alexwengg's picture
Add laya_multilingual_fp16_L256_options32
e9ef51f verified
Raw History Blame Contribute Delete
384 kB
program(1.0)
[buildInfo = dict<tensor<string, []>, tensor<string, []>>({{"coremlc-component-MIL", "3600.16.1"}, {"coremlc-version", "3600.25.2"}, {"coremltools-component-torch", "2.7.0"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "9.0"}})]
{
func main<ios17>(tensor<int32, [1, 256]> attention_mask, tensor<int32, [1, 256]> input_ids, tensor<fp32, [1, 32, 256]> marker_map, tensor<fp32, [1, 3]> question_type) {
tensor<fp16, []> var_1093_to_fp16 = const()[name = tensor<string, []>("op_1093_to_fp16"), val = tensor<fp16, []>(0x1p+0)];
tensor<string, []> var_1092_to_fp16_dtype_0 = const()[name = tensor<string, []>("op_1092_to_fp16_dtype_0"), val = tensor<string, []>("fp16")];
tensor<fp16, [1, 256]> attention_mask_to_fp16 = cast(dtype = var_1092_to_fp16_dtype_0, x = attention_mask)[name = tensor<string, []>("cast_63")];
tensor<fp16, [1, 256]> var_1095_cast_fp16 = sub(x = var_1093_to_fp16, y = attention_mask_to_fp16)[name = tensor<string, []>("op_1095_cast_fp16")];
tensor<int32, [4]> var_1100 = const()[name = tensor<string, []>("op_1100"), val = tensor<int32, [4]>([1, 1, 1, 256])];
tensor<fp16, [1, 1, 1, 256]> var_1101_cast_fp16 = reshape(shape = var_1100, x = var_1095_cast_fp16)[name = tensor<string, []>("op_1101_cast_fp16")];
tensor<fp16, []> var_1102_to_fp16 = const()[name = tensor<string, []>("op_1102_to_fp16"), val = tensor<fp16, []>(-0x1.388p+13)];
tensor<fp16, [1, 1, 1, 256]> pad_cast_fp16 = mul(x = var_1101_cast_fp16, y = var_1102_to_fp16)[name = tensor<string, []>("pad_cast_fp16")];
tensor<fp16, [1, 1, 256, 256]> full_mask_to_fp16 = const()[name = tensor<string, []>("full_mask_to_fp16"), val = tensor<fp16, [1, 1, 256, 256]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(64)))];
tensor<fp16, [1, 1, 256, 256]> attention_mask_3_cast_fp16 = add(x = full_mask_to_fp16, y = pad_cast_fp16)[name = tensor<string, []>("attention_mask_3_cast_fp16")];
tensor<fp16, [1, 1, 256, 256]> band_mask_to_fp16 = const()[name = tensor<string, []>("band_mask_to_fp16"), val = tensor<fp16, [1, 1, 256, 256]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(131200)))];
tensor<fp16, [1, 1, 256, 256]> attention_mask_cast_fp16 = add(x = band_mask_to_fp16, y = pad_cast_fp16)[name = tensor<string, []>("attention_mask_cast_fp16")];
tensor<int32, []> input_3_batch_dims_0 = const()[name = tensor<string, []>("input_3_batch_dims_0"), val = tensor<int32, []>(0)];
tensor<bool, []> input_3_validate_indices_0 = const()[name = tensor<string, []>("input_3_validate_indices_0"), val = tensor<bool, []>(false)];
tensor<fp16, [256000, 768]> model_encoder_embeddings_tok_embeddings_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_embeddings_tok_embeddings_weight_to_fp16"), val = tensor<fp16, [256000, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(262336)))];
tensor<int32, []> greater_equal_0_y_0 = const()[name = tensor<string, []>("greater_equal_0_y_0"), val = tensor<int32, []>(0)];
tensor<bool, [1, 256]> greater_equal_0 = greater_equal(x = input_ids, y = greater_equal_0_y_0)[name = tensor<string, []>("greater_equal_0")];
tensor<int32, []> slice_by_index_0 = const()[name = tensor<string, []>("slice_by_index_0"), val = tensor<int32, []>(256000)];
tensor<int32, [1, 256]> add_0 = add(x = input_ids, y = slice_by_index_0)[name = tensor<string, []>("add_0")];
tensor<int32, [1, 256]> select_0 = select(a = input_ids, b = add_0, cond = greater_equal_0)[name = tensor<string, []>("select_0")];
tensor<int32, []> input_3_cast_fp16_axis_0 = const()[name = tensor<string, []>("input_3_cast_fp16_axis_0"), val = tensor<int32, []>(0)];
tensor<fp16, [1, 256, 768]> input_3_cast_fp16 = gather(axis = input_3_cast_fp16_axis_0, batch_dims = input_3_batch_dims_0, indices = select_0, validate_indices = input_3_validate_indices_0, x = model_encoder_embeddings_tok_embeddings_weight_to_fp16)[name = tensor<string, []>("input_3_cast_fp16")];
tensor<int32, [1]> input_5_axes_0 = const()[name = tensor<string, []>("input_5_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_embeddings_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_embeddings_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(393478400)))];
tensor<fp16, []> var_1116_to_fp16 = const()[name = tensor<string, []>("op_1116_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> input_5_cast_fp16 = layer_norm(axes = input_5_axes_0, epsilon = var_1116_to_fp16, gamma = model_encoder_embeddings_norm_weight_to_fp16, x = input_3_cast_fp16)[name = tensor<string, []>("input_5_cast_fp16")];
tensor<int32, []> var_1131 = const()[name = tensor<string, []>("op_1131"), val = tensor<int32, []>(-1)];
tensor<fp16, [2304, 768]> model_encoder_layers_0_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_0_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(393480000)))];
tensor<fp16, [2304]> linear_0_bias_0_to_fp16 = const()[name = tensor<string, []>("linear_0_bias_0_to_fp16"), val = tensor<fp16, [2304]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(397019008)))];
tensor<fp16, [1, 256, 2304]> linear_0_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_0_attn_Wqkv_weight_to_fp16, x = input_5_cast_fp16)[name = tensor<string, []>("linear_0_cast_fp16")];
tensor<int32, [5]> var_1144 = const()[name = tensor<string, []>("op_1144"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_3_cast_fp16 = reshape(shape = var_1144, x = linear_0_cast_fp16)[name = tensor<string, []>("qkv_3_cast_fp16")];
tensor<int32, [3]> var_1146_split_sizes_0 = const()[name = tensor<string, []>("op_1146_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_1146_axis_0 = const()[name = tensor<string, []>("op_1146_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_1146_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_1146_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_1146_cast_fp16_2 = split(axis = var_1146_axis_0, split_sizes = var_1146_split_sizes_0, x = qkv_3_cast_fp16)[name = tensor<string, []>("op_1146_cast_fp16")];
tensor<int32, [1]> squeeze_0_axes_0 = const()[name = tensor<string, []>("squeeze_0_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_0_cast_fp16 = squeeze(axes = squeeze_0_axes_0, x = var_1146_cast_fp16_0)[name = tensor<string, []>("squeeze_0_cast_fp16")];
tensor<int32, [1]> squeeze_1_axes_0 = const()[name = tensor<string, []>("squeeze_1_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_1_cast_fp16 = squeeze(axes = squeeze_1_axes_0, x = var_1146_cast_fp16_1)[name = tensor<string, []>("squeeze_1_cast_fp16")];
tensor<int32, [1]> squeeze_2_axes_0 = const()[name = tensor<string, []>("squeeze_2_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_2_cast_fp16 = squeeze(axes = squeeze_2_axes_0, x = var_1146_cast_fp16_2)[name = tensor<string, []>("squeeze_2_cast_fp16")];
tensor<int32, [4]> q_1_perm_0 = const()[name = tensor<string, []>("q_1_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_1_perm_0 = const()[name = tensor<string, []>("k_1_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_1_perm_0 = const()[name = tensor<string, []>("value_1_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 1, 256, 64]> cos_3_to_fp16 = const()[name = tensor<string, []>("cos_3_to_fp16"), val = tensor<fp16, [1, 1, 256, 64]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(397023680)))];
tensor<fp16, [1, 12, 256, 64]> q_1_cast_fp16 = transpose(perm = q_1_perm_0, x = squeeze_0_cast_fp16)[name = tensor<string, []>("transpose_106")];
tensor<fp16, [1, 12, 256, 64]> var_1156_cast_fp16 = mul(x = q_1_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1156_cast_fp16")];
tensor<int32, [4]> x1_1_begin_0 = const()[name = tensor<string, []>("x1_1_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_1_end_0 = const()[name = tensor<string, []>("x1_1_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_1_end_mask_0 = const()[name = tensor<string, []>("x1_1_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_1_cast_fp16 = slice_by_index(begin = x1_1_begin_0, end = x1_1_end_0, end_mask = x1_1_end_mask_0, x = q_1_cast_fp16)[name = tensor<string, []>("x1_1_cast_fp16")];
tensor<int32, [4]> x2_1_begin_0 = const()[name = tensor<string, []>("x2_1_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_1_end_0 = const()[name = tensor<string, []>("x2_1_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_1_end_mask_0 = const()[name = tensor<string, []>("x2_1_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_1_cast_fp16 = slice_by_index(begin = x2_1_begin_0, end = x2_1_end_0, end_mask = x2_1_end_mask_0, x = q_1_cast_fp16)[name = tensor<string, []>("x2_1_cast_fp16")];
tensor<fp16, []> const_4_promoted_to_fp16 = const()[name = tensor<string, []>("const_4_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1168_cast_fp16 = mul(x = x2_1_cast_fp16, y = const_4_promoted_to_fp16)[name = tensor<string, []>("op_1168_cast_fp16")];
tensor<bool, []> var_1170_interleave_0 = const()[name = tensor<string, []>("op_1170_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1170_cast_fp16 = concat(axis = var_1131, interleave = var_1170_interleave_0, values = (var_1168_cast_fp16, x1_1_cast_fp16))[name = tensor<string, []>("op_1170_cast_fp16")];
tensor<fp16, [1, 1, 256, 64]> sin_3_to_fp16 = const()[name = tensor<string, []>("sin_3_to_fp16"), val = tensor<fp16, [1, 1, 256, 64]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(397056512)))];
tensor<fp16, [1, 12, 256, 64]> var_1171_cast_fp16 = mul(x = var_1170_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1171_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_1_cast_fp16 = add(x = var_1156_cast_fp16, y = var_1171_cast_fp16)[name = tensor<string, []>("q_embed_1_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_1_cast_fp16 = transpose(perm = k_1_perm_0, x = squeeze_1_cast_fp16)[name = tensor<string, []>("transpose_105")];
tensor<fp16, [1, 12, 256, 64]> var_1174_cast_fp16 = mul(x = k_1_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1174_cast_fp16")];
tensor<int32, [4]> x1_3_begin_0 = const()[name = tensor<string, []>("x1_3_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_3_end_0 = const()[name = tensor<string, []>("x1_3_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_3_end_mask_0 = const()[name = tensor<string, []>("x1_3_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_3_cast_fp16 = slice_by_index(begin = x1_3_begin_0, end = x1_3_end_0, end_mask = x1_3_end_mask_0, x = k_1_cast_fp16)[name = tensor<string, []>("x1_3_cast_fp16")];
tensor<int32, [4]> x2_3_begin_0 = const()[name = tensor<string, []>("x2_3_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_3_end_0 = const()[name = tensor<string, []>("x2_3_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_3_end_mask_0 = const()[name = tensor<string, []>("x2_3_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_3_cast_fp16 = slice_by_index(begin = x2_3_begin_0, end = x2_3_end_0, end_mask = x2_3_end_mask_0, x = k_1_cast_fp16)[name = tensor<string, []>("x2_3_cast_fp16")];
tensor<fp16, []> const_7_promoted_to_fp16 = const()[name = tensor<string, []>("const_7_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1186_cast_fp16 = mul(x = x2_3_cast_fp16, y = const_7_promoted_to_fp16)[name = tensor<string, []>("op_1186_cast_fp16")];
tensor<bool, []> var_1188_interleave_0 = const()[name = tensor<string, []>("op_1188_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1188_cast_fp16 = concat(axis = var_1131, interleave = var_1188_interleave_0, values = (var_1186_cast_fp16, x1_3_cast_fp16))[name = tensor<string, []>("op_1188_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1189_cast_fp16 = mul(x = var_1188_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1189_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_1_cast_fp16 = add(x = var_1174_cast_fp16, y = var_1189_cast_fp16)[name = tensor<string, []>("k_embed_1_cast_fp16")];
tensor<bool, []> var_1194_transpose_x_1 = const()[name = tensor<string, []>("op_1194_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_1194_transpose_y_1 = const()[name = tensor<string, []>("op_1194_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_1194_cast_fp16 = matmul(transpose_x = var_1194_transpose_x_1, transpose_y = var_1194_transpose_y_1, x = q_embed_1_cast_fp16, y = k_embed_1_cast_fp16)[name = tensor<string, []>("op_1194_cast_fp16")];
tensor<fp16, []> var_1195_to_fp16 = const()[name = tensor<string, []>("op_1195_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_1_cast_fp16 = mul(x = var_1194_cast_fp16, y = var_1195_to_fp16)[name = tensor<string, []>("attn_weights_1_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_7_cast_fp16 = add(x = attn_weights_1_cast_fp16, y = attention_mask_3_cast_fp16)[name = tensor<string, []>("input_7_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_1198_cast_fp16 = softmax(axis = var_1131, x = input_7_cast_fp16)[name = tensor<string, []>("op_1198_cast_fp16")];
tensor<bool, []> attn_output_1_transpose_x_0 = const()[name = tensor<string, []>("attn_output_1_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_1_transpose_y_0 = const()[name = tensor<string, []>("attn_output_1_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_1_cast_fp16 = transpose(perm = value_1_perm_0, x = squeeze_2_cast_fp16)[name = tensor<string, []>("transpose_104")];
tensor<fp16, [1, 12, 256, 64]> attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = var_1198_cast_fp16, y = value_1_cast_fp16)[name = tensor<string, []>("attn_output_1_cast_fp16")];
tensor<int32, [4]> var_1202_perm_0 = const()[name = tensor<string, []>("op_1202_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_1204 = const()[name = tensor<string, []>("op_1204"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_1202_cast_fp16 = transpose(perm = var_1202_perm_0, x = attn_output_1_cast_fp16)[name = tensor<string, []>("transpose_103")];
tensor<fp16, [1, 256, 768]> var_1205_cast_fp16 = reshape(shape = var_1204, x = var_1202_cast_fp16)[name = tensor<string, []>("op_1205_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_0_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_0_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(397089344)))];
tensor<fp16, [768]> linear_1_bias_0_to_fp16 = const()[name = tensor<string, []>("linear_1_bias_0_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(398269056)))];
tensor<fp16, [1, 256, 768]> linear_1_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_0_attn_Wo_weight_to_fp16, x = var_1205_cast_fp16)[name = tensor<string, []>("linear_1_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_13_cast_fp16 = add(x = input_5_cast_fp16, y = linear_1_cast_fp16)[name = tensor<string, []>("input_13_cast_fp16")];
tensor<int32, [1]> input_15_axes_0 = const()[name = tensor<string, []>("input_15_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_0_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_0_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(398270656)))];
tensor<fp16, []> var_1126_to_fp16 = const()[name = tensor<string, []>("op_1126_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> input_15_cast_fp16 = layer_norm(axes = input_15_axes_0, epsilon = var_1126_to_fp16, gamma = model_encoder_layers_0_mlp_norm_weight_to_fp16, x = input_13_cast_fp16)[name = tensor<string, []>("input_15_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_0_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_0_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(398272256)))];
tensor<fp16, [1, 256, 2304]> linear_2_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_0_mlp_Wi_weight_to_fp16, x = input_15_cast_fp16)[name = tensor<string, []>("linear_2_cast_fp16")];
tensor<int32, [2]> var_1212_split_sizes_0 = const()[name = tensor<string, []>("op_1212_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_1212_axis_0 = const()[name = tensor<string, []>("op_1212_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_1212_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_1212_cast_fp16_1 = split(axis = var_1212_axis_0, split_sizes = var_1212_split_sizes_0, x = linear_2_cast_fp16)[name = tensor<string, []>("op_1212_cast_fp16")];
tensor<string, []> var_1214_mode_0 = const()[name = tensor<string, []>("op_1214_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_1214_cast_fp16 = gelu(mode = var_1214_mode_0, x = var_1212_cast_fp16_0)[name = tensor<string, []>("op_1214_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_19_cast_fp16 = mul(x = var_1214_cast_fp16, y = var_1212_cast_fp16_1)[name = tensor<string, []>("input_19_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_0_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_0_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(401811264)))];
tensor<fp16, [1, 256, 768]> linear_3_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_0_mlp_Wo_weight_to_fp16, x = input_19_cast_fp16)[name = tensor<string, []>("linear_3_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_23_cast_fp16 = add(x = input_13_cast_fp16, y = linear_3_cast_fp16)[name = tensor<string, []>("input_23_cast_fp16")];
tensor<int32, []> var_1223 = const()[name = tensor<string, []>("op_1223"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_3_axes_0 = const()[name = tensor<string, []>("hidden_states_3_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_1_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_1_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(403580800)))];
tensor<fp16, []> var_1234_to_fp16 = const()[name = tensor<string, []>("op_1234_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_3_cast_fp16 = layer_norm(axes = hidden_states_3_axes_0, epsilon = var_1234_to_fp16, gamma = model_encoder_layers_1_attn_norm_weight_to_fp16, x = input_23_cast_fp16)[name = tensor<string, []>("hidden_states_3_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_1_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_1_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(403582400)))];
tensor<fp16, [1, 256, 2304]> linear_4_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_1_attn_Wqkv_weight_to_fp16, x = hidden_states_3_cast_fp16)[name = tensor<string, []>("linear_4_cast_fp16")];
tensor<int32, [5]> var_1241 = const()[name = tensor<string, []>("op_1241"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_7_cast_fp16 = reshape(shape = var_1241, x = linear_4_cast_fp16)[name = tensor<string, []>("qkv_7_cast_fp16")];
tensor<int32, [3]> var_1243_split_sizes_0 = const()[name = tensor<string, []>("op_1243_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_1243_axis_0 = const()[name = tensor<string, []>("op_1243_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_1243_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_1243_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_1243_cast_fp16_2 = split(axis = var_1243_axis_0, split_sizes = var_1243_split_sizes_0, x = qkv_7_cast_fp16)[name = tensor<string, []>("op_1243_cast_fp16")];
tensor<int32, [1]> squeeze_3_axes_0 = const()[name = tensor<string, []>("squeeze_3_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_3_cast_fp16 = squeeze(axes = squeeze_3_axes_0, x = var_1243_cast_fp16_0)[name = tensor<string, []>("squeeze_3_cast_fp16")];
tensor<int32, [1]> squeeze_4_axes_0 = const()[name = tensor<string, []>("squeeze_4_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_4_cast_fp16 = squeeze(axes = squeeze_4_axes_0, x = var_1243_cast_fp16_1)[name = tensor<string, []>("squeeze_4_cast_fp16")];
tensor<int32, [1]> squeeze_5_axes_0 = const()[name = tensor<string, []>("squeeze_5_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_5_cast_fp16 = squeeze(axes = squeeze_5_axes_0, x = var_1243_cast_fp16_2)[name = tensor<string, []>("squeeze_5_cast_fp16")];
tensor<int32, [4]> q_5_perm_0 = const()[name = tensor<string, []>("q_5_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_5_perm_0 = const()[name = tensor<string, []>("k_5_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_3_perm_0 = const()[name = tensor<string, []>("value_3_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_5_cast_fp16 = transpose(perm = q_5_perm_0, x = squeeze_3_cast_fp16)[name = tensor<string, []>("transpose_102")];
tensor<fp16, [1, 12, 256, 64]> var_1253_cast_fp16 = mul(x = q_5_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1253_cast_fp16")];
tensor<int32, [4]> x1_5_begin_0 = const()[name = tensor<string, []>("x1_5_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_5_end_0 = const()[name = tensor<string, []>("x1_5_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_5_end_mask_0 = const()[name = tensor<string, []>("x1_5_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_5_cast_fp16 = slice_by_index(begin = x1_5_begin_0, end = x1_5_end_0, end_mask = x1_5_end_mask_0, x = q_5_cast_fp16)[name = tensor<string, []>("x1_5_cast_fp16")];
tensor<int32, [4]> x2_5_begin_0 = const()[name = tensor<string, []>("x2_5_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_5_end_0 = const()[name = tensor<string, []>("x2_5_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_5_end_mask_0 = const()[name = tensor<string, []>("x2_5_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_5_cast_fp16 = slice_by_index(begin = x2_5_begin_0, end = x2_5_end_0, end_mask = x2_5_end_mask_0, x = q_5_cast_fp16)[name = tensor<string, []>("x2_5_cast_fp16")];
tensor<fp16, []> const_12_promoted_to_fp16 = const()[name = tensor<string, []>("const_12_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1265_cast_fp16 = mul(x = x2_5_cast_fp16, y = const_12_promoted_to_fp16)[name = tensor<string, []>("op_1265_cast_fp16")];
tensor<bool, []> var_1267_interleave_0 = const()[name = tensor<string, []>("op_1267_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1267_cast_fp16 = concat(axis = var_1223, interleave = var_1267_interleave_0, values = (var_1265_cast_fp16, x1_5_cast_fp16))[name = tensor<string, []>("op_1267_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1268_cast_fp16 = mul(x = var_1267_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1268_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_3_cast_fp16 = add(x = var_1253_cast_fp16, y = var_1268_cast_fp16)[name = tensor<string, []>("q_embed_3_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_5_cast_fp16 = transpose(perm = k_5_perm_0, x = squeeze_4_cast_fp16)[name = tensor<string, []>("transpose_101")];
tensor<fp16, [1, 12, 256, 64]> var_1271_cast_fp16 = mul(x = k_5_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1271_cast_fp16")];
tensor<int32, [4]> x1_7_begin_0 = const()[name = tensor<string, []>("x1_7_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_7_end_0 = const()[name = tensor<string, []>("x1_7_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_7_end_mask_0 = const()[name = tensor<string, []>("x1_7_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_7_cast_fp16 = slice_by_index(begin = x1_7_begin_0, end = x1_7_end_0, end_mask = x1_7_end_mask_0, x = k_5_cast_fp16)[name = tensor<string, []>("x1_7_cast_fp16")];
tensor<int32, [4]> x2_7_begin_0 = const()[name = tensor<string, []>("x2_7_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_7_end_0 = const()[name = tensor<string, []>("x2_7_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_7_end_mask_0 = const()[name = tensor<string, []>("x2_7_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_7_cast_fp16 = slice_by_index(begin = x2_7_begin_0, end = x2_7_end_0, end_mask = x2_7_end_mask_0, x = k_5_cast_fp16)[name = tensor<string, []>("x2_7_cast_fp16")];
tensor<fp16, []> const_15_promoted_to_fp16 = const()[name = tensor<string, []>("const_15_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1283_cast_fp16 = mul(x = x2_7_cast_fp16, y = const_15_promoted_to_fp16)[name = tensor<string, []>("op_1283_cast_fp16")];
tensor<bool, []> var_1285_interleave_0 = const()[name = tensor<string, []>("op_1285_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1285_cast_fp16 = concat(axis = var_1223, interleave = var_1285_interleave_0, values = (var_1283_cast_fp16, x1_7_cast_fp16))[name = tensor<string, []>("op_1285_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1286_cast_fp16 = mul(x = var_1285_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1286_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_3_cast_fp16 = add(x = var_1271_cast_fp16, y = var_1286_cast_fp16)[name = tensor<string, []>("k_embed_3_cast_fp16")];
tensor<bool, []> var_1291_transpose_x_1 = const()[name = tensor<string, []>("op_1291_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_1291_transpose_y_1 = const()[name = tensor<string, []>("op_1291_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_1291_cast_fp16 = matmul(transpose_x = var_1291_transpose_x_1, transpose_y = var_1291_transpose_y_1, x = q_embed_3_cast_fp16, y = k_embed_3_cast_fp16)[name = tensor<string, []>("op_1291_cast_fp16")];
tensor<fp16, []> var_1292_to_fp16 = const()[name = tensor<string, []>("op_1292_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_5_cast_fp16 = mul(x = var_1291_cast_fp16, y = var_1292_to_fp16)[name = tensor<string, []>("attn_weights_5_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_25_cast_fp16 = add(x = attn_weights_5_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_25_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_1295_cast_fp16 = softmax(axis = var_1223, x = input_25_cast_fp16)[name = tensor<string, []>("op_1295_cast_fp16")];
tensor<bool, []> attn_output_7_transpose_x_0 = const()[name = tensor<string, []>("attn_output_7_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_7_transpose_y_0 = const()[name = tensor<string, []>("attn_output_7_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_3_cast_fp16 = transpose(perm = value_3_perm_0, x = squeeze_5_cast_fp16)[name = tensor<string, []>("transpose_100")];
tensor<fp16, [1, 12, 256, 64]> attn_output_7_cast_fp16 = matmul(transpose_x = attn_output_7_transpose_x_0, transpose_y = attn_output_7_transpose_y_0, x = var_1295_cast_fp16, y = value_3_cast_fp16)[name = tensor<string, []>("attn_output_7_cast_fp16")];
tensor<int32, [4]> var_1299_perm_0 = const()[name = tensor<string, []>("op_1299_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_1301 = const()[name = tensor<string, []>("op_1301"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_1299_cast_fp16 = transpose(perm = var_1299_perm_0, x = attn_output_7_cast_fp16)[name = tensor<string, []>("transpose_99")];
tensor<fp16, [1, 256, 768]> var_1302_cast_fp16 = reshape(shape = var_1301, x = var_1299_cast_fp16)[name = tensor<string, []>("op_1302_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_1_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_1_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(407121408)))];
tensor<fp16, [1, 256, 768]> linear_5_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_1_attn_Wo_weight_to_fp16, x = var_1302_cast_fp16)[name = tensor<string, []>("linear_5_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_31_cast_fp16 = add(x = input_23_cast_fp16, y = linear_5_cast_fp16)[name = tensor<string, []>("input_31_cast_fp16")];
tensor<int32, [1]> input_33_axes_0 = const()[name = tensor<string, []>("input_33_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_1_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_1_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(408301120)))];
tensor<fp16, [1, 256, 768]> input_33_cast_fp16 = layer_norm(axes = input_33_axes_0, epsilon = var_1234_to_fp16, gamma = model_encoder_layers_1_mlp_norm_weight_to_fp16, x = input_31_cast_fp16)[name = tensor<string, []>("input_33_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_1_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_1_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(408302720)))];
tensor<fp16, [1, 256, 2304]> linear_6_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_1_mlp_Wi_weight_to_fp16, x = input_33_cast_fp16)[name = tensor<string, []>("linear_6_cast_fp16")];
tensor<int32, [2]> var_1309_split_sizes_0 = const()[name = tensor<string, []>("op_1309_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_1309_axis_0 = const()[name = tensor<string, []>("op_1309_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_1309_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_1309_cast_fp16_1 = split(axis = var_1309_axis_0, split_sizes = var_1309_split_sizes_0, x = linear_6_cast_fp16)[name = tensor<string, []>("op_1309_cast_fp16")];
tensor<string, []> var_1311_mode_0 = const()[name = tensor<string, []>("op_1311_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_1311_cast_fp16 = gelu(mode = var_1311_mode_0, x = var_1309_cast_fp16_0)[name = tensor<string, []>("op_1311_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_37_cast_fp16 = mul(x = var_1311_cast_fp16, y = var_1309_cast_fp16_1)[name = tensor<string, []>("input_37_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_1_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_1_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(411841728)))];
tensor<fp16, [1, 256, 768]> linear_7_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_1_mlp_Wo_weight_to_fp16, x = input_37_cast_fp16)[name = tensor<string, []>("linear_7_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_41_cast_fp16 = add(x = input_31_cast_fp16, y = linear_7_cast_fp16)[name = tensor<string, []>("input_41_cast_fp16")];
tensor<int32, []> var_1320 = const()[name = tensor<string, []>("op_1320"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_5_axes_0 = const()[name = tensor<string, []>("hidden_states_5_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_2_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_2_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(413611264)))];
tensor<fp16, []> var_1331_to_fp16 = const()[name = tensor<string, []>("op_1331_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_5_cast_fp16 = layer_norm(axes = hidden_states_5_axes_0, epsilon = var_1331_to_fp16, gamma = model_encoder_layers_2_attn_norm_weight_to_fp16, x = input_41_cast_fp16)[name = tensor<string, []>("hidden_states_5_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_2_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_2_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(413612864)))];
tensor<fp16, [1, 256, 2304]> linear_8_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_2_attn_Wqkv_weight_to_fp16, x = hidden_states_5_cast_fp16)[name = tensor<string, []>("linear_8_cast_fp16")];
tensor<int32, [5]> var_1338 = const()[name = tensor<string, []>("op_1338"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_11_cast_fp16 = reshape(shape = var_1338, x = linear_8_cast_fp16)[name = tensor<string, []>("qkv_11_cast_fp16")];
tensor<int32, [3]> var_1340_split_sizes_0 = const()[name = tensor<string, []>("op_1340_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_1340_axis_0 = const()[name = tensor<string, []>("op_1340_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_1340_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_1340_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_1340_cast_fp16_2 = split(axis = var_1340_axis_0, split_sizes = var_1340_split_sizes_0, x = qkv_11_cast_fp16)[name = tensor<string, []>("op_1340_cast_fp16")];
tensor<int32, [1]> squeeze_6_axes_0 = const()[name = tensor<string, []>("squeeze_6_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_6_cast_fp16 = squeeze(axes = squeeze_6_axes_0, x = var_1340_cast_fp16_0)[name = tensor<string, []>("squeeze_6_cast_fp16")];
tensor<int32, [1]> squeeze_7_axes_0 = const()[name = tensor<string, []>("squeeze_7_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_7_cast_fp16 = squeeze(axes = squeeze_7_axes_0, x = var_1340_cast_fp16_1)[name = tensor<string, []>("squeeze_7_cast_fp16")];
tensor<int32, [1]> squeeze_8_axes_0 = const()[name = tensor<string, []>("squeeze_8_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_8_cast_fp16 = squeeze(axes = squeeze_8_axes_0, x = var_1340_cast_fp16_2)[name = tensor<string, []>("squeeze_8_cast_fp16")];
tensor<int32, [4]> q_9_perm_0 = const()[name = tensor<string, []>("q_9_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_9_perm_0 = const()[name = tensor<string, []>("k_9_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_5_perm_0 = const()[name = tensor<string, []>("value_5_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_9_cast_fp16 = transpose(perm = q_9_perm_0, x = squeeze_6_cast_fp16)[name = tensor<string, []>("transpose_98")];
tensor<fp16, [1, 12, 256, 64]> var_1350_cast_fp16 = mul(x = q_9_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1350_cast_fp16")];
tensor<int32, [4]> x1_9_begin_0 = const()[name = tensor<string, []>("x1_9_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_9_end_0 = const()[name = tensor<string, []>("x1_9_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_9_end_mask_0 = const()[name = tensor<string, []>("x1_9_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_9_cast_fp16 = slice_by_index(begin = x1_9_begin_0, end = x1_9_end_0, end_mask = x1_9_end_mask_0, x = q_9_cast_fp16)[name = tensor<string, []>("x1_9_cast_fp16")];
tensor<int32, [4]> x2_9_begin_0 = const()[name = tensor<string, []>("x2_9_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_9_end_0 = const()[name = tensor<string, []>("x2_9_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_9_end_mask_0 = const()[name = tensor<string, []>("x2_9_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_9_cast_fp16 = slice_by_index(begin = x2_9_begin_0, end = x2_9_end_0, end_mask = x2_9_end_mask_0, x = q_9_cast_fp16)[name = tensor<string, []>("x2_9_cast_fp16")];
tensor<fp16, []> const_20_promoted_to_fp16 = const()[name = tensor<string, []>("const_20_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1362_cast_fp16 = mul(x = x2_9_cast_fp16, y = const_20_promoted_to_fp16)[name = tensor<string, []>("op_1362_cast_fp16")];
tensor<bool, []> var_1364_interleave_0 = const()[name = tensor<string, []>("op_1364_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1364_cast_fp16 = concat(axis = var_1320, interleave = var_1364_interleave_0, values = (var_1362_cast_fp16, x1_9_cast_fp16))[name = tensor<string, []>("op_1364_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1365_cast_fp16 = mul(x = var_1364_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1365_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_5_cast_fp16 = add(x = var_1350_cast_fp16, y = var_1365_cast_fp16)[name = tensor<string, []>("q_embed_5_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_9_cast_fp16 = transpose(perm = k_9_perm_0, x = squeeze_7_cast_fp16)[name = tensor<string, []>("transpose_97")];
tensor<fp16, [1, 12, 256, 64]> var_1368_cast_fp16 = mul(x = k_9_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1368_cast_fp16")];
tensor<int32, [4]> x1_11_begin_0 = const()[name = tensor<string, []>("x1_11_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_11_end_0 = const()[name = tensor<string, []>("x1_11_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_11_end_mask_0 = const()[name = tensor<string, []>("x1_11_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_11_cast_fp16 = slice_by_index(begin = x1_11_begin_0, end = x1_11_end_0, end_mask = x1_11_end_mask_0, x = k_9_cast_fp16)[name = tensor<string, []>("x1_11_cast_fp16")];
tensor<int32, [4]> x2_11_begin_0 = const()[name = tensor<string, []>("x2_11_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_11_end_0 = const()[name = tensor<string, []>("x2_11_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_11_end_mask_0 = const()[name = tensor<string, []>("x2_11_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_11_cast_fp16 = slice_by_index(begin = x2_11_begin_0, end = x2_11_end_0, end_mask = x2_11_end_mask_0, x = k_9_cast_fp16)[name = tensor<string, []>("x2_11_cast_fp16")];
tensor<fp16, []> const_23_promoted_to_fp16 = const()[name = tensor<string, []>("const_23_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1380_cast_fp16 = mul(x = x2_11_cast_fp16, y = const_23_promoted_to_fp16)[name = tensor<string, []>("op_1380_cast_fp16")];
tensor<bool, []> var_1382_interleave_0 = const()[name = tensor<string, []>("op_1382_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1382_cast_fp16 = concat(axis = var_1320, interleave = var_1382_interleave_0, values = (var_1380_cast_fp16, x1_11_cast_fp16))[name = tensor<string, []>("op_1382_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1383_cast_fp16 = mul(x = var_1382_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1383_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_5_cast_fp16 = add(x = var_1368_cast_fp16, y = var_1383_cast_fp16)[name = tensor<string, []>("k_embed_5_cast_fp16")];
tensor<bool, []> var_1388_transpose_x_1 = const()[name = tensor<string, []>("op_1388_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_1388_transpose_y_1 = const()[name = tensor<string, []>("op_1388_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_1388_cast_fp16 = matmul(transpose_x = var_1388_transpose_x_1, transpose_y = var_1388_transpose_y_1, x = q_embed_5_cast_fp16, y = k_embed_5_cast_fp16)[name = tensor<string, []>("op_1388_cast_fp16")];
tensor<fp16, []> var_1389_to_fp16 = const()[name = tensor<string, []>("op_1389_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_9_cast_fp16 = mul(x = var_1388_cast_fp16, y = var_1389_to_fp16)[name = tensor<string, []>("attn_weights_9_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_43_cast_fp16 = add(x = attn_weights_9_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_43_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_1392_cast_fp16 = softmax(axis = var_1320, x = input_43_cast_fp16)[name = tensor<string, []>("op_1392_cast_fp16")];
tensor<bool, []> attn_output_13_transpose_x_0 = const()[name = tensor<string, []>("attn_output_13_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_13_transpose_y_0 = const()[name = tensor<string, []>("attn_output_13_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_5_cast_fp16 = transpose(perm = value_5_perm_0, x = squeeze_8_cast_fp16)[name = tensor<string, []>("transpose_96")];
tensor<fp16, [1, 12, 256, 64]> attn_output_13_cast_fp16 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = var_1392_cast_fp16, y = value_5_cast_fp16)[name = tensor<string, []>("attn_output_13_cast_fp16")];
tensor<int32, [4]> var_1396_perm_0 = const()[name = tensor<string, []>("op_1396_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_1398 = const()[name = tensor<string, []>("op_1398"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_1396_cast_fp16 = transpose(perm = var_1396_perm_0, x = attn_output_13_cast_fp16)[name = tensor<string, []>("transpose_95")];
tensor<fp16, [1, 256, 768]> var_1399_cast_fp16 = reshape(shape = var_1398, x = var_1396_cast_fp16)[name = tensor<string, []>("op_1399_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_2_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_2_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(417151872)))];
tensor<fp16, [1, 256, 768]> linear_9_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_2_attn_Wo_weight_to_fp16, x = var_1399_cast_fp16)[name = tensor<string, []>("linear_9_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_49_cast_fp16 = add(x = input_41_cast_fp16, y = linear_9_cast_fp16)[name = tensor<string, []>("input_49_cast_fp16")];
tensor<int32, [1]> input_51_axes_0 = const()[name = tensor<string, []>("input_51_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_2_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_2_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(418331584)))];
tensor<fp16, [1, 256, 768]> input_51_cast_fp16 = layer_norm(axes = input_51_axes_0, epsilon = var_1331_to_fp16, gamma = model_encoder_layers_2_mlp_norm_weight_to_fp16, x = input_49_cast_fp16)[name = tensor<string, []>("input_51_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_2_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_2_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(418333184)))];
tensor<fp16, [1, 256, 2304]> linear_10_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_2_mlp_Wi_weight_to_fp16, x = input_51_cast_fp16)[name = tensor<string, []>("linear_10_cast_fp16")];
tensor<int32, [2]> var_1406_split_sizes_0 = const()[name = tensor<string, []>("op_1406_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_1406_axis_0 = const()[name = tensor<string, []>("op_1406_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_1406_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_1406_cast_fp16_1 = split(axis = var_1406_axis_0, split_sizes = var_1406_split_sizes_0, x = linear_10_cast_fp16)[name = tensor<string, []>("op_1406_cast_fp16")];
tensor<string, []> var_1408_mode_0 = const()[name = tensor<string, []>("op_1408_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_1408_cast_fp16 = gelu(mode = var_1408_mode_0, x = var_1406_cast_fp16_0)[name = tensor<string, []>("op_1408_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_55_cast_fp16 = mul(x = var_1408_cast_fp16, y = var_1406_cast_fp16_1)[name = tensor<string, []>("input_55_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_2_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_2_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(421872192)))];
tensor<fp16, [1, 256, 768]> linear_11_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_2_mlp_Wo_weight_to_fp16, x = input_55_cast_fp16)[name = tensor<string, []>("linear_11_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_59_cast_fp16 = add(x = input_49_cast_fp16, y = linear_11_cast_fp16)[name = tensor<string, []>("input_59_cast_fp16")];
tensor<int32, []> var_1417 = const()[name = tensor<string, []>("op_1417"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_7_axes_0 = const()[name = tensor<string, []>("hidden_states_7_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_3_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_3_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(423641728)))];
tensor<fp16, []> var_1428_to_fp16 = const()[name = tensor<string, []>("op_1428_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_7_cast_fp16 = layer_norm(axes = hidden_states_7_axes_0, epsilon = var_1428_to_fp16, gamma = model_encoder_layers_3_attn_norm_weight_to_fp16, x = input_59_cast_fp16)[name = tensor<string, []>("hidden_states_7_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_3_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_3_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(423643328)))];
tensor<fp16, [1, 256, 2304]> linear_12_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_3_attn_Wqkv_weight_to_fp16, x = hidden_states_7_cast_fp16)[name = tensor<string, []>("linear_12_cast_fp16")];
tensor<int32, [5]> var_1435 = const()[name = tensor<string, []>("op_1435"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_15_cast_fp16 = reshape(shape = var_1435, x = linear_12_cast_fp16)[name = tensor<string, []>("qkv_15_cast_fp16")];
tensor<int32, [3]> var_1437_split_sizes_0 = const()[name = tensor<string, []>("op_1437_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_1437_axis_0 = const()[name = tensor<string, []>("op_1437_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_1437_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_1437_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_1437_cast_fp16_2 = split(axis = var_1437_axis_0, split_sizes = var_1437_split_sizes_0, x = qkv_15_cast_fp16)[name = tensor<string, []>("op_1437_cast_fp16")];
tensor<int32, [1]> squeeze_9_axes_0 = const()[name = tensor<string, []>("squeeze_9_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_9_cast_fp16 = squeeze(axes = squeeze_9_axes_0, x = var_1437_cast_fp16_0)[name = tensor<string, []>("squeeze_9_cast_fp16")];
tensor<int32, [1]> squeeze_10_axes_0 = const()[name = tensor<string, []>("squeeze_10_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_10_cast_fp16 = squeeze(axes = squeeze_10_axes_0, x = var_1437_cast_fp16_1)[name = tensor<string, []>("squeeze_10_cast_fp16")];
tensor<int32, [1]> squeeze_11_axes_0 = const()[name = tensor<string, []>("squeeze_11_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_11_cast_fp16 = squeeze(axes = squeeze_11_axes_0, x = var_1437_cast_fp16_2)[name = tensor<string, []>("squeeze_11_cast_fp16")];
tensor<int32, [4]> q_13_perm_0 = const()[name = tensor<string, []>("q_13_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_13_perm_0 = const()[name = tensor<string, []>("k_13_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_7_perm_0 = const()[name = tensor<string, []>("value_7_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_13_cast_fp16 = transpose(perm = q_13_perm_0, x = squeeze_9_cast_fp16)[name = tensor<string, []>("transpose_94")];
tensor<fp16, [1, 12, 256, 64]> var_1447_cast_fp16 = mul(x = q_13_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1447_cast_fp16")];
tensor<int32, [4]> x1_13_begin_0 = const()[name = tensor<string, []>("x1_13_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_13_end_0 = const()[name = tensor<string, []>("x1_13_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_13_end_mask_0 = const()[name = tensor<string, []>("x1_13_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_13_cast_fp16 = slice_by_index(begin = x1_13_begin_0, end = x1_13_end_0, end_mask = x1_13_end_mask_0, x = q_13_cast_fp16)[name = tensor<string, []>("x1_13_cast_fp16")];
tensor<int32, [4]> x2_13_begin_0 = const()[name = tensor<string, []>("x2_13_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_13_end_0 = const()[name = tensor<string, []>("x2_13_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_13_end_mask_0 = const()[name = tensor<string, []>("x2_13_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_13_cast_fp16 = slice_by_index(begin = x2_13_begin_0, end = x2_13_end_0, end_mask = x2_13_end_mask_0, x = q_13_cast_fp16)[name = tensor<string, []>("x2_13_cast_fp16")];
tensor<fp16, []> const_28_promoted_to_fp16 = const()[name = tensor<string, []>("const_28_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1459_cast_fp16 = mul(x = x2_13_cast_fp16, y = const_28_promoted_to_fp16)[name = tensor<string, []>("op_1459_cast_fp16")];
tensor<bool, []> var_1461_interleave_0 = const()[name = tensor<string, []>("op_1461_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1461_cast_fp16 = concat(axis = var_1417, interleave = var_1461_interleave_0, values = (var_1459_cast_fp16, x1_13_cast_fp16))[name = tensor<string, []>("op_1461_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1462_cast_fp16 = mul(x = var_1461_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1462_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_7_cast_fp16 = add(x = var_1447_cast_fp16, y = var_1462_cast_fp16)[name = tensor<string, []>("q_embed_7_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_13_cast_fp16 = transpose(perm = k_13_perm_0, x = squeeze_10_cast_fp16)[name = tensor<string, []>("transpose_93")];
tensor<fp16, [1, 12, 256, 64]> var_1465_cast_fp16 = mul(x = k_13_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1465_cast_fp16")];
tensor<int32, [4]> x1_15_begin_0 = const()[name = tensor<string, []>("x1_15_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_15_end_0 = const()[name = tensor<string, []>("x1_15_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_15_end_mask_0 = const()[name = tensor<string, []>("x1_15_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_15_cast_fp16 = slice_by_index(begin = x1_15_begin_0, end = x1_15_end_0, end_mask = x1_15_end_mask_0, x = k_13_cast_fp16)[name = tensor<string, []>("x1_15_cast_fp16")];
tensor<int32, [4]> x2_15_begin_0 = const()[name = tensor<string, []>("x2_15_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_15_end_0 = const()[name = tensor<string, []>("x2_15_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_15_end_mask_0 = const()[name = tensor<string, []>("x2_15_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_15_cast_fp16 = slice_by_index(begin = x2_15_begin_0, end = x2_15_end_0, end_mask = x2_15_end_mask_0, x = k_13_cast_fp16)[name = tensor<string, []>("x2_15_cast_fp16")];
tensor<fp16, []> const_31_promoted_to_fp16 = const()[name = tensor<string, []>("const_31_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1477_cast_fp16 = mul(x = x2_15_cast_fp16, y = const_31_promoted_to_fp16)[name = tensor<string, []>("op_1477_cast_fp16")];
tensor<bool, []> var_1479_interleave_0 = const()[name = tensor<string, []>("op_1479_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1479_cast_fp16 = concat(axis = var_1417, interleave = var_1479_interleave_0, values = (var_1477_cast_fp16, x1_15_cast_fp16))[name = tensor<string, []>("op_1479_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1480_cast_fp16 = mul(x = var_1479_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1480_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_7_cast_fp16 = add(x = var_1465_cast_fp16, y = var_1480_cast_fp16)[name = tensor<string, []>("k_embed_7_cast_fp16")];
tensor<bool, []> var_1485_transpose_x_1 = const()[name = tensor<string, []>("op_1485_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_1485_transpose_y_1 = const()[name = tensor<string, []>("op_1485_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_1485_cast_fp16 = matmul(transpose_x = var_1485_transpose_x_1, transpose_y = var_1485_transpose_y_1, x = q_embed_7_cast_fp16, y = k_embed_7_cast_fp16)[name = tensor<string, []>("op_1485_cast_fp16")];
tensor<fp16, []> var_1486_to_fp16 = const()[name = tensor<string, []>("op_1486_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_13_cast_fp16 = mul(x = var_1485_cast_fp16, y = var_1486_to_fp16)[name = tensor<string, []>("attn_weights_13_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_61_cast_fp16 = add(x = attn_weights_13_cast_fp16, y = attention_mask_3_cast_fp16)[name = tensor<string, []>("input_61_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_1489_cast_fp16 = softmax(axis = var_1417, x = input_61_cast_fp16)[name = tensor<string, []>("op_1489_cast_fp16")];
tensor<bool, []> attn_output_19_transpose_x_0 = const()[name = tensor<string, []>("attn_output_19_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_19_transpose_y_0 = const()[name = tensor<string, []>("attn_output_19_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_7_cast_fp16 = transpose(perm = value_7_perm_0, x = squeeze_11_cast_fp16)[name = tensor<string, []>("transpose_92")];
tensor<fp16, [1, 12, 256, 64]> attn_output_19_cast_fp16 = matmul(transpose_x = attn_output_19_transpose_x_0, transpose_y = attn_output_19_transpose_y_0, x = var_1489_cast_fp16, y = value_7_cast_fp16)[name = tensor<string, []>("attn_output_19_cast_fp16")];
tensor<int32, [4]> var_1493_perm_0 = const()[name = tensor<string, []>("op_1493_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_1495 = const()[name = tensor<string, []>("op_1495"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_1493_cast_fp16 = transpose(perm = var_1493_perm_0, x = attn_output_19_cast_fp16)[name = tensor<string, []>("transpose_91")];
tensor<fp16, [1, 256, 768]> var_1496_cast_fp16 = reshape(shape = var_1495, x = var_1493_cast_fp16)[name = tensor<string, []>("op_1496_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_3_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_3_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(427182336)))];
tensor<fp16, [1, 256, 768]> linear_13_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_3_attn_Wo_weight_to_fp16, x = var_1496_cast_fp16)[name = tensor<string, []>("linear_13_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_67_cast_fp16 = add(x = input_59_cast_fp16, y = linear_13_cast_fp16)[name = tensor<string, []>("input_67_cast_fp16")];
tensor<int32, [1]> input_69_axes_0 = const()[name = tensor<string, []>("input_69_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_3_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_3_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(428362048)))];
tensor<fp16, [1, 256, 768]> input_69_cast_fp16 = layer_norm(axes = input_69_axes_0, epsilon = var_1428_to_fp16, gamma = model_encoder_layers_3_mlp_norm_weight_to_fp16, x = input_67_cast_fp16)[name = tensor<string, []>("input_69_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_3_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_3_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(428363648)))];
tensor<fp16, [1, 256, 2304]> linear_14_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_3_mlp_Wi_weight_to_fp16, x = input_69_cast_fp16)[name = tensor<string, []>("linear_14_cast_fp16")];
tensor<int32, [2]> var_1503_split_sizes_0 = const()[name = tensor<string, []>("op_1503_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_1503_axis_0 = const()[name = tensor<string, []>("op_1503_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_1503_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_1503_cast_fp16_1 = split(axis = var_1503_axis_0, split_sizes = var_1503_split_sizes_0, x = linear_14_cast_fp16)[name = tensor<string, []>("op_1503_cast_fp16")];
tensor<string, []> var_1505_mode_0 = const()[name = tensor<string, []>("op_1505_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_1505_cast_fp16 = gelu(mode = var_1505_mode_0, x = var_1503_cast_fp16_0)[name = tensor<string, []>("op_1505_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_73_cast_fp16 = mul(x = var_1505_cast_fp16, y = var_1503_cast_fp16_1)[name = tensor<string, []>("input_73_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_3_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_3_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(431902656)))];
tensor<fp16, [1, 256, 768]> linear_15_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_3_mlp_Wo_weight_to_fp16, x = input_73_cast_fp16)[name = tensor<string, []>("linear_15_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_77_cast_fp16 = add(x = input_67_cast_fp16, y = linear_15_cast_fp16)[name = tensor<string, []>("input_77_cast_fp16")];
tensor<int32, []> var_1514 = const()[name = tensor<string, []>("op_1514"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_9_axes_0 = const()[name = tensor<string, []>("hidden_states_9_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_4_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_4_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(433672192)))];
tensor<fp16, []> var_1525_to_fp16 = const()[name = tensor<string, []>("op_1525_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_9_cast_fp16 = layer_norm(axes = hidden_states_9_axes_0, epsilon = var_1525_to_fp16, gamma = model_encoder_layers_4_attn_norm_weight_to_fp16, x = input_77_cast_fp16)[name = tensor<string, []>("hidden_states_9_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_4_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_4_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(433673792)))];
tensor<fp16, [1, 256, 2304]> linear_16_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_4_attn_Wqkv_weight_to_fp16, x = hidden_states_9_cast_fp16)[name = tensor<string, []>("linear_16_cast_fp16")];
tensor<int32, [5]> var_1532 = const()[name = tensor<string, []>("op_1532"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_19_cast_fp16 = reshape(shape = var_1532, x = linear_16_cast_fp16)[name = tensor<string, []>("qkv_19_cast_fp16")];
tensor<int32, [3]> var_1534_split_sizes_0 = const()[name = tensor<string, []>("op_1534_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_1534_axis_0 = const()[name = tensor<string, []>("op_1534_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_1534_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_1534_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_1534_cast_fp16_2 = split(axis = var_1534_axis_0, split_sizes = var_1534_split_sizes_0, x = qkv_19_cast_fp16)[name = tensor<string, []>("op_1534_cast_fp16")];
tensor<int32, [1]> squeeze_12_axes_0 = const()[name = tensor<string, []>("squeeze_12_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_12_cast_fp16 = squeeze(axes = squeeze_12_axes_0, x = var_1534_cast_fp16_0)[name = tensor<string, []>("squeeze_12_cast_fp16")];
tensor<int32, [1]> squeeze_13_axes_0 = const()[name = tensor<string, []>("squeeze_13_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_13_cast_fp16 = squeeze(axes = squeeze_13_axes_0, x = var_1534_cast_fp16_1)[name = tensor<string, []>("squeeze_13_cast_fp16")];
tensor<int32, [1]> squeeze_14_axes_0 = const()[name = tensor<string, []>("squeeze_14_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_14_cast_fp16 = squeeze(axes = squeeze_14_axes_0, x = var_1534_cast_fp16_2)[name = tensor<string, []>("squeeze_14_cast_fp16")];
tensor<int32, [4]> q_17_perm_0 = const()[name = tensor<string, []>("q_17_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_17_perm_0 = const()[name = tensor<string, []>("k_17_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_9_perm_0 = const()[name = tensor<string, []>("value_9_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_17_cast_fp16 = transpose(perm = q_17_perm_0, x = squeeze_12_cast_fp16)[name = tensor<string, []>("transpose_90")];
tensor<fp16, [1, 12, 256, 64]> var_1544_cast_fp16 = mul(x = q_17_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1544_cast_fp16")];
tensor<int32, [4]> x1_17_begin_0 = const()[name = tensor<string, []>("x1_17_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_17_end_0 = const()[name = tensor<string, []>("x1_17_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_17_end_mask_0 = const()[name = tensor<string, []>("x1_17_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_17_cast_fp16 = slice_by_index(begin = x1_17_begin_0, end = x1_17_end_0, end_mask = x1_17_end_mask_0, x = q_17_cast_fp16)[name = tensor<string, []>("x1_17_cast_fp16")];
tensor<int32, [4]> x2_17_begin_0 = const()[name = tensor<string, []>("x2_17_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_17_end_0 = const()[name = tensor<string, []>("x2_17_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_17_end_mask_0 = const()[name = tensor<string, []>("x2_17_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_17_cast_fp16 = slice_by_index(begin = x2_17_begin_0, end = x2_17_end_0, end_mask = x2_17_end_mask_0, x = q_17_cast_fp16)[name = tensor<string, []>("x2_17_cast_fp16")];
tensor<fp16, []> const_36_promoted_to_fp16 = const()[name = tensor<string, []>("const_36_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1556_cast_fp16 = mul(x = x2_17_cast_fp16, y = const_36_promoted_to_fp16)[name = tensor<string, []>("op_1556_cast_fp16")];
tensor<bool, []> var_1558_interleave_0 = const()[name = tensor<string, []>("op_1558_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1558_cast_fp16 = concat(axis = var_1514, interleave = var_1558_interleave_0, values = (var_1556_cast_fp16, x1_17_cast_fp16))[name = tensor<string, []>("op_1558_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1559_cast_fp16 = mul(x = var_1558_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1559_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_9_cast_fp16 = add(x = var_1544_cast_fp16, y = var_1559_cast_fp16)[name = tensor<string, []>("q_embed_9_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_17_cast_fp16 = transpose(perm = k_17_perm_0, x = squeeze_13_cast_fp16)[name = tensor<string, []>("transpose_89")];
tensor<fp16, [1, 12, 256, 64]> var_1562_cast_fp16 = mul(x = k_17_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1562_cast_fp16")];
tensor<int32, [4]> x1_19_begin_0 = const()[name = tensor<string, []>("x1_19_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_19_end_0 = const()[name = tensor<string, []>("x1_19_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_19_end_mask_0 = const()[name = tensor<string, []>("x1_19_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_19_cast_fp16 = slice_by_index(begin = x1_19_begin_0, end = x1_19_end_0, end_mask = x1_19_end_mask_0, x = k_17_cast_fp16)[name = tensor<string, []>("x1_19_cast_fp16")];
tensor<int32, [4]> x2_19_begin_0 = const()[name = tensor<string, []>("x2_19_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_19_end_0 = const()[name = tensor<string, []>("x2_19_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_19_end_mask_0 = const()[name = tensor<string, []>("x2_19_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_19_cast_fp16 = slice_by_index(begin = x2_19_begin_0, end = x2_19_end_0, end_mask = x2_19_end_mask_0, x = k_17_cast_fp16)[name = tensor<string, []>("x2_19_cast_fp16")];
tensor<fp16, []> const_39_promoted_to_fp16 = const()[name = tensor<string, []>("const_39_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1574_cast_fp16 = mul(x = x2_19_cast_fp16, y = const_39_promoted_to_fp16)[name = tensor<string, []>("op_1574_cast_fp16")];
tensor<bool, []> var_1576_interleave_0 = const()[name = tensor<string, []>("op_1576_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1576_cast_fp16 = concat(axis = var_1514, interleave = var_1576_interleave_0, values = (var_1574_cast_fp16, x1_19_cast_fp16))[name = tensor<string, []>("op_1576_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1577_cast_fp16 = mul(x = var_1576_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1577_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_9_cast_fp16 = add(x = var_1562_cast_fp16, y = var_1577_cast_fp16)[name = tensor<string, []>("k_embed_9_cast_fp16")];
tensor<bool, []> var_1582_transpose_x_1 = const()[name = tensor<string, []>("op_1582_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_1582_transpose_y_1 = const()[name = tensor<string, []>("op_1582_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_1582_cast_fp16 = matmul(transpose_x = var_1582_transpose_x_1, transpose_y = var_1582_transpose_y_1, x = q_embed_9_cast_fp16, y = k_embed_9_cast_fp16)[name = tensor<string, []>("op_1582_cast_fp16")];
tensor<fp16, []> var_1583_to_fp16 = const()[name = tensor<string, []>("op_1583_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_17_cast_fp16 = mul(x = var_1582_cast_fp16, y = var_1583_to_fp16)[name = tensor<string, []>("attn_weights_17_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_79_cast_fp16 = add(x = attn_weights_17_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_79_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_1586_cast_fp16 = softmax(axis = var_1514, x = input_79_cast_fp16)[name = tensor<string, []>("op_1586_cast_fp16")];
tensor<bool, []> attn_output_25_transpose_x_0 = const()[name = tensor<string, []>("attn_output_25_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_25_transpose_y_0 = const()[name = tensor<string, []>("attn_output_25_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_9_cast_fp16 = transpose(perm = value_9_perm_0, x = squeeze_14_cast_fp16)[name = tensor<string, []>("transpose_88")];
tensor<fp16, [1, 12, 256, 64]> attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_0, transpose_y = attn_output_25_transpose_y_0, x = var_1586_cast_fp16, y = value_9_cast_fp16)[name = tensor<string, []>("attn_output_25_cast_fp16")];
tensor<int32, [4]> var_1590_perm_0 = const()[name = tensor<string, []>("op_1590_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_1592 = const()[name = tensor<string, []>("op_1592"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_1590_cast_fp16 = transpose(perm = var_1590_perm_0, x = attn_output_25_cast_fp16)[name = tensor<string, []>("transpose_87")];
tensor<fp16, [1, 256, 768]> var_1593_cast_fp16 = reshape(shape = var_1592, x = var_1590_cast_fp16)[name = tensor<string, []>("op_1593_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_4_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_4_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(437212800)))];
tensor<fp16, [1, 256, 768]> linear_17_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_4_attn_Wo_weight_to_fp16, x = var_1593_cast_fp16)[name = tensor<string, []>("linear_17_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_85_cast_fp16 = add(x = input_77_cast_fp16, y = linear_17_cast_fp16)[name = tensor<string, []>("input_85_cast_fp16")];
tensor<int32, [1]> input_87_axes_0 = const()[name = tensor<string, []>("input_87_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_4_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_4_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(438392512)))];
tensor<fp16, [1, 256, 768]> input_87_cast_fp16 = layer_norm(axes = input_87_axes_0, epsilon = var_1525_to_fp16, gamma = model_encoder_layers_4_mlp_norm_weight_to_fp16, x = input_85_cast_fp16)[name = tensor<string, []>("input_87_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_4_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_4_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(438394112)))];
tensor<fp16, [1, 256, 2304]> linear_18_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_4_mlp_Wi_weight_to_fp16, x = input_87_cast_fp16)[name = tensor<string, []>("linear_18_cast_fp16")];
tensor<int32, [2]> var_1600_split_sizes_0 = const()[name = tensor<string, []>("op_1600_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_1600_axis_0 = const()[name = tensor<string, []>("op_1600_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_1600_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_1600_cast_fp16_1 = split(axis = var_1600_axis_0, split_sizes = var_1600_split_sizes_0, x = linear_18_cast_fp16)[name = tensor<string, []>("op_1600_cast_fp16")];
tensor<string, []> var_1602_mode_0 = const()[name = tensor<string, []>("op_1602_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_1602_cast_fp16 = gelu(mode = var_1602_mode_0, x = var_1600_cast_fp16_0)[name = tensor<string, []>("op_1602_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_91_cast_fp16 = mul(x = var_1602_cast_fp16, y = var_1600_cast_fp16_1)[name = tensor<string, []>("input_91_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_4_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_4_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(441933120)))];
tensor<fp16, [1, 256, 768]> linear_19_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_4_mlp_Wo_weight_to_fp16, x = input_91_cast_fp16)[name = tensor<string, []>("linear_19_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_95_cast_fp16 = add(x = input_85_cast_fp16, y = linear_19_cast_fp16)[name = tensor<string, []>("input_95_cast_fp16")];
tensor<int32, []> var_1611 = const()[name = tensor<string, []>("op_1611"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_11_axes_0 = const()[name = tensor<string, []>("hidden_states_11_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_5_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_5_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(443702656)))];
tensor<fp16, []> var_1622_to_fp16 = const()[name = tensor<string, []>("op_1622_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_11_cast_fp16 = layer_norm(axes = hidden_states_11_axes_0, epsilon = var_1622_to_fp16, gamma = model_encoder_layers_5_attn_norm_weight_to_fp16, x = input_95_cast_fp16)[name = tensor<string, []>("hidden_states_11_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_5_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_5_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(443704256)))];
tensor<fp16, [1, 256, 2304]> linear_20_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_5_attn_Wqkv_weight_to_fp16, x = hidden_states_11_cast_fp16)[name = tensor<string, []>("linear_20_cast_fp16")];
tensor<int32, [5]> var_1629 = const()[name = tensor<string, []>("op_1629"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_23_cast_fp16 = reshape(shape = var_1629, x = linear_20_cast_fp16)[name = tensor<string, []>("qkv_23_cast_fp16")];
tensor<int32, [3]> var_1631_split_sizes_0 = const()[name = tensor<string, []>("op_1631_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_1631_axis_0 = const()[name = tensor<string, []>("op_1631_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_1631_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_1631_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_1631_cast_fp16_2 = split(axis = var_1631_axis_0, split_sizes = var_1631_split_sizes_0, x = qkv_23_cast_fp16)[name = tensor<string, []>("op_1631_cast_fp16")];
tensor<int32, [1]> squeeze_15_axes_0 = const()[name = tensor<string, []>("squeeze_15_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_15_cast_fp16 = squeeze(axes = squeeze_15_axes_0, x = var_1631_cast_fp16_0)[name = tensor<string, []>("squeeze_15_cast_fp16")];
tensor<int32, [1]> squeeze_16_axes_0 = const()[name = tensor<string, []>("squeeze_16_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_16_cast_fp16 = squeeze(axes = squeeze_16_axes_0, x = var_1631_cast_fp16_1)[name = tensor<string, []>("squeeze_16_cast_fp16")];
tensor<int32, [1]> squeeze_17_axes_0 = const()[name = tensor<string, []>("squeeze_17_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_17_cast_fp16 = squeeze(axes = squeeze_17_axes_0, x = var_1631_cast_fp16_2)[name = tensor<string, []>("squeeze_17_cast_fp16")];
tensor<int32, [4]> q_21_perm_0 = const()[name = tensor<string, []>("q_21_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_21_perm_0 = const()[name = tensor<string, []>("k_21_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_11_perm_0 = const()[name = tensor<string, []>("value_11_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_21_cast_fp16 = transpose(perm = q_21_perm_0, x = squeeze_15_cast_fp16)[name = tensor<string, []>("transpose_86")];
tensor<fp16, [1, 12, 256, 64]> var_1641_cast_fp16 = mul(x = q_21_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1641_cast_fp16")];
tensor<int32, [4]> x1_21_begin_0 = const()[name = tensor<string, []>("x1_21_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_21_end_0 = const()[name = tensor<string, []>("x1_21_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_21_end_mask_0 = const()[name = tensor<string, []>("x1_21_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_21_cast_fp16 = slice_by_index(begin = x1_21_begin_0, end = x1_21_end_0, end_mask = x1_21_end_mask_0, x = q_21_cast_fp16)[name = tensor<string, []>("x1_21_cast_fp16")];
tensor<int32, [4]> x2_21_begin_0 = const()[name = tensor<string, []>("x2_21_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_21_end_0 = const()[name = tensor<string, []>("x2_21_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_21_end_mask_0 = const()[name = tensor<string, []>("x2_21_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_21_cast_fp16 = slice_by_index(begin = x2_21_begin_0, end = x2_21_end_0, end_mask = x2_21_end_mask_0, x = q_21_cast_fp16)[name = tensor<string, []>("x2_21_cast_fp16")];
tensor<fp16, []> const_44_promoted_to_fp16 = const()[name = tensor<string, []>("const_44_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1653_cast_fp16 = mul(x = x2_21_cast_fp16, y = const_44_promoted_to_fp16)[name = tensor<string, []>("op_1653_cast_fp16")];
tensor<bool, []> var_1655_interleave_0 = const()[name = tensor<string, []>("op_1655_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1655_cast_fp16 = concat(axis = var_1611, interleave = var_1655_interleave_0, values = (var_1653_cast_fp16, x1_21_cast_fp16))[name = tensor<string, []>("op_1655_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1656_cast_fp16 = mul(x = var_1655_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1656_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_11_cast_fp16 = add(x = var_1641_cast_fp16, y = var_1656_cast_fp16)[name = tensor<string, []>("q_embed_11_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_21_cast_fp16 = transpose(perm = k_21_perm_0, x = squeeze_16_cast_fp16)[name = tensor<string, []>("transpose_85")];
tensor<fp16, [1, 12, 256, 64]> var_1659_cast_fp16 = mul(x = k_21_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1659_cast_fp16")];
tensor<int32, [4]> x1_23_begin_0 = const()[name = tensor<string, []>("x1_23_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_23_end_0 = const()[name = tensor<string, []>("x1_23_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_23_end_mask_0 = const()[name = tensor<string, []>("x1_23_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_23_cast_fp16 = slice_by_index(begin = x1_23_begin_0, end = x1_23_end_0, end_mask = x1_23_end_mask_0, x = k_21_cast_fp16)[name = tensor<string, []>("x1_23_cast_fp16")];
tensor<int32, [4]> x2_23_begin_0 = const()[name = tensor<string, []>("x2_23_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_23_end_0 = const()[name = tensor<string, []>("x2_23_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_23_end_mask_0 = const()[name = tensor<string, []>("x2_23_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_23_cast_fp16 = slice_by_index(begin = x2_23_begin_0, end = x2_23_end_0, end_mask = x2_23_end_mask_0, x = k_21_cast_fp16)[name = tensor<string, []>("x2_23_cast_fp16")];
tensor<fp16, []> const_47_promoted_to_fp16 = const()[name = tensor<string, []>("const_47_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1671_cast_fp16 = mul(x = x2_23_cast_fp16, y = const_47_promoted_to_fp16)[name = tensor<string, []>("op_1671_cast_fp16")];
tensor<bool, []> var_1673_interleave_0 = const()[name = tensor<string, []>("op_1673_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1673_cast_fp16 = concat(axis = var_1611, interleave = var_1673_interleave_0, values = (var_1671_cast_fp16, x1_23_cast_fp16))[name = tensor<string, []>("op_1673_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1674_cast_fp16 = mul(x = var_1673_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1674_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_11_cast_fp16 = add(x = var_1659_cast_fp16, y = var_1674_cast_fp16)[name = tensor<string, []>("k_embed_11_cast_fp16")];
tensor<bool, []> var_1679_transpose_x_1 = const()[name = tensor<string, []>("op_1679_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_1679_transpose_y_1 = const()[name = tensor<string, []>("op_1679_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_1679_cast_fp16 = matmul(transpose_x = var_1679_transpose_x_1, transpose_y = var_1679_transpose_y_1, x = q_embed_11_cast_fp16, y = k_embed_11_cast_fp16)[name = tensor<string, []>("op_1679_cast_fp16")];
tensor<fp16, []> var_1680_to_fp16 = const()[name = tensor<string, []>("op_1680_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_21_cast_fp16 = mul(x = var_1679_cast_fp16, y = var_1680_to_fp16)[name = tensor<string, []>("attn_weights_21_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_97_cast_fp16 = add(x = attn_weights_21_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_97_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_1683_cast_fp16 = softmax(axis = var_1611, x = input_97_cast_fp16)[name = tensor<string, []>("op_1683_cast_fp16")];
tensor<bool, []> attn_output_31_transpose_x_0 = const()[name = tensor<string, []>("attn_output_31_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_31_transpose_y_0 = const()[name = tensor<string, []>("attn_output_31_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_11_cast_fp16 = transpose(perm = value_11_perm_0, x = squeeze_17_cast_fp16)[name = tensor<string, []>("transpose_84")];
tensor<fp16, [1, 12, 256, 64]> attn_output_31_cast_fp16 = matmul(transpose_x = attn_output_31_transpose_x_0, transpose_y = attn_output_31_transpose_y_0, x = var_1683_cast_fp16, y = value_11_cast_fp16)[name = tensor<string, []>("attn_output_31_cast_fp16")];
tensor<int32, [4]> var_1687_perm_0 = const()[name = tensor<string, []>("op_1687_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_1689 = const()[name = tensor<string, []>("op_1689"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_1687_cast_fp16 = transpose(perm = var_1687_perm_0, x = attn_output_31_cast_fp16)[name = tensor<string, []>("transpose_83")];
tensor<fp16, [1, 256, 768]> var_1690_cast_fp16 = reshape(shape = var_1689, x = var_1687_cast_fp16)[name = tensor<string, []>("op_1690_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_5_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_5_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(447243264)))];
tensor<fp16, [1, 256, 768]> linear_21_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_5_attn_Wo_weight_to_fp16, x = var_1690_cast_fp16)[name = tensor<string, []>("linear_21_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_103_cast_fp16 = add(x = input_95_cast_fp16, y = linear_21_cast_fp16)[name = tensor<string, []>("input_103_cast_fp16")];
tensor<int32, [1]> input_105_axes_0 = const()[name = tensor<string, []>("input_105_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_5_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_5_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(448422976)))];
tensor<fp16, [1, 256, 768]> input_105_cast_fp16 = layer_norm(axes = input_105_axes_0, epsilon = var_1622_to_fp16, gamma = model_encoder_layers_5_mlp_norm_weight_to_fp16, x = input_103_cast_fp16)[name = tensor<string, []>("input_105_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_5_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_5_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(448424576)))];
tensor<fp16, [1, 256, 2304]> linear_22_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_5_mlp_Wi_weight_to_fp16, x = input_105_cast_fp16)[name = tensor<string, []>("linear_22_cast_fp16")];
tensor<int32, [2]> var_1697_split_sizes_0 = const()[name = tensor<string, []>("op_1697_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_1697_axis_0 = const()[name = tensor<string, []>("op_1697_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_1697_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_1697_cast_fp16_1 = split(axis = var_1697_axis_0, split_sizes = var_1697_split_sizes_0, x = linear_22_cast_fp16)[name = tensor<string, []>("op_1697_cast_fp16")];
tensor<string, []> var_1699_mode_0 = const()[name = tensor<string, []>("op_1699_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_1699_cast_fp16 = gelu(mode = var_1699_mode_0, x = var_1697_cast_fp16_0)[name = tensor<string, []>("op_1699_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_109_cast_fp16 = mul(x = var_1699_cast_fp16, y = var_1697_cast_fp16_1)[name = tensor<string, []>("input_109_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_5_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_5_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(451963584)))];
tensor<fp16, [1, 256, 768]> linear_23_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_5_mlp_Wo_weight_to_fp16, x = input_109_cast_fp16)[name = tensor<string, []>("linear_23_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_113_cast_fp16 = add(x = input_103_cast_fp16, y = linear_23_cast_fp16)[name = tensor<string, []>("input_113_cast_fp16")];
tensor<int32, []> var_1708 = const()[name = tensor<string, []>("op_1708"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_13_axes_0 = const()[name = tensor<string, []>("hidden_states_13_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_6_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_6_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(453733120)))];
tensor<fp16, []> var_1719_to_fp16 = const()[name = tensor<string, []>("op_1719_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_13_cast_fp16 = layer_norm(axes = hidden_states_13_axes_0, epsilon = var_1719_to_fp16, gamma = model_encoder_layers_6_attn_norm_weight_to_fp16, x = input_113_cast_fp16)[name = tensor<string, []>("hidden_states_13_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_6_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_6_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(453734720)))];
tensor<fp16, [1, 256, 2304]> linear_24_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_6_attn_Wqkv_weight_to_fp16, x = hidden_states_13_cast_fp16)[name = tensor<string, []>("linear_24_cast_fp16")];
tensor<int32, [5]> var_1726 = const()[name = tensor<string, []>("op_1726"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_27_cast_fp16 = reshape(shape = var_1726, x = linear_24_cast_fp16)[name = tensor<string, []>("qkv_27_cast_fp16")];
tensor<int32, [3]> var_1728_split_sizes_0 = const()[name = tensor<string, []>("op_1728_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_1728_axis_0 = const()[name = tensor<string, []>("op_1728_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_1728_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_1728_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_1728_cast_fp16_2 = split(axis = var_1728_axis_0, split_sizes = var_1728_split_sizes_0, x = qkv_27_cast_fp16)[name = tensor<string, []>("op_1728_cast_fp16")];
tensor<int32, [1]> squeeze_18_axes_0 = const()[name = tensor<string, []>("squeeze_18_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_18_cast_fp16 = squeeze(axes = squeeze_18_axes_0, x = var_1728_cast_fp16_0)[name = tensor<string, []>("squeeze_18_cast_fp16")];
tensor<int32, [1]> squeeze_19_axes_0 = const()[name = tensor<string, []>("squeeze_19_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_19_cast_fp16 = squeeze(axes = squeeze_19_axes_0, x = var_1728_cast_fp16_1)[name = tensor<string, []>("squeeze_19_cast_fp16")];
tensor<int32, [1]> squeeze_20_axes_0 = const()[name = tensor<string, []>("squeeze_20_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_20_cast_fp16 = squeeze(axes = squeeze_20_axes_0, x = var_1728_cast_fp16_2)[name = tensor<string, []>("squeeze_20_cast_fp16")];
tensor<int32, [4]> q_25_perm_0 = const()[name = tensor<string, []>("q_25_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_25_perm_0 = const()[name = tensor<string, []>("k_25_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_13_perm_0 = const()[name = tensor<string, []>("value_13_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_25_cast_fp16 = transpose(perm = q_25_perm_0, x = squeeze_18_cast_fp16)[name = tensor<string, []>("transpose_82")];
tensor<fp16, [1, 12, 256, 64]> var_1738_cast_fp16 = mul(x = q_25_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1738_cast_fp16")];
tensor<int32, [4]> x1_25_begin_0 = const()[name = tensor<string, []>("x1_25_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_25_end_0 = const()[name = tensor<string, []>("x1_25_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_25_end_mask_0 = const()[name = tensor<string, []>("x1_25_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_25_cast_fp16 = slice_by_index(begin = x1_25_begin_0, end = x1_25_end_0, end_mask = x1_25_end_mask_0, x = q_25_cast_fp16)[name = tensor<string, []>("x1_25_cast_fp16")];
tensor<int32, [4]> x2_25_begin_0 = const()[name = tensor<string, []>("x2_25_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_25_end_0 = const()[name = tensor<string, []>("x2_25_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_25_end_mask_0 = const()[name = tensor<string, []>("x2_25_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_25_cast_fp16 = slice_by_index(begin = x2_25_begin_0, end = x2_25_end_0, end_mask = x2_25_end_mask_0, x = q_25_cast_fp16)[name = tensor<string, []>("x2_25_cast_fp16")];
tensor<fp16, []> const_52_promoted_to_fp16 = const()[name = tensor<string, []>("const_52_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1750_cast_fp16 = mul(x = x2_25_cast_fp16, y = const_52_promoted_to_fp16)[name = tensor<string, []>("op_1750_cast_fp16")];
tensor<bool, []> var_1752_interleave_0 = const()[name = tensor<string, []>("op_1752_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1752_cast_fp16 = concat(axis = var_1708, interleave = var_1752_interleave_0, values = (var_1750_cast_fp16, x1_25_cast_fp16))[name = tensor<string, []>("op_1752_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1753_cast_fp16 = mul(x = var_1752_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1753_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_13_cast_fp16 = add(x = var_1738_cast_fp16, y = var_1753_cast_fp16)[name = tensor<string, []>("q_embed_13_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_25_cast_fp16 = transpose(perm = k_25_perm_0, x = squeeze_19_cast_fp16)[name = tensor<string, []>("transpose_81")];
tensor<fp16, [1, 12, 256, 64]> var_1756_cast_fp16 = mul(x = k_25_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1756_cast_fp16")];
tensor<int32, [4]> x1_27_begin_0 = const()[name = tensor<string, []>("x1_27_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_27_end_0 = const()[name = tensor<string, []>("x1_27_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_27_end_mask_0 = const()[name = tensor<string, []>("x1_27_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_27_cast_fp16 = slice_by_index(begin = x1_27_begin_0, end = x1_27_end_0, end_mask = x1_27_end_mask_0, x = k_25_cast_fp16)[name = tensor<string, []>("x1_27_cast_fp16")];
tensor<int32, [4]> x2_27_begin_0 = const()[name = tensor<string, []>("x2_27_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_27_end_0 = const()[name = tensor<string, []>("x2_27_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_27_end_mask_0 = const()[name = tensor<string, []>("x2_27_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_27_cast_fp16 = slice_by_index(begin = x2_27_begin_0, end = x2_27_end_0, end_mask = x2_27_end_mask_0, x = k_25_cast_fp16)[name = tensor<string, []>("x2_27_cast_fp16")];
tensor<fp16, []> const_55_promoted_to_fp16 = const()[name = tensor<string, []>("const_55_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1768_cast_fp16 = mul(x = x2_27_cast_fp16, y = const_55_promoted_to_fp16)[name = tensor<string, []>("op_1768_cast_fp16")];
tensor<bool, []> var_1770_interleave_0 = const()[name = tensor<string, []>("op_1770_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1770_cast_fp16 = concat(axis = var_1708, interleave = var_1770_interleave_0, values = (var_1768_cast_fp16, x1_27_cast_fp16))[name = tensor<string, []>("op_1770_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1771_cast_fp16 = mul(x = var_1770_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1771_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_13_cast_fp16 = add(x = var_1756_cast_fp16, y = var_1771_cast_fp16)[name = tensor<string, []>("k_embed_13_cast_fp16")];
tensor<bool, []> var_1776_transpose_x_1 = const()[name = tensor<string, []>("op_1776_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_1776_transpose_y_1 = const()[name = tensor<string, []>("op_1776_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_1776_cast_fp16 = matmul(transpose_x = var_1776_transpose_x_1, transpose_y = var_1776_transpose_y_1, x = q_embed_13_cast_fp16, y = k_embed_13_cast_fp16)[name = tensor<string, []>("op_1776_cast_fp16")];
tensor<fp16, []> var_1777_to_fp16 = const()[name = tensor<string, []>("op_1777_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_25_cast_fp16 = mul(x = var_1776_cast_fp16, y = var_1777_to_fp16)[name = tensor<string, []>("attn_weights_25_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_115_cast_fp16 = add(x = attn_weights_25_cast_fp16, y = attention_mask_3_cast_fp16)[name = tensor<string, []>("input_115_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_1780_cast_fp16 = softmax(axis = var_1708, x = input_115_cast_fp16)[name = tensor<string, []>("op_1780_cast_fp16")];
tensor<bool, []> attn_output_37_transpose_x_0 = const()[name = tensor<string, []>("attn_output_37_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_37_transpose_y_0 = const()[name = tensor<string, []>("attn_output_37_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_13_cast_fp16 = transpose(perm = value_13_perm_0, x = squeeze_20_cast_fp16)[name = tensor<string, []>("transpose_80")];
tensor<fp16, [1, 12, 256, 64]> attn_output_37_cast_fp16 = matmul(transpose_x = attn_output_37_transpose_x_0, transpose_y = attn_output_37_transpose_y_0, x = var_1780_cast_fp16, y = value_13_cast_fp16)[name = tensor<string, []>("attn_output_37_cast_fp16")];
tensor<int32, [4]> var_1784_perm_0 = const()[name = tensor<string, []>("op_1784_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_1786 = const()[name = tensor<string, []>("op_1786"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_1784_cast_fp16 = transpose(perm = var_1784_perm_0, x = attn_output_37_cast_fp16)[name = tensor<string, []>("transpose_79")];
tensor<fp16, [1, 256, 768]> var_1787_cast_fp16 = reshape(shape = var_1786, x = var_1784_cast_fp16)[name = tensor<string, []>("op_1787_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_6_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_6_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(457273728)))];
tensor<fp16, [1, 256, 768]> linear_25_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_6_attn_Wo_weight_to_fp16, x = var_1787_cast_fp16)[name = tensor<string, []>("linear_25_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_121_cast_fp16 = add(x = input_113_cast_fp16, y = linear_25_cast_fp16)[name = tensor<string, []>("input_121_cast_fp16")];
tensor<int32, [1]> input_123_axes_0 = const()[name = tensor<string, []>("input_123_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_6_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_6_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(458453440)))];
tensor<fp16, [1, 256, 768]> input_123_cast_fp16 = layer_norm(axes = input_123_axes_0, epsilon = var_1719_to_fp16, gamma = model_encoder_layers_6_mlp_norm_weight_to_fp16, x = input_121_cast_fp16)[name = tensor<string, []>("input_123_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_6_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_6_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(458455040)))];
tensor<fp16, [1, 256, 2304]> linear_26_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_6_mlp_Wi_weight_to_fp16, x = input_123_cast_fp16)[name = tensor<string, []>("linear_26_cast_fp16")];
tensor<int32, [2]> var_1794_split_sizes_0 = const()[name = tensor<string, []>("op_1794_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_1794_axis_0 = const()[name = tensor<string, []>("op_1794_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_1794_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_1794_cast_fp16_1 = split(axis = var_1794_axis_0, split_sizes = var_1794_split_sizes_0, x = linear_26_cast_fp16)[name = tensor<string, []>("op_1794_cast_fp16")];
tensor<string, []> var_1796_mode_0 = const()[name = tensor<string, []>("op_1796_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_1796_cast_fp16 = gelu(mode = var_1796_mode_0, x = var_1794_cast_fp16_0)[name = tensor<string, []>("op_1796_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_127_cast_fp16 = mul(x = var_1796_cast_fp16, y = var_1794_cast_fp16_1)[name = tensor<string, []>("input_127_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_6_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_6_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(461994048)))];
tensor<fp16, [1, 256, 768]> linear_27_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_6_mlp_Wo_weight_to_fp16, x = input_127_cast_fp16)[name = tensor<string, []>("linear_27_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_131_cast_fp16 = add(x = input_121_cast_fp16, y = linear_27_cast_fp16)[name = tensor<string, []>("input_131_cast_fp16")];
tensor<int32, []> var_1805 = const()[name = tensor<string, []>("op_1805"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_15_axes_0 = const()[name = tensor<string, []>("hidden_states_15_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_7_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_7_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(463763584)))];
tensor<fp16, []> var_1816_to_fp16 = const()[name = tensor<string, []>("op_1816_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_15_cast_fp16 = layer_norm(axes = hidden_states_15_axes_0, epsilon = var_1816_to_fp16, gamma = model_encoder_layers_7_attn_norm_weight_to_fp16, x = input_131_cast_fp16)[name = tensor<string, []>("hidden_states_15_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_7_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_7_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(463765184)))];
tensor<fp16, [1, 256, 2304]> linear_28_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_7_attn_Wqkv_weight_to_fp16, x = hidden_states_15_cast_fp16)[name = tensor<string, []>("linear_28_cast_fp16")];
tensor<int32, [5]> var_1823 = const()[name = tensor<string, []>("op_1823"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_31_cast_fp16 = reshape(shape = var_1823, x = linear_28_cast_fp16)[name = tensor<string, []>("qkv_31_cast_fp16")];
tensor<int32, [3]> var_1825_split_sizes_0 = const()[name = tensor<string, []>("op_1825_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_1825_axis_0 = const()[name = tensor<string, []>("op_1825_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_1825_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_1825_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_1825_cast_fp16_2 = split(axis = var_1825_axis_0, split_sizes = var_1825_split_sizes_0, x = qkv_31_cast_fp16)[name = tensor<string, []>("op_1825_cast_fp16")];
tensor<int32, [1]> squeeze_21_axes_0 = const()[name = tensor<string, []>("squeeze_21_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_21_cast_fp16 = squeeze(axes = squeeze_21_axes_0, x = var_1825_cast_fp16_0)[name = tensor<string, []>("squeeze_21_cast_fp16")];
tensor<int32, [1]> squeeze_22_axes_0 = const()[name = tensor<string, []>("squeeze_22_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_22_cast_fp16 = squeeze(axes = squeeze_22_axes_0, x = var_1825_cast_fp16_1)[name = tensor<string, []>("squeeze_22_cast_fp16")];
tensor<int32, [1]> squeeze_23_axes_0 = const()[name = tensor<string, []>("squeeze_23_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_23_cast_fp16 = squeeze(axes = squeeze_23_axes_0, x = var_1825_cast_fp16_2)[name = tensor<string, []>("squeeze_23_cast_fp16")];
tensor<int32, [4]> q_29_perm_0 = const()[name = tensor<string, []>("q_29_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_29_perm_0 = const()[name = tensor<string, []>("k_29_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_15_perm_0 = const()[name = tensor<string, []>("value_15_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_29_cast_fp16 = transpose(perm = q_29_perm_0, x = squeeze_21_cast_fp16)[name = tensor<string, []>("transpose_78")];
tensor<fp16, [1, 12, 256, 64]> var_1835_cast_fp16 = mul(x = q_29_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1835_cast_fp16")];
tensor<int32, [4]> x1_29_begin_0 = const()[name = tensor<string, []>("x1_29_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_29_end_0 = const()[name = tensor<string, []>("x1_29_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_29_end_mask_0 = const()[name = tensor<string, []>("x1_29_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_29_cast_fp16 = slice_by_index(begin = x1_29_begin_0, end = x1_29_end_0, end_mask = x1_29_end_mask_0, x = q_29_cast_fp16)[name = tensor<string, []>("x1_29_cast_fp16")];
tensor<int32, [4]> x2_29_begin_0 = const()[name = tensor<string, []>("x2_29_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_29_end_0 = const()[name = tensor<string, []>("x2_29_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_29_end_mask_0 = const()[name = tensor<string, []>("x2_29_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_29_cast_fp16 = slice_by_index(begin = x2_29_begin_0, end = x2_29_end_0, end_mask = x2_29_end_mask_0, x = q_29_cast_fp16)[name = tensor<string, []>("x2_29_cast_fp16")];
tensor<fp16, []> const_60_promoted_to_fp16 = const()[name = tensor<string, []>("const_60_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1847_cast_fp16 = mul(x = x2_29_cast_fp16, y = const_60_promoted_to_fp16)[name = tensor<string, []>("op_1847_cast_fp16")];
tensor<bool, []> var_1849_interleave_0 = const()[name = tensor<string, []>("op_1849_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1849_cast_fp16 = concat(axis = var_1805, interleave = var_1849_interleave_0, values = (var_1847_cast_fp16, x1_29_cast_fp16))[name = tensor<string, []>("op_1849_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1850_cast_fp16 = mul(x = var_1849_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1850_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_15_cast_fp16 = add(x = var_1835_cast_fp16, y = var_1850_cast_fp16)[name = tensor<string, []>("q_embed_15_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_29_cast_fp16 = transpose(perm = k_29_perm_0, x = squeeze_22_cast_fp16)[name = tensor<string, []>("transpose_77")];
tensor<fp16, [1, 12, 256, 64]> var_1853_cast_fp16 = mul(x = k_29_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1853_cast_fp16")];
tensor<int32, [4]> x1_31_begin_0 = const()[name = tensor<string, []>("x1_31_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_31_end_0 = const()[name = tensor<string, []>("x1_31_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_31_end_mask_0 = const()[name = tensor<string, []>("x1_31_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_31_cast_fp16 = slice_by_index(begin = x1_31_begin_0, end = x1_31_end_0, end_mask = x1_31_end_mask_0, x = k_29_cast_fp16)[name = tensor<string, []>("x1_31_cast_fp16")];
tensor<int32, [4]> x2_31_begin_0 = const()[name = tensor<string, []>("x2_31_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_31_end_0 = const()[name = tensor<string, []>("x2_31_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_31_end_mask_0 = const()[name = tensor<string, []>("x2_31_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_31_cast_fp16 = slice_by_index(begin = x2_31_begin_0, end = x2_31_end_0, end_mask = x2_31_end_mask_0, x = k_29_cast_fp16)[name = tensor<string, []>("x2_31_cast_fp16")];
tensor<fp16, []> const_63_promoted_to_fp16 = const()[name = tensor<string, []>("const_63_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1865_cast_fp16 = mul(x = x2_31_cast_fp16, y = const_63_promoted_to_fp16)[name = tensor<string, []>("op_1865_cast_fp16")];
tensor<bool, []> var_1867_interleave_0 = const()[name = tensor<string, []>("op_1867_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1867_cast_fp16 = concat(axis = var_1805, interleave = var_1867_interleave_0, values = (var_1865_cast_fp16, x1_31_cast_fp16))[name = tensor<string, []>("op_1867_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1868_cast_fp16 = mul(x = var_1867_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1868_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_15_cast_fp16 = add(x = var_1853_cast_fp16, y = var_1868_cast_fp16)[name = tensor<string, []>("k_embed_15_cast_fp16")];
tensor<bool, []> var_1873_transpose_x_1 = const()[name = tensor<string, []>("op_1873_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_1873_transpose_y_1 = const()[name = tensor<string, []>("op_1873_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_1873_cast_fp16 = matmul(transpose_x = var_1873_transpose_x_1, transpose_y = var_1873_transpose_y_1, x = q_embed_15_cast_fp16, y = k_embed_15_cast_fp16)[name = tensor<string, []>("op_1873_cast_fp16")];
tensor<fp16, []> var_1874_to_fp16 = const()[name = tensor<string, []>("op_1874_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_29_cast_fp16 = mul(x = var_1873_cast_fp16, y = var_1874_to_fp16)[name = tensor<string, []>("attn_weights_29_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_133_cast_fp16 = add(x = attn_weights_29_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_133_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_1877_cast_fp16 = softmax(axis = var_1805, x = input_133_cast_fp16)[name = tensor<string, []>("op_1877_cast_fp16")];
tensor<bool, []> attn_output_43_transpose_x_0 = const()[name = tensor<string, []>("attn_output_43_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_43_transpose_y_0 = const()[name = tensor<string, []>("attn_output_43_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_15_cast_fp16 = transpose(perm = value_15_perm_0, x = squeeze_23_cast_fp16)[name = tensor<string, []>("transpose_76")];
tensor<fp16, [1, 12, 256, 64]> attn_output_43_cast_fp16 = matmul(transpose_x = attn_output_43_transpose_x_0, transpose_y = attn_output_43_transpose_y_0, x = var_1877_cast_fp16, y = value_15_cast_fp16)[name = tensor<string, []>("attn_output_43_cast_fp16")];
tensor<int32, [4]> var_1881_perm_0 = const()[name = tensor<string, []>("op_1881_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_1883 = const()[name = tensor<string, []>("op_1883"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_1881_cast_fp16 = transpose(perm = var_1881_perm_0, x = attn_output_43_cast_fp16)[name = tensor<string, []>("transpose_75")];
tensor<fp16, [1, 256, 768]> var_1884_cast_fp16 = reshape(shape = var_1883, x = var_1881_cast_fp16)[name = tensor<string, []>("op_1884_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_7_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_7_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(467304192)))];
tensor<fp16, [1, 256, 768]> linear_29_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_7_attn_Wo_weight_to_fp16, x = var_1884_cast_fp16)[name = tensor<string, []>("linear_29_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_139_cast_fp16 = add(x = input_131_cast_fp16, y = linear_29_cast_fp16)[name = tensor<string, []>("input_139_cast_fp16")];
tensor<int32, [1]> input_141_axes_0 = const()[name = tensor<string, []>("input_141_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_7_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_7_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(468483904)))];
tensor<fp16, [1, 256, 768]> input_141_cast_fp16 = layer_norm(axes = input_141_axes_0, epsilon = var_1816_to_fp16, gamma = model_encoder_layers_7_mlp_norm_weight_to_fp16, x = input_139_cast_fp16)[name = tensor<string, []>("input_141_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_7_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_7_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(468485504)))];
tensor<fp16, [1, 256, 2304]> linear_30_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_7_mlp_Wi_weight_to_fp16, x = input_141_cast_fp16)[name = tensor<string, []>("linear_30_cast_fp16")];
tensor<int32, [2]> var_1891_split_sizes_0 = const()[name = tensor<string, []>("op_1891_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_1891_axis_0 = const()[name = tensor<string, []>("op_1891_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_1891_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_1891_cast_fp16_1 = split(axis = var_1891_axis_0, split_sizes = var_1891_split_sizes_0, x = linear_30_cast_fp16)[name = tensor<string, []>("op_1891_cast_fp16")];
tensor<string, []> var_1893_mode_0 = const()[name = tensor<string, []>("op_1893_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_1893_cast_fp16 = gelu(mode = var_1893_mode_0, x = var_1891_cast_fp16_0)[name = tensor<string, []>("op_1893_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_145_cast_fp16 = mul(x = var_1893_cast_fp16, y = var_1891_cast_fp16_1)[name = tensor<string, []>("input_145_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_7_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_7_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(472024512)))];
tensor<fp16, [1, 256, 768]> linear_31_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_7_mlp_Wo_weight_to_fp16, x = input_145_cast_fp16)[name = tensor<string, []>("linear_31_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_149_cast_fp16 = add(x = input_139_cast_fp16, y = linear_31_cast_fp16)[name = tensor<string, []>("input_149_cast_fp16")];
tensor<int32, []> var_1902 = const()[name = tensor<string, []>("op_1902"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_17_axes_0 = const()[name = tensor<string, []>("hidden_states_17_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_8_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_8_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(473794048)))];
tensor<fp16, []> var_1913_to_fp16 = const()[name = tensor<string, []>("op_1913_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_17_cast_fp16 = layer_norm(axes = hidden_states_17_axes_0, epsilon = var_1913_to_fp16, gamma = model_encoder_layers_8_attn_norm_weight_to_fp16, x = input_149_cast_fp16)[name = tensor<string, []>("hidden_states_17_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_8_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_8_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(473795648)))];
tensor<fp16, [1, 256, 2304]> linear_32_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_8_attn_Wqkv_weight_to_fp16, x = hidden_states_17_cast_fp16)[name = tensor<string, []>("linear_32_cast_fp16")];
tensor<int32, [5]> var_1920 = const()[name = tensor<string, []>("op_1920"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_35_cast_fp16 = reshape(shape = var_1920, x = linear_32_cast_fp16)[name = tensor<string, []>("qkv_35_cast_fp16")];
tensor<int32, [3]> var_1922_split_sizes_0 = const()[name = tensor<string, []>("op_1922_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_1922_axis_0 = const()[name = tensor<string, []>("op_1922_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_1922_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_1922_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_1922_cast_fp16_2 = split(axis = var_1922_axis_0, split_sizes = var_1922_split_sizes_0, x = qkv_35_cast_fp16)[name = tensor<string, []>("op_1922_cast_fp16")];
tensor<int32, [1]> squeeze_24_axes_0 = const()[name = tensor<string, []>("squeeze_24_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_24_cast_fp16 = squeeze(axes = squeeze_24_axes_0, x = var_1922_cast_fp16_0)[name = tensor<string, []>("squeeze_24_cast_fp16")];
tensor<int32, [1]> squeeze_25_axes_0 = const()[name = tensor<string, []>("squeeze_25_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_25_cast_fp16 = squeeze(axes = squeeze_25_axes_0, x = var_1922_cast_fp16_1)[name = tensor<string, []>("squeeze_25_cast_fp16")];
tensor<int32, [1]> squeeze_26_axes_0 = const()[name = tensor<string, []>("squeeze_26_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_26_cast_fp16 = squeeze(axes = squeeze_26_axes_0, x = var_1922_cast_fp16_2)[name = tensor<string, []>("squeeze_26_cast_fp16")];
tensor<int32, [4]> q_33_perm_0 = const()[name = tensor<string, []>("q_33_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_33_perm_0 = const()[name = tensor<string, []>("k_33_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_17_perm_0 = const()[name = tensor<string, []>("value_17_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_33_cast_fp16 = transpose(perm = q_33_perm_0, x = squeeze_24_cast_fp16)[name = tensor<string, []>("transpose_74")];
tensor<fp16, [1, 12, 256, 64]> var_1932_cast_fp16 = mul(x = q_33_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1932_cast_fp16")];
tensor<int32, [4]> x1_33_begin_0 = const()[name = tensor<string, []>("x1_33_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_33_end_0 = const()[name = tensor<string, []>("x1_33_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_33_end_mask_0 = const()[name = tensor<string, []>("x1_33_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_33_cast_fp16 = slice_by_index(begin = x1_33_begin_0, end = x1_33_end_0, end_mask = x1_33_end_mask_0, x = q_33_cast_fp16)[name = tensor<string, []>("x1_33_cast_fp16")];
tensor<int32, [4]> x2_33_begin_0 = const()[name = tensor<string, []>("x2_33_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_33_end_0 = const()[name = tensor<string, []>("x2_33_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_33_end_mask_0 = const()[name = tensor<string, []>("x2_33_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_33_cast_fp16 = slice_by_index(begin = x2_33_begin_0, end = x2_33_end_0, end_mask = x2_33_end_mask_0, x = q_33_cast_fp16)[name = tensor<string, []>("x2_33_cast_fp16")];
tensor<fp16, []> const_68_promoted_to_fp16 = const()[name = tensor<string, []>("const_68_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1944_cast_fp16 = mul(x = x2_33_cast_fp16, y = const_68_promoted_to_fp16)[name = tensor<string, []>("op_1944_cast_fp16")];
tensor<bool, []> var_1946_interleave_0 = const()[name = tensor<string, []>("op_1946_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1946_cast_fp16 = concat(axis = var_1902, interleave = var_1946_interleave_0, values = (var_1944_cast_fp16, x1_33_cast_fp16))[name = tensor<string, []>("op_1946_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1947_cast_fp16 = mul(x = var_1946_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1947_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_17_cast_fp16 = add(x = var_1932_cast_fp16, y = var_1947_cast_fp16)[name = tensor<string, []>("q_embed_17_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_33_cast_fp16 = transpose(perm = k_33_perm_0, x = squeeze_25_cast_fp16)[name = tensor<string, []>("transpose_73")];
tensor<fp16, [1, 12, 256, 64]> var_1950_cast_fp16 = mul(x = k_33_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_1950_cast_fp16")];
tensor<int32, [4]> x1_35_begin_0 = const()[name = tensor<string, []>("x1_35_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_35_end_0 = const()[name = tensor<string, []>("x1_35_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_35_end_mask_0 = const()[name = tensor<string, []>("x1_35_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_35_cast_fp16 = slice_by_index(begin = x1_35_begin_0, end = x1_35_end_0, end_mask = x1_35_end_mask_0, x = k_33_cast_fp16)[name = tensor<string, []>("x1_35_cast_fp16")];
tensor<int32, [4]> x2_35_begin_0 = const()[name = tensor<string, []>("x2_35_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_35_end_0 = const()[name = tensor<string, []>("x2_35_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_35_end_mask_0 = const()[name = tensor<string, []>("x2_35_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_35_cast_fp16 = slice_by_index(begin = x2_35_begin_0, end = x2_35_end_0, end_mask = x2_35_end_mask_0, x = k_33_cast_fp16)[name = tensor<string, []>("x2_35_cast_fp16")];
tensor<fp16, []> const_71_promoted_to_fp16 = const()[name = tensor<string, []>("const_71_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_1962_cast_fp16 = mul(x = x2_35_cast_fp16, y = const_71_promoted_to_fp16)[name = tensor<string, []>("op_1962_cast_fp16")];
tensor<bool, []> var_1964_interleave_0 = const()[name = tensor<string, []>("op_1964_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_1964_cast_fp16 = concat(axis = var_1902, interleave = var_1964_interleave_0, values = (var_1962_cast_fp16, x1_35_cast_fp16))[name = tensor<string, []>("op_1964_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_1965_cast_fp16 = mul(x = var_1964_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_1965_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_17_cast_fp16 = add(x = var_1950_cast_fp16, y = var_1965_cast_fp16)[name = tensor<string, []>("k_embed_17_cast_fp16")];
tensor<bool, []> var_1970_transpose_x_1 = const()[name = tensor<string, []>("op_1970_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_1970_transpose_y_1 = const()[name = tensor<string, []>("op_1970_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_1970_cast_fp16 = matmul(transpose_x = var_1970_transpose_x_1, transpose_y = var_1970_transpose_y_1, x = q_embed_17_cast_fp16, y = k_embed_17_cast_fp16)[name = tensor<string, []>("op_1970_cast_fp16")];
tensor<fp16, []> var_1971_to_fp16 = const()[name = tensor<string, []>("op_1971_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_33_cast_fp16 = mul(x = var_1970_cast_fp16, y = var_1971_to_fp16)[name = tensor<string, []>("attn_weights_33_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_151_cast_fp16 = add(x = attn_weights_33_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_151_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_1974_cast_fp16 = softmax(axis = var_1902, x = input_151_cast_fp16)[name = tensor<string, []>("op_1974_cast_fp16")];
tensor<bool, []> attn_output_49_transpose_x_0 = const()[name = tensor<string, []>("attn_output_49_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_49_transpose_y_0 = const()[name = tensor<string, []>("attn_output_49_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_17_cast_fp16 = transpose(perm = value_17_perm_0, x = squeeze_26_cast_fp16)[name = tensor<string, []>("transpose_72")];
tensor<fp16, [1, 12, 256, 64]> attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_0, transpose_y = attn_output_49_transpose_y_0, x = var_1974_cast_fp16, y = value_17_cast_fp16)[name = tensor<string, []>("attn_output_49_cast_fp16")];
tensor<int32, [4]> var_1978_perm_0 = const()[name = tensor<string, []>("op_1978_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_1980 = const()[name = tensor<string, []>("op_1980"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_1978_cast_fp16 = transpose(perm = var_1978_perm_0, x = attn_output_49_cast_fp16)[name = tensor<string, []>("transpose_71")];
tensor<fp16, [1, 256, 768]> var_1981_cast_fp16 = reshape(shape = var_1980, x = var_1978_cast_fp16)[name = tensor<string, []>("op_1981_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_8_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_8_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(477334656)))];
tensor<fp16, [1, 256, 768]> linear_33_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_8_attn_Wo_weight_to_fp16, x = var_1981_cast_fp16)[name = tensor<string, []>("linear_33_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_157_cast_fp16 = add(x = input_149_cast_fp16, y = linear_33_cast_fp16)[name = tensor<string, []>("input_157_cast_fp16")];
tensor<int32, [1]> input_159_axes_0 = const()[name = tensor<string, []>("input_159_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_8_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_8_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(478514368)))];
tensor<fp16, [1, 256, 768]> input_159_cast_fp16 = layer_norm(axes = input_159_axes_0, epsilon = var_1913_to_fp16, gamma = model_encoder_layers_8_mlp_norm_weight_to_fp16, x = input_157_cast_fp16)[name = tensor<string, []>("input_159_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_8_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_8_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(478515968)))];
tensor<fp16, [1, 256, 2304]> linear_34_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_8_mlp_Wi_weight_to_fp16, x = input_159_cast_fp16)[name = tensor<string, []>("linear_34_cast_fp16")];
tensor<int32, [2]> var_1988_split_sizes_0 = const()[name = tensor<string, []>("op_1988_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_1988_axis_0 = const()[name = tensor<string, []>("op_1988_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_1988_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_1988_cast_fp16_1 = split(axis = var_1988_axis_0, split_sizes = var_1988_split_sizes_0, x = linear_34_cast_fp16)[name = tensor<string, []>("op_1988_cast_fp16")];
tensor<string, []> var_1990_mode_0 = const()[name = tensor<string, []>("op_1990_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_1990_cast_fp16 = gelu(mode = var_1990_mode_0, x = var_1988_cast_fp16_0)[name = tensor<string, []>("op_1990_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_163_cast_fp16 = mul(x = var_1990_cast_fp16, y = var_1988_cast_fp16_1)[name = tensor<string, []>("input_163_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_8_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_8_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(482054976)))];
tensor<fp16, [1, 256, 768]> linear_35_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_8_mlp_Wo_weight_to_fp16, x = input_163_cast_fp16)[name = tensor<string, []>("linear_35_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_167_cast_fp16 = add(x = input_157_cast_fp16, y = linear_35_cast_fp16)[name = tensor<string, []>("input_167_cast_fp16")];
tensor<int32, []> var_1999 = const()[name = tensor<string, []>("op_1999"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_19_axes_0 = const()[name = tensor<string, []>("hidden_states_19_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_9_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_9_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(483824512)))];
tensor<fp16, []> var_2010_to_fp16 = const()[name = tensor<string, []>("op_2010_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_19_cast_fp16 = layer_norm(axes = hidden_states_19_axes_0, epsilon = var_2010_to_fp16, gamma = model_encoder_layers_9_attn_norm_weight_to_fp16, x = input_167_cast_fp16)[name = tensor<string, []>("hidden_states_19_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_9_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_9_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(483826112)))];
tensor<fp16, [1, 256, 2304]> linear_36_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_9_attn_Wqkv_weight_to_fp16, x = hidden_states_19_cast_fp16)[name = tensor<string, []>("linear_36_cast_fp16")];
tensor<int32, [5]> var_2017 = const()[name = tensor<string, []>("op_2017"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_39_cast_fp16 = reshape(shape = var_2017, x = linear_36_cast_fp16)[name = tensor<string, []>("qkv_39_cast_fp16")];
tensor<int32, [3]> var_2019_split_sizes_0 = const()[name = tensor<string, []>("op_2019_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_2019_axis_0 = const()[name = tensor<string, []>("op_2019_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_2019_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_2019_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_2019_cast_fp16_2 = split(axis = var_2019_axis_0, split_sizes = var_2019_split_sizes_0, x = qkv_39_cast_fp16)[name = tensor<string, []>("op_2019_cast_fp16")];
tensor<int32, [1]> squeeze_27_axes_0 = const()[name = tensor<string, []>("squeeze_27_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_27_cast_fp16 = squeeze(axes = squeeze_27_axes_0, x = var_2019_cast_fp16_0)[name = tensor<string, []>("squeeze_27_cast_fp16")];
tensor<int32, [1]> squeeze_28_axes_0 = const()[name = tensor<string, []>("squeeze_28_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_28_cast_fp16 = squeeze(axes = squeeze_28_axes_0, x = var_2019_cast_fp16_1)[name = tensor<string, []>("squeeze_28_cast_fp16")];
tensor<int32, [1]> squeeze_29_axes_0 = const()[name = tensor<string, []>("squeeze_29_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_29_cast_fp16 = squeeze(axes = squeeze_29_axes_0, x = var_2019_cast_fp16_2)[name = tensor<string, []>("squeeze_29_cast_fp16")];
tensor<int32, [4]> q_37_perm_0 = const()[name = tensor<string, []>("q_37_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_37_perm_0 = const()[name = tensor<string, []>("k_37_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_19_perm_0 = const()[name = tensor<string, []>("value_19_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_37_cast_fp16 = transpose(perm = q_37_perm_0, x = squeeze_27_cast_fp16)[name = tensor<string, []>("transpose_70")];
tensor<fp16, [1, 12, 256, 64]> var_2029_cast_fp16 = mul(x = q_37_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2029_cast_fp16")];
tensor<int32, [4]> x1_37_begin_0 = const()[name = tensor<string, []>("x1_37_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_37_end_0 = const()[name = tensor<string, []>("x1_37_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_37_end_mask_0 = const()[name = tensor<string, []>("x1_37_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_37_cast_fp16 = slice_by_index(begin = x1_37_begin_0, end = x1_37_end_0, end_mask = x1_37_end_mask_0, x = q_37_cast_fp16)[name = tensor<string, []>("x1_37_cast_fp16")];
tensor<int32, [4]> x2_37_begin_0 = const()[name = tensor<string, []>("x2_37_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_37_end_0 = const()[name = tensor<string, []>("x2_37_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_37_end_mask_0 = const()[name = tensor<string, []>("x2_37_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_37_cast_fp16 = slice_by_index(begin = x2_37_begin_0, end = x2_37_end_0, end_mask = x2_37_end_mask_0, x = q_37_cast_fp16)[name = tensor<string, []>("x2_37_cast_fp16")];
tensor<fp16, []> const_76_promoted_to_fp16 = const()[name = tensor<string, []>("const_76_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2041_cast_fp16 = mul(x = x2_37_cast_fp16, y = const_76_promoted_to_fp16)[name = tensor<string, []>("op_2041_cast_fp16")];
tensor<bool, []> var_2043_interleave_0 = const()[name = tensor<string, []>("op_2043_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2043_cast_fp16 = concat(axis = var_1999, interleave = var_2043_interleave_0, values = (var_2041_cast_fp16, x1_37_cast_fp16))[name = tensor<string, []>("op_2043_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2044_cast_fp16 = mul(x = var_2043_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2044_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_19_cast_fp16 = add(x = var_2029_cast_fp16, y = var_2044_cast_fp16)[name = tensor<string, []>("q_embed_19_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_37_cast_fp16 = transpose(perm = k_37_perm_0, x = squeeze_28_cast_fp16)[name = tensor<string, []>("transpose_69")];
tensor<fp16, [1, 12, 256, 64]> var_2047_cast_fp16 = mul(x = k_37_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2047_cast_fp16")];
tensor<int32, [4]> x1_39_begin_0 = const()[name = tensor<string, []>("x1_39_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_39_end_0 = const()[name = tensor<string, []>("x1_39_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_39_end_mask_0 = const()[name = tensor<string, []>("x1_39_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_39_cast_fp16 = slice_by_index(begin = x1_39_begin_0, end = x1_39_end_0, end_mask = x1_39_end_mask_0, x = k_37_cast_fp16)[name = tensor<string, []>("x1_39_cast_fp16")];
tensor<int32, [4]> x2_39_begin_0 = const()[name = tensor<string, []>("x2_39_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_39_end_0 = const()[name = tensor<string, []>("x2_39_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_39_end_mask_0 = const()[name = tensor<string, []>("x2_39_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_39_cast_fp16 = slice_by_index(begin = x2_39_begin_0, end = x2_39_end_0, end_mask = x2_39_end_mask_0, x = k_37_cast_fp16)[name = tensor<string, []>("x2_39_cast_fp16")];
tensor<fp16, []> const_79_promoted_to_fp16 = const()[name = tensor<string, []>("const_79_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2059_cast_fp16 = mul(x = x2_39_cast_fp16, y = const_79_promoted_to_fp16)[name = tensor<string, []>("op_2059_cast_fp16")];
tensor<bool, []> var_2061_interleave_0 = const()[name = tensor<string, []>("op_2061_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2061_cast_fp16 = concat(axis = var_1999, interleave = var_2061_interleave_0, values = (var_2059_cast_fp16, x1_39_cast_fp16))[name = tensor<string, []>("op_2061_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2062_cast_fp16 = mul(x = var_2061_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2062_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_19_cast_fp16 = add(x = var_2047_cast_fp16, y = var_2062_cast_fp16)[name = tensor<string, []>("k_embed_19_cast_fp16")];
tensor<bool, []> var_2067_transpose_x_1 = const()[name = tensor<string, []>("op_2067_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_2067_transpose_y_1 = const()[name = tensor<string, []>("op_2067_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_2067_cast_fp16 = matmul(transpose_x = var_2067_transpose_x_1, transpose_y = var_2067_transpose_y_1, x = q_embed_19_cast_fp16, y = k_embed_19_cast_fp16)[name = tensor<string, []>("op_2067_cast_fp16")];
tensor<fp16, []> var_2068_to_fp16 = const()[name = tensor<string, []>("op_2068_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_37_cast_fp16 = mul(x = var_2067_cast_fp16, y = var_2068_to_fp16)[name = tensor<string, []>("attn_weights_37_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_169_cast_fp16 = add(x = attn_weights_37_cast_fp16, y = attention_mask_3_cast_fp16)[name = tensor<string, []>("input_169_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_2071_cast_fp16 = softmax(axis = var_1999, x = input_169_cast_fp16)[name = tensor<string, []>("op_2071_cast_fp16")];
tensor<bool, []> attn_output_55_transpose_x_0 = const()[name = tensor<string, []>("attn_output_55_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_55_transpose_y_0 = const()[name = tensor<string, []>("attn_output_55_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_19_cast_fp16 = transpose(perm = value_19_perm_0, x = squeeze_29_cast_fp16)[name = tensor<string, []>("transpose_68")];
tensor<fp16, [1, 12, 256, 64]> attn_output_55_cast_fp16 = matmul(transpose_x = attn_output_55_transpose_x_0, transpose_y = attn_output_55_transpose_y_0, x = var_2071_cast_fp16, y = value_19_cast_fp16)[name = tensor<string, []>("attn_output_55_cast_fp16")];
tensor<int32, [4]> var_2075_perm_0 = const()[name = tensor<string, []>("op_2075_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_2077 = const()[name = tensor<string, []>("op_2077"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_2075_cast_fp16 = transpose(perm = var_2075_perm_0, x = attn_output_55_cast_fp16)[name = tensor<string, []>("transpose_67")];
tensor<fp16, [1, 256, 768]> var_2078_cast_fp16 = reshape(shape = var_2077, x = var_2075_cast_fp16)[name = tensor<string, []>("op_2078_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_9_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_9_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(487365120)))];
tensor<fp16, [1, 256, 768]> linear_37_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_9_attn_Wo_weight_to_fp16, x = var_2078_cast_fp16)[name = tensor<string, []>("linear_37_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_175_cast_fp16 = add(x = input_167_cast_fp16, y = linear_37_cast_fp16)[name = tensor<string, []>("input_175_cast_fp16")];
tensor<int32, [1]> input_177_axes_0 = const()[name = tensor<string, []>("input_177_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_9_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_9_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(488544832)))];
tensor<fp16, [1, 256, 768]> input_177_cast_fp16 = layer_norm(axes = input_177_axes_0, epsilon = var_2010_to_fp16, gamma = model_encoder_layers_9_mlp_norm_weight_to_fp16, x = input_175_cast_fp16)[name = tensor<string, []>("input_177_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_9_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_9_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(488546432)))];
tensor<fp16, [1, 256, 2304]> linear_38_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_9_mlp_Wi_weight_to_fp16, x = input_177_cast_fp16)[name = tensor<string, []>("linear_38_cast_fp16")];
tensor<int32, [2]> var_2085_split_sizes_0 = const()[name = tensor<string, []>("op_2085_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_2085_axis_0 = const()[name = tensor<string, []>("op_2085_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_2085_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_2085_cast_fp16_1 = split(axis = var_2085_axis_0, split_sizes = var_2085_split_sizes_0, x = linear_38_cast_fp16)[name = tensor<string, []>("op_2085_cast_fp16")];
tensor<string, []> var_2087_mode_0 = const()[name = tensor<string, []>("op_2087_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_2087_cast_fp16 = gelu(mode = var_2087_mode_0, x = var_2085_cast_fp16_0)[name = tensor<string, []>("op_2087_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_181_cast_fp16 = mul(x = var_2087_cast_fp16, y = var_2085_cast_fp16_1)[name = tensor<string, []>("input_181_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_9_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_9_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(492085440)))];
tensor<fp16, [1, 256, 768]> linear_39_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_9_mlp_Wo_weight_to_fp16, x = input_181_cast_fp16)[name = tensor<string, []>("linear_39_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_185_cast_fp16 = add(x = input_175_cast_fp16, y = linear_39_cast_fp16)[name = tensor<string, []>("input_185_cast_fp16")];
tensor<int32, []> var_2096 = const()[name = tensor<string, []>("op_2096"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_21_axes_0 = const()[name = tensor<string, []>("hidden_states_21_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_10_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_10_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(493854976)))];
tensor<fp16, []> var_2107_to_fp16 = const()[name = tensor<string, []>("op_2107_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_21_cast_fp16 = layer_norm(axes = hidden_states_21_axes_0, epsilon = var_2107_to_fp16, gamma = model_encoder_layers_10_attn_norm_weight_to_fp16, x = input_185_cast_fp16)[name = tensor<string, []>("hidden_states_21_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_10_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_10_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(493856576)))];
tensor<fp16, [1, 256, 2304]> linear_40_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_10_attn_Wqkv_weight_to_fp16, x = hidden_states_21_cast_fp16)[name = tensor<string, []>("linear_40_cast_fp16")];
tensor<int32, [5]> var_2114 = const()[name = tensor<string, []>("op_2114"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_43_cast_fp16 = reshape(shape = var_2114, x = linear_40_cast_fp16)[name = tensor<string, []>("qkv_43_cast_fp16")];
tensor<int32, [3]> var_2116_split_sizes_0 = const()[name = tensor<string, []>("op_2116_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_2116_axis_0 = const()[name = tensor<string, []>("op_2116_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_2116_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_2116_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_2116_cast_fp16_2 = split(axis = var_2116_axis_0, split_sizes = var_2116_split_sizes_0, x = qkv_43_cast_fp16)[name = tensor<string, []>("op_2116_cast_fp16")];
tensor<int32, [1]> squeeze_30_axes_0 = const()[name = tensor<string, []>("squeeze_30_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_30_cast_fp16 = squeeze(axes = squeeze_30_axes_0, x = var_2116_cast_fp16_0)[name = tensor<string, []>("squeeze_30_cast_fp16")];
tensor<int32, [1]> squeeze_31_axes_0 = const()[name = tensor<string, []>("squeeze_31_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_31_cast_fp16 = squeeze(axes = squeeze_31_axes_0, x = var_2116_cast_fp16_1)[name = tensor<string, []>("squeeze_31_cast_fp16")];
tensor<int32, [1]> squeeze_32_axes_0 = const()[name = tensor<string, []>("squeeze_32_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_32_cast_fp16 = squeeze(axes = squeeze_32_axes_0, x = var_2116_cast_fp16_2)[name = tensor<string, []>("squeeze_32_cast_fp16")];
tensor<int32, [4]> q_41_perm_0 = const()[name = tensor<string, []>("q_41_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_41_perm_0 = const()[name = tensor<string, []>("k_41_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_21_perm_0 = const()[name = tensor<string, []>("value_21_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_41_cast_fp16 = transpose(perm = q_41_perm_0, x = squeeze_30_cast_fp16)[name = tensor<string, []>("transpose_66")];
tensor<fp16, [1, 12, 256, 64]> var_2126_cast_fp16 = mul(x = q_41_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2126_cast_fp16")];
tensor<int32, [4]> x1_41_begin_0 = const()[name = tensor<string, []>("x1_41_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_41_end_0 = const()[name = tensor<string, []>("x1_41_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_41_end_mask_0 = const()[name = tensor<string, []>("x1_41_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_41_cast_fp16 = slice_by_index(begin = x1_41_begin_0, end = x1_41_end_0, end_mask = x1_41_end_mask_0, x = q_41_cast_fp16)[name = tensor<string, []>("x1_41_cast_fp16")];
tensor<int32, [4]> x2_41_begin_0 = const()[name = tensor<string, []>("x2_41_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_41_end_0 = const()[name = tensor<string, []>("x2_41_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_41_end_mask_0 = const()[name = tensor<string, []>("x2_41_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_41_cast_fp16 = slice_by_index(begin = x2_41_begin_0, end = x2_41_end_0, end_mask = x2_41_end_mask_0, x = q_41_cast_fp16)[name = tensor<string, []>("x2_41_cast_fp16")];
tensor<fp16, []> const_84_promoted_to_fp16 = const()[name = tensor<string, []>("const_84_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2138_cast_fp16 = mul(x = x2_41_cast_fp16, y = const_84_promoted_to_fp16)[name = tensor<string, []>("op_2138_cast_fp16")];
tensor<bool, []> var_2140_interleave_0 = const()[name = tensor<string, []>("op_2140_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2140_cast_fp16 = concat(axis = var_2096, interleave = var_2140_interleave_0, values = (var_2138_cast_fp16, x1_41_cast_fp16))[name = tensor<string, []>("op_2140_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2141_cast_fp16 = mul(x = var_2140_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2141_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_21_cast_fp16 = add(x = var_2126_cast_fp16, y = var_2141_cast_fp16)[name = tensor<string, []>("q_embed_21_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_41_cast_fp16 = transpose(perm = k_41_perm_0, x = squeeze_31_cast_fp16)[name = tensor<string, []>("transpose_65")];
tensor<fp16, [1, 12, 256, 64]> var_2144_cast_fp16 = mul(x = k_41_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2144_cast_fp16")];
tensor<int32, [4]> x1_43_begin_0 = const()[name = tensor<string, []>("x1_43_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_43_end_0 = const()[name = tensor<string, []>("x1_43_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_43_end_mask_0 = const()[name = tensor<string, []>("x1_43_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_43_cast_fp16 = slice_by_index(begin = x1_43_begin_0, end = x1_43_end_0, end_mask = x1_43_end_mask_0, x = k_41_cast_fp16)[name = tensor<string, []>("x1_43_cast_fp16")];
tensor<int32, [4]> x2_43_begin_0 = const()[name = tensor<string, []>("x2_43_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_43_end_0 = const()[name = tensor<string, []>("x2_43_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_43_end_mask_0 = const()[name = tensor<string, []>("x2_43_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_43_cast_fp16 = slice_by_index(begin = x2_43_begin_0, end = x2_43_end_0, end_mask = x2_43_end_mask_0, x = k_41_cast_fp16)[name = tensor<string, []>("x2_43_cast_fp16")];
tensor<fp16, []> const_87_promoted_to_fp16 = const()[name = tensor<string, []>("const_87_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2156_cast_fp16 = mul(x = x2_43_cast_fp16, y = const_87_promoted_to_fp16)[name = tensor<string, []>("op_2156_cast_fp16")];
tensor<bool, []> var_2158_interleave_0 = const()[name = tensor<string, []>("op_2158_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2158_cast_fp16 = concat(axis = var_2096, interleave = var_2158_interleave_0, values = (var_2156_cast_fp16, x1_43_cast_fp16))[name = tensor<string, []>("op_2158_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2159_cast_fp16 = mul(x = var_2158_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2159_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_21_cast_fp16 = add(x = var_2144_cast_fp16, y = var_2159_cast_fp16)[name = tensor<string, []>("k_embed_21_cast_fp16")];
tensor<bool, []> var_2164_transpose_x_1 = const()[name = tensor<string, []>("op_2164_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_2164_transpose_y_1 = const()[name = tensor<string, []>("op_2164_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_2164_cast_fp16 = matmul(transpose_x = var_2164_transpose_x_1, transpose_y = var_2164_transpose_y_1, x = q_embed_21_cast_fp16, y = k_embed_21_cast_fp16)[name = tensor<string, []>("op_2164_cast_fp16")];
tensor<fp16, []> var_2165_to_fp16 = const()[name = tensor<string, []>("op_2165_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_41_cast_fp16 = mul(x = var_2164_cast_fp16, y = var_2165_to_fp16)[name = tensor<string, []>("attn_weights_41_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_187_cast_fp16 = add(x = attn_weights_41_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_187_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_2168_cast_fp16 = softmax(axis = var_2096, x = input_187_cast_fp16)[name = tensor<string, []>("op_2168_cast_fp16")];
tensor<bool, []> attn_output_61_transpose_x_0 = const()[name = tensor<string, []>("attn_output_61_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_61_transpose_y_0 = const()[name = tensor<string, []>("attn_output_61_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_21_cast_fp16 = transpose(perm = value_21_perm_0, x = squeeze_32_cast_fp16)[name = tensor<string, []>("transpose_64")];
tensor<fp16, [1, 12, 256, 64]> attn_output_61_cast_fp16 = matmul(transpose_x = attn_output_61_transpose_x_0, transpose_y = attn_output_61_transpose_y_0, x = var_2168_cast_fp16, y = value_21_cast_fp16)[name = tensor<string, []>("attn_output_61_cast_fp16")];
tensor<int32, [4]> var_2172_perm_0 = const()[name = tensor<string, []>("op_2172_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_2174 = const()[name = tensor<string, []>("op_2174"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_2172_cast_fp16 = transpose(perm = var_2172_perm_0, x = attn_output_61_cast_fp16)[name = tensor<string, []>("transpose_63")];
tensor<fp16, [1, 256, 768]> var_2175_cast_fp16 = reshape(shape = var_2174, x = var_2172_cast_fp16)[name = tensor<string, []>("op_2175_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_10_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_10_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(497395584)))];
tensor<fp16, [1, 256, 768]> linear_41_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_10_attn_Wo_weight_to_fp16, x = var_2175_cast_fp16)[name = tensor<string, []>("linear_41_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_193_cast_fp16 = add(x = input_185_cast_fp16, y = linear_41_cast_fp16)[name = tensor<string, []>("input_193_cast_fp16")];
tensor<int32, [1]> input_195_axes_0 = const()[name = tensor<string, []>("input_195_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_10_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_10_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(498575296)))];
tensor<fp16, [1, 256, 768]> input_195_cast_fp16 = layer_norm(axes = input_195_axes_0, epsilon = var_2107_to_fp16, gamma = model_encoder_layers_10_mlp_norm_weight_to_fp16, x = input_193_cast_fp16)[name = tensor<string, []>("input_195_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_10_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_10_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(498576896)))];
tensor<fp16, [1, 256, 2304]> linear_42_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_10_mlp_Wi_weight_to_fp16, x = input_195_cast_fp16)[name = tensor<string, []>("linear_42_cast_fp16")];
tensor<int32, [2]> var_2182_split_sizes_0 = const()[name = tensor<string, []>("op_2182_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_2182_axis_0 = const()[name = tensor<string, []>("op_2182_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_2182_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_2182_cast_fp16_1 = split(axis = var_2182_axis_0, split_sizes = var_2182_split_sizes_0, x = linear_42_cast_fp16)[name = tensor<string, []>("op_2182_cast_fp16")];
tensor<string, []> var_2184_mode_0 = const()[name = tensor<string, []>("op_2184_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_2184_cast_fp16 = gelu(mode = var_2184_mode_0, x = var_2182_cast_fp16_0)[name = tensor<string, []>("op_2184_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_199_cast_fp16 = mul(x = var_2184_cast_fp16, y = var_2182_cast_fp16_1)[name = tensor<string, []>("input_199_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_10_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_10_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(502115904)))];
tensor<fp16, [1, 256, 768]> linear_43_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_10_mlp_Wo_weight_to_fp16, x = input_199_cast_fp16)[name = tensor<string, []>("linear_43_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_203_cast_fp16 = add(x = input_193_cast_fp16, y = linear_43_cast_fp16)[name = tensor<string, []>("input_203_cast_fp16")];
tensor<int32, []> var_2193 = const()[name = tensor<string, []>("op_2193"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_23_axes_0 = const()[name = tensor<string, []>("hidden_states_23_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_11_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_11_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(503885440)))];
tensor<fp16, []> var_2204_to_fp16 = const()[name = tensor<string, []>("op_2204_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_23_cast_fp16 = layer_norm(axes = hidden_states_23_axes_0, epsilon = var_2204_to_fp16, gamma = model_encoder_layers_11_attn_norm_weight_to_fp16, x = input_203_cast_fp16)[name = tensor<string, []>("hidden_states_23_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_11_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_11_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(503887040)))];
tensor<fp16, [1, 256, 2304]> linear_44_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_11_attn_Wqkv_weight_to_fp16, x = hidden_states_23_cast_fp16)[name = tensor<string, []>("linear_44_cast_fp16")];
tensor<int32, [5]> var_2211 = const()[name = tensor<string, []>("op_2211"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_47_cast_fp16 = reshape(shape = var_2211, x = linear_44_cast_fp16)[name = tensor<string, []>("qkv_47_cast_fp16")];
tensor<int32, [3]> var_2213_split_sizes_0 = const()[name = tensor<string, []>("op_2213_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_2213_axis_0 = const()[name = tensor<string, []>("op_2213_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_2213_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_2213_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_2213_cast_fp16_2 = split(axis = var_2213_axis_0, split_sizes = var_2213_split_sizes_0, x = qkv_47_cast_fp16)[name = tensor<string, []>("op_2213_cast_fp16")];
tensor<int32, [1]> squeeze_33_axes_0 = const()[name = tensor<string, []>("squeeze_33_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_33_cast_fp16 = squeeze(axes = squeeze_33_axes_0, x = var_2213_cast_fp16_0)[name = tensor<string, []>("squeeze_33_cast_fp16")];
tensor<int32, [1]> squeeze_34_axes_0 = const()[name = tensor<string, []>("squeeze_34_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_34_cast_fp16 = squeeze(axes = squeeze_34_axes_0, x = var_2213_cast_fp16_1)[name = tensor<string, []>("squeeze_34_cast_fp16")];
tensor<int32, [1]> squeeze_35_axes_0 = const()[name = tensor<string, []>("squeeze_35_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_35_cast_fp16 = squeeze(axes = squeeze_35_axes_0, x = var_2213_cast_fp16_2)[name = tensor<string, []>("squeeze_35_cast_fp16")];
tensor<int32, [4]> q_45_perm_0 = const()[name = tensor<string, []>("q_45_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_45_perm_0 = const()[name = tensor<string, []>("k_45_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_23_perm_0 = const()[name = tensor<string, []>("value_23_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_45_cast_fp16 = transpose(perm = q_45_perm_0, x = squeeze_33_cast_fp16)[name = tensor<string, []>("transpose_62")];
tensor<fp16, [1, 12, 256, 64]> var_2223_cast_fp16 = mul(x = q_45_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2223_cast_fp16")];
tensor<int32, [4]> x1_45_begin_0 = const()[name = tensor<string, []>("x1_45_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_45_end_0 = const()[name = tensor<string, []>("x1_45_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_45_end_mask_0 = const()[name = tensor<string, []>("x1_45_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_45_cast_fp16 = slice_by_index(begin = x1_45_begin_0, end = x1_45_end_0, end_mask = x1_45_end_mask_0, x = q_45_cast_fp16)[name = tensor<string, []>("x1_45_cast_fp16")];
tensor<int32, [4]> x2_45_begin_0 = const()[name = tensor<string, []>("x2_45_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_45_end_0 = const()[name = tensor<string, []>("x2_45_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_45_end_mask_0 = const()[name = tensor<string, []>("x2_45_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_45_cast_fp16 = slice_by_index(begin = x2_45_begin_0, end = x2_45_end_0, end_mask = x2_45_end_mask_0, x = q_45_cast_fp16)[name = tensor<string, []>("x2_45_cast_fp16")];
tensor<fp16, []> const_92_promoted_to_fp16 = const()[name = tensor<string, []>("const_92_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2235_cast_fp16 = mul(x = x2_45_cast_fp16, y = const_92_promoted_to_fp16)[name = tensor<string, []>("op_2235_cast_fp16")];
tensor<bool, []> var_2237_interleave_0 = const()[name = tensor<string, []>("op_2237_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2237_cast_fp16 = concat(axis = var_2193, interleave = var_2237_interleave_0, values = (var_2235_cast_fp16, x1_45_cast_fp16))[name = tensor<string, []>("op_2237_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2238_cast_fp16 = mul(x = var_2237_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2238_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_23_cast_fp16 = add(x = var_2223_cast_fp16, y = var_2238_cast_fp16)[name = tensor<string, []>("q_embed_23_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_45_cast_fp16 = transpose(perm = k_45_perm_0, x = squeeze_34_cast_fp16)[name = tensor<string, []>("transpose_61")];
tensor<fp16, [1, 12, 256, 64]> var_2241_cast_fp16 = mul(x = k_45_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2241_cast_fp16")];
tensor<int32, [4]> x1_47_begin_0 = const()[name = tensor<string, []>("x1_47_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_47_end_0 = const()[name = tensor<string, []>("x1_47_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_47_end_mask_0 = const()[name = tensor<string, []>("x1_47_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_47_cast_fp16 = slice_by_index(begin = x1_47_begin_0, end = x1_47_end_0, end_mask = x1_47_end_mask_0, x = k_45_cast_fp16)[name = tensor<string, []>("x1_47_cast_fp16")];
tensor<int32, [4]> x2_47_begin_0 = const()[name = tensor<string, []>("x2_47_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_47_end_0 = const()[name = tensor<string, []>("x2_47_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_47_end_mask_0 = const()[name = tensor<string, []>("x2_47_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_47_cast_fp16 = slice_by_index(begin = x2_47_begin_0, end = x2_47_end_0, end_mask = x2_47_end_mask_0, x = k_45_cast_fp16)[name = tensor<string, []>("x2_47_cast_fp16")];
tensor<fp16, []> const_95_promoted_to_fp16 = const()[name = tensor<string, []>("const_95_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2253_cast_fp16 = mul(x = x2_47_cast_fp16, y = const_95_promoted_to_fp16)[name = tensor<string, []>("op_2253_cast_fp16")];
tensor<bool, []> var_2255_interleave_0 = const()[name = tensor<string, []>("op_2255_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2255_cast_fp16 = concat(axis = var_2193, interleave = var_2255_interleave_0, values = (var_2253_cast_fp16, x1_47_cast_fp16))[name = tensor<string, []>("op_2255_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2256_cast_fp16 = mul(x = var_2255_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2256_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_23_cast_fp16 = add(x = var_2241_cast_fp16, y = var_2256_cast_fp16)[name = tensor<string, []>("k_embed_23_cast_fp16")];
tensor<bool, []> var_2261_transpose_x_1 = const()[name = tensor<string, []>("op_2261_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_2261_transpose_y_1 = const()[name = tensor<string, []>("op_2261_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_2261_cast_fp16 = matmul(transpose_x = var_2261_transpose_x_1, transpose_y = var_2261_transpose_y_1, x = q_embed_23_cast_fp16, y = k_embed_23_cast_fp16)[name = tensor<string, []>("op_2261_cast_fp16")];
tensor<fp16, []> var_2262_to_fp16 = const()[name = tensor<string, []>("op_2262_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_45_cast_fp16 = mul(x = var_2261_cast_fp16, y = var_2262_to_fp16)[name = tensor<string, []>("attn_weights_45_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_205_cast_fp16 = add(x = attn_weights_45_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_205_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_2265_cast_fp16 = softmax(axis = var_2193, x = input_205_cast_fp16)[name = tensor<string, []>("op_2265_cast_fp16")];
tensor<bool, []> attn_output_67_transpose_x_0 = const()[name = tensor<string, []>("attn_output_67_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_67_transpose_y_0 = const()[name = tensor<string, []>("attn_output_67_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_23_cast_fp16 = transpose(perm = value_23_perm_0, x = squeeze_35_cast_fp16)[name = tensor<string, []>("transpose_60")];
tensor<fp16, [1, 12, 256, 64]> attn_output_67_cast_fp16 = matmul(transpose_x = attn_output_67_transpose_x_0, transpose_y = attn_output_67_transpose_y_0, x = var_2265_cast_fp16, y = value_23_cast_fp16)[name = tensor<string, []>("attn_output_67_cast_fp16")];
tensor<int32, [4]> var_2269_perm_0 = const()[name = tensor<string, []>("op_2269_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_2271 = const()[name = tensor<string, []>("op_2271"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_2269_cast_fp16 = transpose(perm = var_2269_perm_0, x = attn_output_67_cast_fp16)[name = tensor<string, []>("transpose_59")];
tensor<fp16, [1, 256, 768]> var_2272_cast_fp16 = reshape(shape = var_2271, x = var_2269_cast_fp16)[name = tensor<string, []>("op_2272_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_11_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_11_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(507426048)))];
tensor<fp16, [1, 256, 768]> linear_45_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_11_attn_Wo_weight_to_fp16, x = var_2272_cast_fp16)[name = tensor<string, []>("linear_45_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_211_cast_fp16 = add(x = input_203_cast_fp16, y = linear_45_cast_fp16)[name = tensor<string, []>("input_211_cast_fp16")];
tensor<int32, [1]> input_213_axes_0 = const()[name = tensor<string, []>("input_213_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_11_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_11_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(508605760)))];
tensor<fp16, [1, 256, 768]> input_213_cast_fp16 = layer_norm(axes = input_213_axes_0, epsilon = var_2204_to_fp16, gamma = model_encoder_layers_11_mlp_norm_weight_to_fp16, x = input_211_cast_fp16)[name = tensor<string, []>("input_213_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_11_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_11_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(508607360)))];
tensor<fp16, [1, 256, 2304]> linear_46_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_11_mlp_Wi_weight_to_fp16, x = input_213_cast_fp16)[name = tensor<string, []>("linear_46_cast_fp16")];
tensor<int32, [2]> var_2279_split_sizes_0 = const()[name = tensor<string, []>("op_2279_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_2279_axis_0 = const()[name = tensor<string, []>("op_2279_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_2279_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_2279_cast_fp16_1 = split(axis = var_2279_axis_0, split_sizes = var_2279_split_sizes_0, x = linear_46_cast_fp16)[name = tensor<string, []>("op_2279_cast_fp16")];
tensor<string, []> var_2281_mode_0 = const()[name = tensor<string, []>("op_2281_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_2281_cast_fp16 = gelu(mode = var_2281_mode_0, x = var_2279_cast_fp16_0)[name = tensor<string, []>("op_2281_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_217_cast_fp16 = mul(x = var_2281_cast_fp16, y = var_2279_cast_fp16_1)[name = tensor<string, []>("input_217_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_11_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_11_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(512146368)))];
tensor<fp16, [1, 256, 768]> linear_47_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_11_mlp_Wo_weight_to_fp16, x = input_217_cast_fp16)[name = tensor<string, []>("linear_47_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_221_cast_fp16 = add(x = input_211_cast_fp16, y = linear_47_cast_fp16)[name = tensor<string, []>("input_221_cast_fp16")];
tensor<int32, []> var_2290 = const()[name = tensor<string, []>("op_2290"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_25_axes_0 = const()[name = tensor<string, []>("hidden_states_25_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_12_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_12_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(513915904)))];
tensor<fp16, []> var_2301_to_fp16 = const()[name = tensor<string, []>("op_2301_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_25_cast_fp16 = layer_norm(axes = hidden_states_25_axes_0, epsilon = var_2301_to_fp16, gamma = model_encoder_layers_12_attn_norm_weight_to_fp16, x = input_221_cast_fp16)[name = tensor<string, []>("hidden_states_25_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_12_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_12_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(513917504)))];
tensor<fp16, [1, 256, 2304]> linear_48_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_12_attn_Wqkv_weight_to_fp16, x = hidden_states_25_cast_fp16)[name = tensor<string, []>("linear_48_cast_fp16")];
tensor<int32, [5]> var_2308 = const()[name = tensor<string, []>("op_2308"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_51_cast_fp16 = reshape(shape = var_2308, x = linear_48_cast_fp16)[name = tensor<string, []>("qkv_51_cast_fp16")];
tensor<int32, [3]> var_2310_split_sizes_0 = const()[name = tensor<string, []>("op_2310_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_2310_axis_0 = const()[name = tensor<string, []>("op_2310_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_2310_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_2310_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_2310_cast_fp16_2 = split(axis = var_2310_axis_0, split_sizes = var_2310_split_sizes_0, x = qkv_51_cast_fp16)[name = tensor<string, []>("op_2310_cast_fp16")];
tensor<int32, [1]> squeeze_36_axes_0 = const()[name = tensor<string, []>("squeeze_36_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_36_cast_fp16 = squeeze(axes = squeeze_36_axes_0, x = var_2310_cast_fp16_0)[name = tensor<string, []>("squeeze_36_cast_fp16")];
tensor<int32, [1]> squeeze_37_axes_0 = const()[name = tensor<string, []>("squeeze_37_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_37_cast_fp16 = squeeze(axes = squeeze_37_axes_0, x = var_2310_cast_fp16_1)[name = tensor<string, []>("squeeze_37_cast_fp16")];
tensor<int32, [1]> squeeze_38_axes_0 = const()[name = tensor<string, []>("squeeze_38_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_38_cast_fp16 = squeeze(axes = squeeze_38_axes_0, x = var_2310_cast_fp16_2)[name = tensor<string, []>("squeeze_38_cast_fp16")];
tensor<int32, [4]> q_49_perm_0 = const()[name = tensor<string, []>("q_49_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_49_perm_0 = const()[name = tensor<string, []>("k_49_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_25_perm_0 = const()[name = tensor<string, []>("value_25_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_49_cast_fp16 = transpose(perm = q_49_perm_0, x = squeeze_36_cast_fp16)[name = tensor<string, []>("transpose_58")];
tensor<fp16, [1, 12, 256, 64]> var_2320_cast_fp16 = mul(x = q_49_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2320_cast_fp16")];
tensor<int32, [4]> x1_49_begin_0 = const()[name = tensor<string, []>("x1_49_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_49_end_0 = const()[name = tensor<string, []>("x1_49_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_49_end_mask_0 = const()[name = tensor<string, []>("x1_49_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_49_cast_fp16 = slice_by_index(begin = x1_49_begin_0, end = x1_49_end_0, end_mask = x1_49_end_mask_0, x = q_49_cast_fp16)[name = tensor<string, []>("x1_49_cast_fp16")];
tensor<int32, [4]> x2_49_begin_0 = const()[name = tensor<string, []>("x2_49_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_49_end_0 = const()[name = tensor<string, []>("x2_49_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_49_end_mask_0 = const()[name = tensor<string, []>("x2_49_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_49_cast_fp16 = slice_by_index(begin = x2_49_begin_0, end = x2_49_end_0, end_mask = x2_49_end_mask_0, x = q_49_cast_fp16)[name = tensor<string, []>("x2_49_cast_fp16")];
tensor<fp16, []> const_100_promoted_to_fp16 = const()[name = tensor<string, []>("const_100_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2332_cast_fp16 = mul(x = x2_49_cast_fp16, y = const_100_promoted_to_fp16)[name = tensor<string, []>("op_2332_cast_fp16")];
tensor<bool, []> var_2334_interleave_0 = const()[name = tensor<string, []>("op_2334_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2334_cast_fp16 = concat(axis = var_2290, interleave = var_2334_interleave_0, values = (var_2332_cast_fp16, x1_49_cast_fp16))[name = tensor<string, []>("op_2334_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2335_cast_fp16 = mul(x = var_2334_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2335_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_25_cast_fp16 = add(x = var_2320_cast_fp16, y = var_2335_cast_fp16)[name = tensor<string, []>("q_embed_25_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_49_cast_fp16 = transpose(perm = k_49_perm_0, x = squeeze_37_cast_fp16)[name = tensor<string, []>("transpose_57")];
tensor<fp16, [1, 12, 256, 64]> var_2338_cast_fp16 = mul(x = k_49_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2338_cast_fp16")];
tensor<int32, [4]> x1_51_begin_0 = const()[name = tensor<string, []>("x1_51_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_51_end_0 = const()[name = tensor<string, []>("x1_51_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_51_end_mask_0 = const()[name = tensor<string, []>("x1_51_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_51_cast_fp16 = slice_by_index(begin = x1_51_begin_0, end = x1_51_end_0, end_mask = x1_51_end_mask_0, x = k_49_cast_fp16)[name = tensor<string, []>("x1_51_cast_fp16")];
tensor<int32, [4]> x2_51_begin_0 = const()[name = tensor<string, []>("x2_51_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_51_end_0 = const()[name = tensor<string, []>("x2_51_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_51_end_mask_0 = const()[name = tensor<string, []>("x2_51_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_51_cast_fp16 = slice_by_index(begin = x2_51_begin_0, end = x2_51_end_0, end_mask = x2_51_end_mask_0, x = k_49_cast_fp16)[name = tensor<string, []>("x2_51_cast_fp16")];
tensor<fp16, []> const_103_promoted_to_fp16 = const()[name = tensor<string, []>("const_103_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2350_cast_fp16 = mul(x = x2_51_cast_fp16, y = const_103_promoted_to_fp16)[name = tensor<string, []>("op_2350_cast_fp16")];
tensor<bool, []> var_2352_interleave_0 = const()[name = tensor<string, []>("op_2352_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2352_cast_fp16 = concat(axis = var_2290, interleave = var_2352_interleave_0, values = (var_2350_cast_fp16, x1_51_cast_fp16))[name = tensor<string, []>("op_2352_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2353_cast_fp16 = mul(x = var_2352_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2353_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_25_cast_fp16 = add(x = var_2338_cast_fp16, y = var_2353_cast_fp16)[name = tensor<string, []>("k_embed_25_cast_fp16")];
tensor<bool, []> var_2358_transpose_x_1 = const()[name = tensor<string, []>("op_2358_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_2358_transpose_y_1 = const()[name = tensor<string, []>("op_2358_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_2358_cast_fp16 = matmul(transpose_x = var_2358_transpose_x_1, transpose_y = var_2358_transpose_y_1, x = q_embed_25_cast_fp16, y = k_embed_25_cast_fp16)[name = tensor<string, []>("op_2358_cast_fp16")];
tensor<fp16, []> var_2359_to_fp16 = const()[name = tensor<string, []>("op_2359_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_49_cast_fp16 = mul(x = var_2358_cast_fp16, y = var_2359_to_fp16)[name = tensor<string, []>("attn_weights_49_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_223_cast_fp16 = add(x = attn_weights_49_cast_fp16, y = attention_mask_3_cast_fp16)[name = tensor<string, []>("input_223_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_2362_cast_fp16 = softmax(axis = var_2290, x = input_223_cast_fp16)[name = tensor<string, []>("op_2362_cast_fp16")];
tensor<bool, []> attn_output_73_transpose_x_0 = const()[name = tensor<string, []>("attn_output_73_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_73_transpose_y_0 = const()[name = tensor<string, []>("attn_output_73_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_25_cast_fp16 = transpose(perm = value_25_perm_0, x = squeeze_38_cast_fp16)[name = tensor<string, []>("transpose_56")];
tensor<fp16, [1, 12, 256, 64]> attn_output_73_cast_fp16 = matmul(transpose_x = attn_output_73_transpose_x_0, transpose_y = attn_output_73_transpose_y_0, x = var_2362_cast_fp16, y = value_25_cast_fp16)[name = tensor<string, []>("attn_output_73_cast_fp16")];
tensor<int32, [4]> var_2366_perm_0 = const()[name = tensor<string, []>("op_2366_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_2368 = const()[name = tensor<string, []>("op_2368"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_2366_cast_fp16 = transpose(perm = var_2366_perm_0, x = attn_output_73_cast_fp16)[name = tensor<string, []>("transpose_55")];
tensor<fp16, [1, 256, 768]> var_2369_cast_fp16 = reshape(shape = var_2368, x = var_2366_cast_fp16)[name = tensor<string, []>("op_2369_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_12_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_12_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(517456512)))];
tensor<fp16, [1, 256, 768]> linear_49_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_12_attn_Wo_weight_to_fp16, x = var_2369_cast_fp16)[name = tensor<string, []>("linear_49_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_229_cast_fp16 = add(x = input_221_cast_fp16, y = linear_49_cast_fp16)[name = tensor<string, []>("input_229_cast_fp16")];
tensor<int32, [1]> input_231_axes_0 = const()[name = tensor<string, []>("input_231_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_12_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_12_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(518636224)))];
tensor<fp16, [1, 256, 768]> input_231_cast_fp16 = layer_norm(axes = input_231_axes_0, epsilon = var_2301_to_fp16, gamma = model_encoder_layers_12_mlp_norm_weight_to_fp16, x = input_229_cast_fp16)[name = tensor<string, []>("input_231_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_12_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_12_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(518637824)))];
tensor<fp16, [1, 256, 2304]> linear_50_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_12_mlp_Wi_weight_to_fp16, x = input_231_cast_fp16)[name = tensor<string, []>("linear_50_cast_fp16")];
tensor<int32, [2]> var_2376_split_sizes_0 = const()[name = tensor<string, []>("op_2376_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_2376_axis_0 = const()[name = tensor<string, []>("op_2376_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_2376_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_2376_cast_fp16_1 = split(axis = var_2376_axis_0, split_sizes = var_2376_split_sizes_0, x = linear_50_cast_fp16)[name = tensor<string, []>("op_2376_cast_fp16")];
tensor<string, []> var_2378_mode_0 = const()[name = tensor<string, []>("op_2378_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_2378_cast_fp16 = gelu(mode = var_2378_mode_0, x = var_2376_cast_fp16_0)[name = tensor<string, []>("op_2378_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_235_cast_fp16 = mul(x = var_2378_cast_fp16, y = var_2376_cast_fp16_1)[name = tensor<string, []>("input_235_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_12_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_12_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(522176832)))];
tensor<fp16, [1, 256, 768]> linear_51_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_12_mlp_Wo_weight_to_fp16, x = input_235_cast_fp16)[name = tensor<string, []>("linear_51_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_239_cast_fp16 = add(x = input_229_cast_fp16, y = linear_51_cast_fp16)[name = tensor<string, []>("input_239_cast_fp16")];
tensor<int32, []> var_2387 = const()[name = tensor<string, []>("op_2387"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_27_axes_0 = const()[name = tensor<string, []>("hidden_states_27_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_13_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_13_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(523946368)))];
tensor<fp16, []> var_2398_to_fp16 = const()[name = tensor<string, []>("op_2398_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_27_cast_fp16 = layer_norm(axes = hidden_states_27_axes_0, epsilon = var_2398_to_fp16, gamma = model_encoder_layers_13_attn_norm_weight_to_fp16, x = input_239_cast_fp16)[name = tensor<string, []>("hidden_states_27_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_13_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_13_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(523947968)))];
tensor<fp16, [1, 256, 2304]> linear_52_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_13_attn_Wqkv_weight_to_fp16, x = hidden_states_27_cast_fp16)[name = tensor<string, []>("linear_52_cast_fp16")];
tensor<int32, [5]> var_2405 = const()[name = tensor<string, []>("op_2405"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_55_cast_fp16 = reshape(shape = var_2405, x = linear_52_cast_fp16)[name = tensor<string, []>("qkv_55_cast_fp16")];
tensor<int32, [3]> var_2407_split_sizes_0 = const()[name = tensor<string, []>("op_2407_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_2407_axis_0 = const()[name = tensor<string, []>("op_2407_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_2407_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_2407_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_2407_cast_fp16_2 = split(axis = var_2407_axis_0, split_sizes = var_2407_split_sizes_0, x = qkv_55_cast_fp16)[name = tensor<string, []>("op_2407_cast_fp16")];
tensor<int32, [1]> squeeze_39_axes_0 = const()[name = tensor<string, []>("squeeze_39_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_39_cast_fp16 = squeeze(axes = squeeze_39_axes_0, x = var_2407_cast_fp16_0)[name = tensor<string, []>("squeeze_39_cast_fp16")];
tensor<int32, [1]> squeeze_40_axes_0 = const()[name = tensor<string, []>("squeeze_40_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_40_cast_fp16 = squeeze(axes = squeeze_40_axes_0, x = var_2407_cast_fp16_1)[name = tensor<string, []>("squeeze_40_cast_fp16")];
tensor<int32, [1]> squeeze_41_axes_0 = const()[name = tensor<string, []>("squeeze_41_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_41_cast_fp16 = squeeze(axes = squeeze_41_axes_0, x = var_2407_cast_fp16_2)[name = tensor<string, []>("squeeze_41_cast_fp16")];
tensor<int32, [4]> q_53_perm_0 = const()[name = tensor<string, []>("q_53_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_53_perm_0 = const()[name = tensor<string, []>("k_53_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_27_perm_0 = const()[name = tensor<string, []>("value_27_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_53_cast_fp16 = transpose(perm = q_53_perm_0, x = squeeze_39_cast_fp16)[name = tensor<string, []>("transpose_54")];
tensor<fp16, [1, 12, 256, 64]> var_2417_cast_fp16 = mul(x = q_53_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2417_cast_fp16")];
tensor<int32, [4]> x1_53_begin_0 = const()[name = tensor<string, []>("x1_53_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_53_end_0 = const()[name = tensor<string, []>("x1_53_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_53_end_mask_0 = const()[name = tensor<string, []>("x1_53_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_53_cast_fp16 = slice_by_index(begin = x1_53_begin_0, end = x1_53_end_0, end_mask = x1_53_end_mask_0, x = q_53_cast_fp16)[name = tensor<string, []>("x1_53_cast_fp16")];
tensor<int32, [4]> x2_53_begin_0 = const()[name = tensor<string, []>("x2_53_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_53_end_0 = const()[name = tensor<string, []>("x2_53_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_53_end_mask_0 = const()[name = tensor<string, []>("x2_53_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_53_cast_fp16 = slice_by_index(begin = x2_53_begin_0, end = x2_53_end_0, end_mask = x2_53_end_mask_0, x = q_53_cast_fp16)[name = tensor<string, []>("x2_53_cast_fp16")];
tensor<fp16, []> const_108_promoted_to_fp16 = const()[name = tensor<string, []>("const_108_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2429_cast_fp16 = mul(x = x2_53_cast_fp16, y = const_108_promoted_to_fp16)[name = tensor<string, []>("op_2429_cast_fp16")];
tensor<bool, []> var_2431_interleave_0 = const()[name = tensor<string, []>("op_2431_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2431_cast_fp16 = concat(axis = var_2387, interleave = var_2431_interleave_0, values = (var_2429_cast_fp16, x1_53_cast_fp16))[name = tensor<string, []>("op_2431_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2432_cast_fp16 = mul(x = var_2431_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2432_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_27_cast_fp16 = add(x = var_2417_cast_fp16, y = var_2432_cast_fp16)[name = tensor<string, []>("q_embed_27_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_53_cast_fp16 = transpose(perm = k_53_perm_0, x = squeeze_40_cast_fp16)[name = tensor<string, []>("transpose_53")];
tensor<fp16, [1, 12, 256, 64]> var_2435_cast_fp16 = mul(x = k_53_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2435_cast_fp16")];
tensor<int32, [4]> x1_55_begin_0 = const()[name = tensor<string, []>("x1_55_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_55_end_0 = const()[name = tensor<string, []>("x1_55_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_55_end_mask_0 = const()[name = tensor<string, []>("x1_55_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_55_cast_fp16 = slice_by_index(begin = x1_55_begin_0, end = x1_55_end_0, end_mask = x1_55_end_mask_0, x = k_53_cast_fp16)[name = tensor<string, []>("x1_55_cast_fp16")];
tensor<int32, [4]> x2_55_begin_0 = const()[name = tensor<string, []>("x2_55_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_55_end_0 = const()[name = tensor<string, []>("x2_55_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_55_end_mask_0 = const()[name = tensor<string, []>("x2_55_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_55_cast_fp16 = slice_by_index(begin = x2_55_begin_0, end = x2_55_end_0, end_mask = x2_55_end_mask_0, x = k_53_cast_fp16)[name = tensor<string, []>("x2_55_cast_fp16")];
tensor<fp16, []> const_111_promoted_to_fp16 = const()[name = tensor<string, []>("const_111_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2447_cast_fp16 = mul(x = x2_55_cast_fp16, y = const_111_promoted_to_fp16)[name = tensor<string, []>("op_2447_cast_fp16")];
tensor<bool, []> var_2449_interleave_0 = const()[name = tensor<string, []>("op_2449_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2449_cast_fp16 = concat(axis = var_2387, interleave = var_2449_interleave_0, values = (var_2447_cast_fp16, x1_55_cast_fp16))[name = tensor<string, []>("op_2449_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2450_cast_fp16 = mul(x = var_2449_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2450_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_27_cast_fp16 = add(x = var_2435_cast_fp16, y = var_2450_cast_fp16)[name = tensor<string, []>("k_embed_27_cast_fp16")];
tensor<bool, []> var_2455_transpose_x_1 = const()[name = tensor<string, []>("op_2455_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_2455_transpose_y_1 = const()[name = tensor<string, []>("op_2455_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_2455_cast_fp16 = matmul(transpose_x = var_2455_transpose_x_1, transpose_y = var_2455_transpose_y_1, x = q_embed_27_cast_fp16, y = k_embed_27_cast_fp16)[name = tensor<string, []>("op_2455_cast_fp16")];
tensor<fp16, []> var_2456_to_fp16 = const()[name = tensor<string, []>("op_2456_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_53_cast_fp16 = mul(x = var_2455_cast_fp16, y = var_2456_to_fp16)[name = tensor<string, []>("attn_weights_53_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_241_cast_fp16 = add(x = attn_weights_53_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_241_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_2459_cast_fp16 = softmax(axis = var_2387, x = input_241_cast_fp16)[name = tensor<string, []>("op_2459_cast_fp16")];
tensor<bool, []> attn_output_79_transpose_x_0 = const()[name = tensor<string, []>("attn_output_79_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_79_transpose_y_0 = const()[name = tensor<string, []>("attn_output_79_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_27_cast_fp16 = transpose(perm = value_27_perm_0, x = squeeze_41_cast_fp16)[name = tensor<string, []>("transpose_52")];
tensor<fp16, [1, 12, 256, 64]> attn_output_79_cast_fp16 = matmul(transpose_x = attn_output_79_transpose_x_0, transpose_y = attn_output_79_transpose_y_0, x = var_2459_cast_fp16, y = value_27_cast_fp16)[name = tensor<string, []>("attn_output_79_cast_fp16")];
tensor<int32, [4]> var_2463_perm_0 = const()[name = tensor<string, []>("op_2463_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_2465 = const()[name = tensor<string, []>("op_2465"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_2463_cast_fp16 = transpose(perm = var_2463_perm_0, x = attn_output_79_cast_fp16)[name = tensor<string, []>("transpose_51")];
tensor<fp16, [1, 256, 768]> var_2466_cast_fp16 = reshape(shape = var_2465, x = var_2463_cast_fp16)[name = tensor<string, []>("op_2466_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_13_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_13_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(527486976)))];
tensor<fp16, [1, 256, 768]> linear_53_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_13_attn_Wo_weight_to_fp16, x = var_2466_cast_fp16)[name = tensor<string, []>("linear_53_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_247_cast_fp16 = add(x = input_239_cast_fp16, y = linear_53_cast_fp16)[name = tensor<string, []>("input_247_cast_fp16")];
tensor<int32, [1]> input_249_axes_0 = const()[name = tensor<string, []>("input_249_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_13_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_13_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(528666688)))];
tensor<fp16, [1, 256, 768]> input_249_cast_fp16 = layer_norm(axes = input_249_axes_0, epsilon = var_2398_to_fp16, gamma = model_encoder_layers_13_mlp_norm_weight_to_fp16, x = input_247_cast_fp16)[name = tensor<string, []>("input_249_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_13_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_13_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(528668288)))];
tensor<fp16, [1, 256, 2304]> linear_54_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_13_mlp_Wi_weight_to_fp16, x = input_249_cast_fp16)[name = tensor<string, []>("linear_54_cast_fp16")];
tensor<int32, [2]> var_2473_split_sizes_0 = const()[name = tensor<string, []>("op_2473_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_2473_axis_0 = const()[name = tensor<string, []>("op_2473_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_2473_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_2473_cast_fp16_1 = split(axis = var_2473_axis_0, split_sizes = var_2473_split_sizes_0, x = linear_54_cast_fp16)[name = tensor<string, []>("op_2473_cast_fp16")];
tensor<string, []> var_2475_mode_0 = const()[name = tensor<string, []>("op_2475_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_2475_cast_fp16 = gelu(mode = var_2475_mode_0, x = var_2473_cast_fp16_0)[name = tensor<string, []>("op_2475_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_253_cast_fp16 = mul(x = var_2475_cast_fp16, y = var_2473_cast_fp16_1)[name = tensor<string, []>("input_253_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_13_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_13_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(532207296)))];
tensor<fp16, [1, 256, 768]> linear_55_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_13_mlp_Wo_weight_to_fp16, x = input_253_cast_fp16)[name = tensor<string, []>("linear_55_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_257_cast_fp16 = add(x = input_247_cast_fp16, y = linear_55_cast_fp16)[name = tensor<string, []>("input_257_cast_fp16")];
tensor<int32, []> var_2484 = const()[name = tensor<string, []>("op_2484"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_29_axes_0 = const()[name = tensor<string, []>("hidden_states_29_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_14_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_14_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(533976832)))];
tensor<fp16, []> var_2495_to_fp16 = const()[name = tensor<string, []>("op_2495_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_29_cast_fp16 = layer_norm(axes = hidden_states_29_axes_0, epsilon = var_2495_to_fp16, gamma = model_encoder_layers_14_attn_norm_weight_to_fp16, x = input_257_cast_fp16)[name = tensor<string, []>("hidden_states_29_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_14_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_14_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(533978432)))];
tensor<fp16, [1, 256, 2304]> linear_56_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_14_attn_Wqkv_weight_to_fp16, x = hidden_states_29_cast_fp16)[name = tensor<string, []>("linear_56_cast_fp16")];
tensor<int32, [5]> var_2502 = const()[name = tensor<string, []>("op_2502"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_59_cast_fp16 = reshape(shape = var_2502, x = linear_56_cast_fp16)[name = tensor<string, []>("qkv_59_cast_fp16")];
tensor<int32, [3]> var_2504_split_sizes_0 = const()[name = tensor<string, []>("op_2504_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_2504_axis_0 = const()[name = tensor<string, []>("op_2504_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_2504_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_2504_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_2504_cast_fp16_2 = split(axis = var_2504_axis_0, split_sizes = var_2504_split_sizes_0, x = qkv_59_cast_fp16)[name = tensor<string, []>("op_2504_cast_fp16")];
tensor<int32, [1]> squeeze_42_axes_0 = const()[name = tensor<string, []>("squeeze_42_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_42_cast_fp16 = squeeze(axes = squeeze_42_axes_0, x = var_2504_cast_fp16_0)[name = tensor<string, []>("squeeze_42_cast_fp16")];
tensor<int32, [1]> squeeze_43_axes_0 = const()[name = tensor<string, []>("squeeze_43_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_43_cast_fp16 = squeeze(axes = squeeze_43_axes_0, x = var_2504_cast_fp16_1)[name = tensor<string, []>("squeeze_43_cast_fp16")];
tensor<int32, [1]> squeeze_44_axes_0 = const()[name = tensor<string, []>("squeeze_44_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_44_cast_fp16 = squeeze(axes = squeeze_44_axes_0, x = var_2504_cast_fp16_2)[name = tensor<string, []>("squeeze_44_cast_fp16")];
tensor<int32, [4]> q_57_perm_0 = const()[name = tensor<string, []>("q_57_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_57_perm_0 = const()[name = tensor<string, []>("k_57_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_29_perm_0 = const()[name = tensor<string, []>("value_29_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_57_cast_fp16 = transpose(perm = q_57_perm_0, x = squeeze_42_cast_fp16)[name = tensor<string, []>("transpose_50")];
tensor<fp16, [1, 12, 256, 64]> var_2514_cast_fp16 = mul(x = q_57_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2514_cast_fp16")];
tensor<int32, [4]> x1_57_begin_0 = const()[name = tensor<string, []>("x1_57_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_57_end_0 = const()[name = tensor<string, []>("x1_57_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_57_end_mask_0 = const()[name = tensor<string, []>("x1_57_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_57_cast_fp16 = slice_by_index(begin = x1_57_begin_0, end = x1_57_end_0, end_mask = x1_57_end_mask_0, x = q_57_cast_fp16)[name = tensor<string, []>("x1_57_cast_fp16")];
tensor<int32, [4]> x2_57_begin_0 = const()[name = tensor<string, []>("x2_57_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_57_end_0 = const()[name = tensor<string, []>("x2_57_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_57_end_mask_0 = const()[name = tensor<string, []>("x2_57_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_57_cast_fp16 = slice_by_index(begin = x2_57_begin_0, end = x2_57_end_0, end_mask = x2_57_end_mask_0, x = q_57_cast_fp16)[name = tensor<string, []>("x2_57_cast_fp16")];
tensor<fp16, []> const_116_promoted_to_fp16 = const()[name = tensor<string, []>("const_116_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2526_cast_fp16 = mul(x = x2_57_cast_fp16, y = const_116_promoted_to_fp16)[name = tensor<string, []>("op_2526_cast_fp16")];
tensor<bool, []> var_2528_interleave_0 = const()[name = tensor<string, []>("op_2528_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2528_cast_fp16 = concat(axis = var_2484, interleave = var_2528_interleave_0, values = (var_2526_cast_fp16, x1_57_cast_fp16))[name = tensor<string, []>("op_2528_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2529_cast_fp16 = mul(x = var_2528_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2529_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_29_cast_fp16 = add(x = var_2514_cast_fp16, y = var_2529_cast_fp16)[name = tensor<string, []>("q_embed_29_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_57_cast_fp16 = transpose(perm = k_57_perm_0, x = squeeze_43_cast_fp16)[name = tensor<string, []>("transpose_49")];
tensor<fp16, [1, 12, 256, 64]> var_2532_cast_fp16 = mul(x = k_57_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2532_cast_fp16")];
tensor<int32, [4]> x1_59_begin_0 = const()[name = tensor<string, []>("x1_59_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_59_end_0 = const()[name = tensor<string, []>("x1_59_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_59_end_mask_0 = const()[name = tensor<string, []>("x1_59_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_59_cast_fp16 = slice_by_index(begin = x1_59_begin_0, end = x1_59_end_0, end_mask = x1_59_end_mask_0, x = k_57_cast_fp16)[name = tensor<string, []>("x1_59_cast_fp16")];
tensor<int32, [4]> x2_59_begin_0 = const()[name = tensor<string, []>("x2_59_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_59_end_0 = const()[name = tensor<string, []>("x2_59_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_59_end_mask_0 = const()[name = tensor<string, []>("x2_59_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_59_cast_fp16 = slice_by_index(begin = x2_59_begin_0, end = x2_59_end_0, end_mask = x2_59_end_mask_0, x = k_57_cast_fp16)[name = tensor<string, []>("x2_59_cast_fp16")];
tensor<fp16, []> const_119_promoted_to_fp16 = const()[name = tensor<string, []>("const_119_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2544_cast_fp16 = mul(x = x2_59_cast_fp16, y = const_119_promoted_to_fp16)[name = tensor<string, []>("op_2544_cast_fp16")];
tensor<bool, []> var_2546_interleave_0 = const()[name = tensor<string, []>("op_2546_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2546_cast_fp16 = concat(axis = var_2484, interleave = var_2546_interleave_0, values = (var_2544_cast_fp16, x1_59_cast_fp16))[name = tensor<string, []>("op_2546_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2547_cast_fp16 = mul(x = var_2546_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2547_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_29_cast_fp16 = add(x = var_2532_cast_fp16, y = var_2547_cast_fp16)[name = tensor<string, []>("k_embed_29_cast_fp16")];
tensor<bool, []> var_2552_transpose_x_1 = const()[name = tensor<string, []>("op_2552_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_2552_transpose_y_1 = const()[name = tensor<string, []>("op_2552_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_2552_cast_fp16 = matmul(transpose_x = var_2552_transpose_x_1, transpose_y = var_2552_transpose_y_1, x = q_embed_29_cast_fp16, y = k_embed_29_cast_fp16)[name = tensor<string, []>("op_2552_cast_fp16")];
tensor<fp16, []> var_2553_to_fp16 = const()[name = tensor<string, []>("op_2553_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_57_cast_fp16 = mul(x = var_2552_cast_fp16, y = var_2553_to_fp16)[name = tensor<string, []>("attn_weights_57_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_259_cast_fp16 = add(x = attn_weights_57_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_259_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_2556_cast_fp16 = softmax(axis = var_2484, x = input_259_cast_fp16)[name = tensor<string, []>("op_2556_cast_fp16")];
tensor<bool, []> attn_output_85_transpose_x_0 = const()[name = tensor<string, []>("attn_output_85_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_85_transpose_y_0 = const()[name = tensor<string, []>("attn_output_85_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_29_cast_fp16 = transpose(perm = value_29_perm_0, x = squeeze_44_cast_fp16)[name = tensor<string, []>("transpose_48")];
tensor<fp16, [1, 12, 256, 64]> attn_output_85_cast_fp16 = matmul(transpose_x = attn_output_85_transpose_x_0, transpose_y = attn_output_85_transpose_y_0, x = var_2556_cast_fp16, y = value_29_cast_fp16)[name = tensor<string, []>("attn_output_85_cast_fp16")];
tensor<int32, [4]> var_2560_perm_0 = const()[name = tensor<string, []>("op_2560_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_2562 = const()[name = tensor<string, []>("op_2562"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_2560_cast_fp16 = transpose(perm = var_2560_perm_0, x = attn_output_85_cast_fp16)[name = tensor<string, []>("transpose_47")];
tensor<fp16, [1, 256, 768]> var_2563_cast_fp16 = reshape(shape = var_2562, x = var_2560_cast_fp16)[name = tensor<string, []>("op_2563_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_14_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_14_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(537517440)))];
tensor<fp16, [1, 256, 768]> linear_57_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_14_attn_Wo_weight_to_fp16, x = var_2563_cast_fp16)[name = tensor<string, []>("linear_57_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_265_cast_fp16 = add(x = input_257_cast_fp16, y = linear_57_cast_fp16)[name = tensor<string, []>("input_265_cast_fp16")];
tensor<int32, [1]> input_267_axes_0 = const()[name = tensor<string, []>("input_267_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_14_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_14_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(538697152)))];
tensor<fp16, [1, 256, 768]> input_267_cast_fp16 = layer_norm(axes = input_267_axes_0, epsilon = var_2495_to_fp16, gamma = model_encoder_layers_14_mlp_norm_weight_to_fp16, x = input_265_cast_fp16)[name = tensor<string, []>("input_267_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_14_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_14_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(538698752)))];
tensor<fp16, [1, 256, 2304]> linear_58_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_14_mlp_Wi_weight_to_fp16, x = input_267_cast_fp16)[name = tensor<string, []>("linear_58_cast_fp16")];
tensor<int32, [2]> var_2570_split_sizes_0 = const()[name = tensor<string, []>("op_2570_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_2570_axis_0 = const()[name = tensor<string, []>("op_2570_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_2570_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_2570_cast_fp16_1 = split(axis = var_2570_axis_0, split_sizes = var_2570_split_sizes_0, x = linear_58_cast_fp16)[name = tensor<string, []>("op_2570_cast_fp16")];
tensor<string, []> var_2572_mode_0 = const()[name = tensor<string, []>("op_2572_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_2572_cast_fp16 = gelu(mode = var_2572_mode_0, x = var_2570_cast_fp16_0)[name = tensor<string, []>("op_2572_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_271_cast_fp16 = mul(x = var_2572_cast_fp16, y = var_2570_cast_fp16_1)[name = tensor<string, []>("input_271_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_14_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_14_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(542237760)))];
tensor<fp16, [1, 256, 768]> linear_59_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_14_mlp_Wo_weight_to_fp16, x = input_271_cast_fp16)[name = tensor<string, []>("linear_59_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_275_cast_fp16 = add(x = input_265_cast_fp16, y = linear_59_cast_fp16)[name = tensor<string, []>("input_275_cast_fp16")];
tensor<int32, []> var_2581 = const()[name = tensor<string, []>("op_2581"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_31_axes_0 = const()[name = tensor<string, []>("hidden_states_31_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_15_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_15_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(544007296)))];
tensor<fp16, []> var_2592_to_fp16 = const()[name = tensor<string, []>("op_2592_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_31_cast_fp16 = layer_norm(axes = hidden_states_31_axes_0, epsilon = var_2592_to_fp16, gamma = model_encoder_layers_15_attn_norm_weight_to_fp16, x = input_275_cast_fp16)[name = tensor<string, []>("hidden_states_31_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_15_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_15_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(544008896)))];
tensor<fp16, [1, 256, 2304]> linear_60_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_15_attn_Wqkv_weight_to_fp16, x = hidden_states_31_cast_fp16)[name = tensor<string, []>("linear_60_cast_fp16")];
tensor<int32, [5]> var_2599 = const()[name = tensor<string, []>("op_2599"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_63_cast_fp16 = reshape(shape = var_2599, x = linear_60_cast_fp16)[name = tensor<string, []>("qkv_63_cast_fp16")];
tensor<int32, [3]> var_2601_split_sizes_0 = const()[name = tensor<string, []>("op_2601_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_2601_axis_0 = const()[name = tensor<string, []>("op_2601_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_2601_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_2601_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_2601_cast_fp16_2 = split(axis = var_2601_axis_0, split_sizes = var_2601_split_sizes_0, x = qkv_63_cast_fp16)[name = tensor<string, []>("op_2601_cast_fp16")];
tensor<int32, [1]> squeeze_45_axes_0 = const()[name = tensor<string, []>("squeeze_45_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_45_cast_fp16 = squeeze(axes = squeeze_45_axes_0, x = var_2601_cast_fp16_0)[name = tensor<string, []>("squeeze_45_cast_fp16")];
tensor<int32, [1]> squeeze_46_axes_0 = const()[name = tensor<string, []>("squeeze_46_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_46_cast_fp16 = squeeze(axes = squeeze_46_axes_0, x = var_2601_cast_fp16_1)[name = tensor<string, []>("squeeze_46_cast_fp16")];
tensor<int32, [1]> squeeze_47_axes_0 = const()[name = tensor<string, []>("squeeze_47_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_47_cast_fp16 = squeeze(axes = squeeze_47_axes_0, x = var_2601_cast_fp16_2)[name = tensor<string, []>("squeeze_47_cast_fp16")];
tensor<int32, [4]> q_61_perm_0 = const()[name = tensor<string, []>("q_61_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_61_perm_0 = const()[name = tensor<string, []>("k_61_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_31_perm_0 = const()[name = tensor<string, []>("value_31_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_61_cast_fp16 = transpose(perm = q_61_perm_0, x = squeeze_45_cast_fp16)[name = tensor<string, []>("transpose_46")];
tensor<fp16, [1, 12, 256, 64]> var_2611_cast_fp16 = mul(x = q_61_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2611_cast_fp16")];
tensor<int32, [4]> x1_61_begin_0 = const()[name = tensor<string, []>("x1_61_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_61_end_0 = const()[name = tensor<string, []>("x1_61_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_61_end_mask_0 = const()[name = tensor<string, []>("x1_61_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_61_cast_fp16 = slice_by_index(begin = x1_61_begin_0, end = x1_61_end_0, end_mask = x1_61_end_mask_0, x = q_61_cast_fp16)[name = tensor<string, []>("x1_61_cast_fp16")];
tensor<int32, [4]> x2_61_begin_0 = const()[name = tensor<string, []>("x2_61_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_61_end_0 = const()[name = tensor<string, []>("x2_61_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_61_end_mask_0 = const()[name = tensor<string, []>("x2_61_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_61_cast_fp16 = slice_by_index(begin = x2_61_begin_0, end = x2_61_end_0, end_mask = x2_61_end_mask_0, x = q_61_cast_fp16)[name = tensor<string, []>("x2_61_cast_fp16")];
tensor<fp16, []> const_124_promoted_to_fp16 = const()[name = tensor<string, []>("const_124_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2623_cast_fp16 = mul(x = x2_61_cast_fp16, y = const_124_promoted_to_fp16)[name = tensor<string, []>("op_2623_cast_fp16")];
tensor<bool, []> var_2625_interleave_0 = const()[name = tensor<string, []>("op_2625_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2625_cast_fp16 = concat(axis = var_2581, interleave = var_2625_interleave_0, values = (var_2623_cast_fp16, x1_61_cast_fp16))[name = tensor<string, []>("op_2625_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2626_cast_fp16 = mul(x = var_2625_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2626_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_31_cast_fp16 = add(x = var_2611_cast_fp16, y = var_2626_cast_fp16)[name = tensor<string, []>("q_embed_31_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_61_cast_fp16 = transpose(perm = k_61_perm_0, x = squeeze_46_cast_fp16)[name = tensor<string, []>("transpose_45")];
tensor<fp16, [1, 12, 256, 64]> var_2629_cast_fp16 = mul(x = k_61_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2629_cast_fp16")];
tensor<int32, [4]> x1_63_begin_0 = const()[name = tensor<string, []>("x1_63_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_63_end_0 = const()[name = tensor<string, []>("x1_63_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_63_end_mask_0 = const()[name = tensor<string, []>("x1_63_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_63_cast_fp16 = slice_by_index(begin = x1_63_begin_0, end = x1_63_end_0, end_mask = x1_63_end_mask_0, x = k_61_cast_fp16)[name = tensor<string, []>("x1_63_cast_fp16")];
tensor<int32, [4]> x2_63_begin_0 = const()[name = tensor<string, []>("x2_63_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_63_end_0 = const()[name = tensor<string, []>("x2_63_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_63_end_mask_0 = const()[name = tensor<string, []>("x2_63_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_63_cast_fp16 = slice_by_index(begin = x2_63_begin_0, end = x2_63_end_0, end_mask = x2_63_end_mask_0, x = k_61_cast_fp16)[name = tensor<string, []>("x2_63_cast_fp16")];
tensor<fp16, []> const_127_promoted_to_fp16 = const()[name = tensor<string, []>("const_127_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2641_cast_fp16 = mul(x = x2_63_cast_fp16, y = const_127_promoted_to_fp16)[name = tensor<string, []>("op_2641_cast_fp16")];
tensor<bool, []> var_2643_interleave_0 = const()[name = tensor<string, []>("op_2643_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2643_cast_fp16 = concat(axis = var_2581, interleave = var_2643_interleave_0, values = (var_2641_cast_fp16, x1_63_cast_fp16))[name = tensor<string, []>("op_2643_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2644_cast_fp16 = mul(x = var_2643_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2644_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_31_cast_fp16 = add(x = var_2629_cast_fp16, y = var_2644_cast_fp16)[name = tensor<string, []>("k_embed_31_cast_fp16")];
tensor<bool, []> var_2649_transpose_x_1 = const()[name = tensor<string, []>("op_2649_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_2649_transpose_y_1 = const()[name = tensor<string, []>("op_2649_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_2649_cast_fp16 = matmul(transpose_x = var_2649_transpose_x_1, transpose_y = var_2649_transpose_y_1, x = q_embed_31_cast_fp16, y = k_embed_31_cast_fp16)[name = tensor<string, []>("op_2649_cast_fp16")];
tensor<fp16, []> var_2650_to_fp16 = const()[name = tensor<string, []>("op_2650_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_61_cast_fp16 = mul(x = var_2649_cast_fp16, y = var_2650_to_fp16)[name = tensor<string, []>("attn_weights_61_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_277_cast_fp16 = add(x = attn_weights_61_cast_fp16, y = attention_mask_3_cast_fp16)[name = tensor<string, []>("input_277_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_2653_cast_fp16 = softmax(axis = var_2581, x = input_277_cast_fp16)[name = tensor<string, []>("op_2653_cast_fp16")];
tensor<bool, []> attn_output_91_transpose_x_0 = const()[name = tensor<string, []>("attn_output_91_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_91_transpose_y_0 = const()[name = tensor<string, []>("attn_output_91_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_31_cast_fp16 = transpose(perm = value_31_perm_0, x = squeeze_47_cast_fp16)[name = tensor<string, []>("transpose_44")];
tensor<fp16, [1, 12, 256, 64]> attn_output_91_cast_fp16 = matmul(transpose_x = attn_output_91_transpose_x_0, transpose_y = attn_output_91_transpose_y_0, x = var_2653_cast_fp16, y = value_31_cast_fp16)[name = tensor<string, []>("attn_output_91_cast_fp16")];
tensor<int32, [4]> var_2657_perm_0 = const()[name = tensor<string, []>("op_2657_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_2659 = const()[name = tensor<string, []>("op_2659"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_2657_cast_fp16 = transpose(perm = var_2657_perm_0, x = attn_output_91_cast_fp16)[name = tensor<string, []>("transpose_43")];
tensor<fp16, [1, 256, 768]> var_2660_cast_fp16 = reshape(shape = var_2659, x = var_2657_cast_fp16)[name = tensor<string, []>("op_2660_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_15_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_15_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(547547904)))];
tensor<fp16, [1, 256, 768]> linear_61_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_15_attn_Wo_weight_to_fp16, x = var_2660_cast_fp16)[name = tensor<string, []>("linear_61_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_283_cast_fp16 = add(x = input_275_cast_fp16, y = linear_61_cast_fp16)[name = tensor<string, []>("input_283_cast_fp16")];
tensor<int32, [1]> input_285_axes_0 = const()[name = tensor<string, []>("input_285_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_15_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_15_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(548727616)))];
tensor<fp16, [1, 256, 768]> input_285_cast_fp16 = layer_norm(axes = input_285_axes_0, epsilon = var_2592_to_fp16, gamma = model_encoder_layers_15_mlp_norm_weight_to_fp16, x = input_283_cast_fp16)[name = tensor<string, []>("input_285_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_15_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_15_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(548729216)))];
tensor<fp16, [1, 256, 2304]> linear_62_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_15_mlp_Wi_weight_to_fp16, x = input_285_cast_fp16)[name = tensor<string, []>("linear_62_cast_fp16")];
tensor<int32, [2]> var_2667_split_sizes_0 = const()[name = tensor<string, []>("op_2667_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_2667_axis_0 = const()[name = tensor<string, []>("op_2667_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_2667_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_2667_cast_fp16_1 = split(axis = var_2667_axis_0, split_sizes = var_2667_split_sizes_0, x = linear_62_cast_fp16)[name = tensor<string, []>("op_2667_cast_fp16")];
tensor<string, []> var_2669_mode_0 = const()[name = tensor<string, []>("op_2669_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_2669_cast_fp16 = gelu(mode = var_2669_mode_0, x = var_2667_cast_fp16_0)[name = tensor<string, []>("op_2669_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_289_cast_fp16 = mul(x = var_2669_cast_fp16, y = var_2667_cast_fp16_1)[name = tensor<string, []>("input_289_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_15_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_15_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(552268224)))];
tensor<fp16, [1, 256, 768]> linear_63_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_15_mlp_Wo_weight_to_fp16, x = input_289_cast_fp16)[name = tensor<string, []>("linear_63_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_293_cast_fp16 = add(x = input_283_cast_fp16, y = linear_63_cast_fp16)[name = tensor<string, []>("input_293_cast_fp16")];
tensor<int32, []> var_2678 = const()[name = tensor<string, []>("op_2678"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_33_axes_0 = const()[name = tensor<string, []>("hidden_states_33_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_16_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_16_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(554037760)))];
tensor<fp16, []> var_2689_to_fp16 = const()[name = tensor<string, []>("op_2689_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_33_cast_fp16 = layer_norm(axes = hidden_states_33_axes_0, epsilon = var_2689_to_fp16, gamma = model_encoder_layers_16_attn_norm_weight_to_fp16, x = input_293_cast_fp16)[name = tensor<string, []>("hidden_states_33_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_16_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_16_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(554039360)))];
tensor<fp16, [1, 256, 2304]> linear_64_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_16_attn_Wqkv_weight_to_fp16, x = hidden_states_33_cast_fp16)[name = tensor<string, []>("linear_64_cast_fp16")];
tensor<int32, [5]> var_2696 = const()[name = tensor<string, []>("op_2696"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_67_cast_fp16 = reshape(shape = var_2696, x = linear_64_cast_fp16)[name = tensor<string, []>("qkv_67_cast_fp16")];
tensor<int32, [3]> var_2698_split_sizes_0 = const()[name = tensor<string, []>("op_2698_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_2698_axis_0 = const()[name = tensor<string, []>("op_2698_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_2698_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_2698_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_2698_cast_fp16_2 = split(axis = var_2698_axis_0, split_sizes = var_2698_split_sizes_0, x = qkv_67_cast_fp16)[name = tensor<string, []>("op_2698_cast_fp16")];
tensor<int32, [1]> squeeze_48_axes_0 = const()[name = tensor<string, []>("squeeze_48_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_48_cast_fp16 = squeeze(axes = squeeze_48_axes_0, x = var_2698_cast_fp16_0)[name = tensor<string, []>("squeeze_48_cast_fp16")];
tensor<int32, [1]> squeeze_49_axes_0 = const()[name = tensor<string, []>("squeeze_49_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_49_cast_fp16 = squeeze(axes = squeeze_49_axes_0, x = var_2698_cast_fp16_1)[name = tensor<string, []>("squeeze_49_cast_fp16")];
tensor<int32, [1]> squeeze_50_axes_0 = const()[name = tensor<string, []>("squeeze_50_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_50_cast_fp16 = squeeze(axes = squeeze_50_axes_0, x = var_2698_cast_fp16_2)[name = tensor<string, []>("squeeze_50_cast_fp16")];
tensor<int32, [4]> q_65_perm_0 = const()[name = tensor<string, []>("q_65_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_65_perm_0 = const()[name = tensor<string, []>("k_65_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_33_perm_0 = const()[name = tensor<string, []>("value_33_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_65_cast_fp16 = transpose(perm = q_65_perm_0, x = squeeze_48_cast_fp16)[name = tensor<string, []>("transpose_42")];
tensor<fp16, [1, 12, 256, 64]> var_2708_cast_fp16 = mul(x = q_65_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2708_cast_fp16")];
tensor<int32, [4]> x1_65_begin_0 = const()[name = tensor<string, []>("x1_65_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_65_end_0 = const()[name = tensor<string, []>("x1_65_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_65_end_mask_0 = const()[name = tensor<string, []>("x1_65_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_65_cast_fp16 = slice_by_index(begin = x1_65_begin_0, end = x1_65_end_0, end_mask = x1_65_end_mask_0, x = q_65_cast_fp16)[name = tensor<string, []>("x1_65_cast_fp16")];
tensor<int32, [4]> x2_65_begin_0 = const()[name = tensor<string, []>("x2_65_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_65_end_0 = const()[name = tensor<string, []>("x2_65_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_65_end_mask_0 = const()[name = tensor<string, []>("x2_65_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_65_cast_fp16 = slice_by_index(begin = x2_65_begin_0, end = x2_65_end_0, end_mask = x2_65_end_mask_0, x = q_65_cast_fp16)[name = tensor<string, []>("x2_65_cast_fp16")];
tensor<fp16, []> const_132_promoted_to_fp16 = const()[name = tensor<string, []>("const_132_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2720_cast_fp16 = mul(x = x2_65_cast_fp16, y = const_132_promoted_to_fp16)[name = tensor<string, []>("op_2720_cast_fp16")];
tensor<bool, []> var_2722_interleave_0 = const()[name = tensor<string, []>("op_2722_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2722_cast_fp16 = concat(axis = var_2678, interleave = var_2722_interleave_0, values = (var_2720_cast_fp16, x1_65_cast_fp16))[name = tensor<string, []>("op_2722_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2723_cast_fp16 = mul(x = var_2722_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2723_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_33_cast_fp16 = add(x = var_2708_cast_fp16, y = var_2723_cast_fp16)[name = tensor<string, []>("q_embed_33_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_65_cast_fp16 = transpose(perm = k_65_perm_0, x = squeeze_49_cast_fp16)[name = tensor<string, []>("transpose_41")];
tensor<fp16, [1, 12, 256, 64]> var_2726_cast_fp16 = mul(x = k_65_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2726_cast_fp16")];
tensor<int32, [4]> x1_67_begin_0 = const()[name = tensor<string, []>("x1_67_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_67_end_0 = const()[name = tensor<string, []>("x1_67_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_67_end_mask_0 = const()[name = tensor<string, []>("x1_67_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_67_cast_fp16 = slice_by_index(begin = x1_67_begin_0, end = x1_67_end_0, end_mask = x1_67_end_mask_0, x = k_65_cast_fp16)[name = tensor<string, []>("x1_67_cast_fp16")];
tensor<int32, [4]> x2_67_begin_0 = const()[name = tensor<string, []>("x2_67_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_67_end_0 = const()[name = tensor<string, []>("x2_67_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_67_end_mask_0 = const()[name = tensor<string, []>("x2_67_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_67_cast_fp16 = slice_by_index(begin = x2_67_begin_0, end = x2_67_end_0, end_mask = x2_67_end_mask_0, x = k_65_cast_fp16)[name = tensor<string, []>("x2_67_cast_fp16")];
tensor<fp16, []> const_135_promoted_to_fp16 = const()[name = tensor<string, []>("const_135_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2738_cast_fp16 = mul(x = x2_67_cast_fp16, y = const_135_promoted_to_fp16)[name = tensor<string, []>("op_2738_cast_fp16")];
tensor<bool, []> var_2740_interleave_0 = const()[name = tensor<string, []>("op_2740_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2740_cast_fp16 = concat(axis = var_2678, interleave = var_2740_interleave_0, values = (var_2738_cast_fp16, x1_67_cast_fp16))[name = tensor<string, []>("op_2740_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2741_cast_fp16 = mul(x = var_2740_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2741_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_33_cast_fp16 = add(x = var_2726_cast_fp16, y = var_2741_cast_fp16)[name = tensor<string, []>("k_embed_33_cast_fp16")];
tensor<bool, []> var_2746_transpose_x_1 = const()[name = tensor<string, []>("op_2746_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_2746_transpose_y_1 = const()[name = tensor<string, []>("op_2746_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_2746_cast_fp16 = matmul(transpose_x = var_2746_transpose_x_1, transpose_y = var_2746_transpose_y_1, x = q_embed_33_cast_fp16, y = k_embed_33_cast_fp16)[name = tensor<string, []>("op_2746_cast_fp16")];
tensor<fp16, []> var_2747_to_fp16 = const()[name = tensor<string, []>("op_2747_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_65_cast_fp16 = mul(x = var_2746_cast_fp16, y = var_2747_to_fp16)[name = tensor<string, []>("attn_weights_65_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_295_cast_fp16 = add(x = attn_weights_65_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_295_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_2750_cast_fp16 = softmax(axis = var_2678, x = input_295_cast_fp16)[name = tensor<string, []>("op_2750_cast_fp16")];
tensor<bool, []> attn_output_97_transpose_x_0 = const()[name = tensor<string, []>("attn_output_97_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_97_transpose_y_0 = const()[name = tensor<string, []>("attn_output_97_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_33_cast_fp16 = transpose(perm = value_33_perm_0, x = squeeze_50_cast_fp16)[name = tensor<string, []>("transpose_40")];
tensor<fp16, [1, 12, 256, 64]> attn_output_97_cast_fp16 = matmul(transpose_x = attn_output_97_transpose_x_0, transpose_y = attn_output_97_transpose_y_0, x = var_2750_cast_fp16, y = value_33_cast_fp16)[name = tensor<string, []>("attn_output_97_cast_fp16")];
tensor<int32, [4]> var_2754_perm_0 = const()[name = tensor<string, []>("op_2754_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_2756 = const()[name = tensor<string, []>("op_2756"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_2754_cast_fp16 = transpose(perm = var_2754_perm_0, x = attn_output_97_cast_fp16)[name = tensor<string, []>("transpose_39")];
tensor<fp16, [1, 256, 768]> var_2757_cast_fp16 = reshape(shape = var_2756, x = var_2754_cast_fp16)[name = tensor<string, []>("op_2757_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_16_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_16_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(557578368)))];
tensor<fp16, [1, 256, 768]> linear_65_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_16_attn_Wo_weight_to_fp16, x = var_2757_cast_fp16)[name = tensor<string, []>("linear_65_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_301_cast_fp16 = add(x = input_293_cast_fp16, y = linear_65_cast_fp16)[name = tensor<string, []>("input_301_cast_fp16")];
tensor<int32, [1]> input_303_axes_0 = const()[name = tensor<string, []>("input_303_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_16_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_16_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(558758080)))];
tensor<fp16, [1, 256, 768]> input_303_cast_fp16 = layer_norm(axes = input_303_axes_0, epsilon = var_2689_to_fp16, gamma = model_encoder_layers_16_mlp_norm_weight_to_fp16, x = input_301_cast_fp16)[name = tensor<string, []>("input_303_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_16_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_16_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(558759680)))];
tensor<fp16, [1, 256, 2304]> linear_66_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_16_mlp_Wi_weight_to_fp16, x = input_303_cast_fp16)[name = tensor<string, []>("linear_66_cast_fp16")];
tensor<int32, [2]> var_2764_split_sizes_0 = const()[name = tensor<string, []>("op_2764_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_2764_axis_0 = const()[name = tensor<string, []>("op_2764_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_2764_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_2764_cast_fp16_1 = split(axis = var_2764_axis_0, split_sizes = var_2764_split_sizes_0, x = linear_66_cast_fp16)[name = tensor<string, []>("op_2764_cast_fp16")];
tensor<string, []> var_2766_mode_0 = const()[name = tensor<string, []>("op_2766_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_2766_cast_fp16 = gelu(mode = var_2766_mode_0, x = var_2764_cast_fp16_0)[name = tensor<string, []>("op_2766_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_307_cast_fp16 = mul(x = var_2766_cast_fp16, y = var_2764_cast_fp16_1)[name = tensor<string, []>("input_307_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_16_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_16_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(562298688)))];
tensor<fp16, [1, 256, 768]> linear_67_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_16_mlp_Wo_weight_to_fp16, x = input_307_cast_fp16)[name = tensor<string, []>("linear_67_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_311_cast_fp16 = add(x = input_301_cast_fp16, y = linear_67_cast_fp16)[name = tensor<string, []>("input_311_cast_fp16")];
tensor<int32, []> var_2775 = const()[name = tensor<string, []>("op_2775"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_35_axes_0 = const()[name = tensor<string, []>("hidden_states_35_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_17_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_17_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(564068224)))];
tensor<fp16, []> var_2786_to_fp16 = const()[name = tensor<string, []>("op_2786_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_35_cast_fp16 = layer_norm(axes = hidden_states_35_axes_0, epsilon = var_2786_to_fp16, gamma = model_encoder_layers_17_attn_norm_weight_to_fp16, x = input_311_cast_fp16)[name = tensor<string, []>("hidden_states_35_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_17_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_17_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(564069824)))];
tensor<fp16, [1, 256, 2304]> linear_68_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_17_attn_Wqkv_weight_to_fp16, x = hidden_states_35_cast_fp16)[name = tensor<string, []>("linear_68_cast_fp16")];
tensor<int32, [5]> var_2793 = const()[name = tensor<string, []>("op_2793"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_71_cast_fp16 = reshape(shape = var_2793, x = linear_68_cast_fp16)[name = tensor<string, []>("qkv_71_cast_fp16")];
tensor<int32, [3]> var_2795_split_sizes_0 = const()[name = tensor<string, []>("op_2795_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_2795_axis_0 = const()[name = tensor<string, []>("op_2795_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_2795_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_2795_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_2795_cast_fp16_2 = split(axis = var_2795_axis_0, split_sizes = var_2795_split_sizes_0, x = qkv_71_cast_fp16)[name = tensor<string, []>("op_2795_cast_fp16")];
tensor<int32, [1]> squeeze_51_axes_0 = const()[name = tensor<string, []>("squeeze_51_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_51_cast_fp16 = squeeze(axes = squeeze_51_axes_0, x = var_2795_cast_fp16_0)[name = tensor<string, []>("squeeze_51_cast_fp16")];
tensor<int32, [1]> squeeze_52_axes_0 = const()[name = tensor<string, []>("squeeze_52_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_52_cast_fp16 = squeeze(axes = squeeze_52_axes_0, x = var_2795_cast_fp16_1)[name = tensor<string, []>("squeeze_52_cast_fp16")];
tensor<int32, [1]> squeeze_53_axes_0 = const()[name = tensor<string, []>("squeeze_53_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_53_cast_fp16 = squeeze(axes = squeeze_53_axes_0, x = var_2795_cast_fp16_2)[name = tensor<string, []>("squeeze_53_cast_fp16")];
tensor<int32, [4]> q_69_perm_0 = const()[name = tensor<string, []>("q_69_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_69_perm_0 = const()[name = tensor<string, []>("k_69_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_35_perm_0 = const()[name = tensor<string, []>("value_35_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_69_cast_fp16 = transpose(perm = q_69_perm_0, x = squeeze_51_cast_fp16)[name = tensor<string, []>("transpose_38")];
tensor<fp16, [1, 12, 256, 64]> var_2805_cast_fp16 = mul(x = q_69_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2805_cast_fp16")];
tensor<int32, [4]> x1_69_begin_0 = const()[name = tensor<string, []>("x1_69_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_69_end_0 = const()[name = tensor<string, []>("x1_69_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_69_end_mask_0 = const()[name = tensor<string, []>("x1_69_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_69_cast_fp16 = slice_by_index(begin = x1_69_begin_0, end = x1_69_end_0, end_mask = x1_69_end_mask_0, x = q_69_cast_fp16)[name = tensor<string, []>("x1_69_cast_fp16")];
tensor<int32, [4]> x2_69_begin_0 = const()[name = tensor<string, []>("x2_69_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_69_end_0 = const()[name = tensor<string, []>("x2_69_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_69_end_mask_0 = const()[name = tensor<string, []>("x2_69_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_69_cast_fp16 = slice_by_index(begin = x2_69_begin_0, end = x2_69_end_0, end_mask = x2_69_end_mask_0, x = q_69_cast_fp16)[name = tensor<string, []>("x2_69_cast_fp16")];
tensor<fp16, []> const_140_promoted_to_fp16 = const()[name = tensor<string, []>("const_140_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2817_cast_fp16 = mul(x = x2_69_cast_fp16, y = const_140_promoted_to_fp16)[name = tensor<string, []>("op_2817_cast_fp16")];
tensor<bool, []> var_2819_interleave_0 = const()[name = tensor<string, []>("op_2819_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2819_cast_fp16 = concat(axis = var_2775, interleave = var_2819_interleave_0, values = (var_2817_cast_fp16, x1_69_cast_fp16))[name = tensor<string, []>("op_2819_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2820_cast_fp16 = mul(x = var_2819_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2820_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_35_cast_fp16 = add(x = var_2805_cast_fp16, y = var_2820_cast_fp16)[name = tensor<string, []>("q_embed_35_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_69_cast_fp16 = transpose(perm = k_69_perm_0, x = squeeze_52_cast_fp16)[name = tensor<string, []>("transpose_37")];
tensor<fp16, [1, 12, 256, 64]> var_2823_cast_fp16 = mul(x = k_69_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2823_cast_fp16")];
tensor<int32, [4]> x1_71_begin_0 = const()[name = tensor<string, []>("x1_71_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_71_end_0 = const()[name = tensor<string, []>("x1_71_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_71_end_mask_0 = const()[name = tensor<string, []>("x1_71_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_71_cast_fp16 = slice_by_index(begin = x1_71_begin_0, end = x1_71_end_0, end_mask = x1_71_end_mask_0, x = k_69_cast_fp16)[name = tensor<string, []>("x1_71_cast_fp16")];
tensor<int32, [4]> x2_71_begin_0 = const()[name = tensor<string, []>("x2_71_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_71_end_0 = const()[name = tensor<string, []>("x2_71_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_71_end_mask_0 = const()[name = tensor<string, []>("x2_71_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_71_cast_fp16 = slice_by_index(begin = x2_71_begin_0, end = x2_71_end_0, end_mask = x2_71_end_mask_0, x = k_69_cast_fp16)[name = tensor<string, []>("x2_71_cast_fp16")];
tensor<fp16, []> const_143_promoted_to_fp16 = const()[name = tensor<string, []>("const_143_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2835_cast_fp16 = mul(x = x2_71_cast_fp16, y = const_143_promoted_to_fp16)[name = tensor<string, []>("op_2835_cast_fp16")];
tensor<bool, []> var_2837_interleave_0 = const()[name = tensor<string, []>("op_2837_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2837_cast_fp16 = concat(axis = var_2775, interleave = var_2837_interleave_0, values = (var_2835_cast_fp16, x1_71_cast_fp16))[name = tensor<string, []>("op_2837_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2838_cast_fp16 = mul(x = var_2837_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2838_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_35_cast_fp16 = add(x = var_2823_cast_fp16, y = var_2838_cast_fp16)[name = tensor<string, []>("k_embed_35_cast_fp16")];
tensor<bool, []> var_2843_transpose_x_1 = const()[name = tensor<string, []>("op_2843_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_2843_transpose_y_1 = const()[name = tensor<string, []>("op_2843_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_2843_cast_fp16 = matmul(transpose_x = var_2843_transpose_x_1, transpose_y = var_2843_transpose_y_1, x = q_embed_35_cast_fp16, y = k_embed_35_cast_fp16)[name = tensor<string, []>("op_2843_cast_fp16")];
tensor<fp16, []> var_2844_to_fp16 = const()[name = tensor<string, []>("op_2844_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_69_cast_fp16 = mul(x = var_2843_cast_fp16, y = var_2844_to_fp16)[name = tensor<string, []>("attn_weights_69_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_313_cast_fp16 = add(x = attn_weights_69_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_313_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_2847_cast_fp16 = softmax(axis = var_2775, x = input_313_cast_fp16)[name = tensor<string, []>("op_2847_cast_fp16")];
tensor<bool, []> attn_output_103_transpose_x_0 = const()[name = tensor<string, []>("attn_output_103_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_103_transpose_y_0 = const()[name = tensor<string, []>("attn_output_103_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_35_cast_fp16 = transpose(perm = value_35_perm_0, x = squeeze_53_cast_fp16)[name = tensor<string, []>("transpose_36")];
tensor<fp16, [1, 12, 256, 64]> attn_output_103_cast_fp16 = matmul(transpose_x = attn_output_103_transpose_x_0, transpose_y = attn_output_103_transpose_y_0, x = var_2847_cast_fp16, y = value_35_cast_fp16)[name = tensor<string, []>("attn_output_103_cast_fp16")];
tensor<int32, [4]> var_2851_perm_0 = const()[name = tensor<string, []>("op_2851_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_2853 = const()[name = tensor<string, []>("op_2853"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_2851_cast_fp16 = transpose(perm = var_2851_perm_0, x = attn_output_103_cast_fp16)[name = tensor<string, []>("transpose_35")];
tensor<fp16, [1, 256, 768]> var_2854_cast_fp16 = reshape(shape = var_2853, x = var_2851_cast_fp16)[name = tensor<string, []>("op_2854_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_17_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_17_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(567608832)))];
tensor<fp16, [1, 256, 768]> linear_69_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_17_attn_Wo_weight_to_fp16, x = var_2854_cast_fp16)[name = tensor<string, []>("linear_69_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_319_cast_fp16 = add(x = input_311_cast_fp16, y = linear_69_cast_fp16)[name = tensor<string, []>("input_319_cast_fp16")];
tensor<int32, [1]> input_321_axes_0 = const()[name = tensor<string, []>("input_321_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_17_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_17_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(568788544)))];
tensor<fp16, [1, 256, 768]> input_321_cast_fp16 = layer_norm(axes = input_321_axes_0, epsilon = var_2786_to_fp16, gamma = model_encoder_layers_17_mlp_norm_weight_to_fp16, x = input_319_cast_fp16)[name = tensor<string, []>("input_321_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_17_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_17_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(568790144)))];
tensor<fp16, [1, 256, 2304]> linear_70_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_17_mlp_Wi_weight_to_fp16, x = input_321_cast_fp16)[name = tensor<string, []>("linear_70_cast_fp16")];
tensor<int32, [2]> var_2861_split_sizes_0 = const()[name = tensor<string, []>("op_2861_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_2861_axis_0 = const()[name = tensor<string, []>("op_2861_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_2861_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_2861_cast_fp16_1 = split(axis = var_2861_axis_0, split_sizes = var_2861_split_sizes_0, x = linear_70_cast_fp16)[name = tensor<string, []>("op_2861_cast_fp16")];
tensor<string, []> var_2863_mode_0 = const()[name = tensor<string, []>("op_2863_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_2863_cast_fp16 = gelu(mode = var_2863_mode_0, x = var_2861_cast_fp16_0)[name = tensor<string, []>("op_2863_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_325_cast_fp16 = mul(x = var_2863_cast_fp16, y = var_2861_cast_fp16_1)[name = tensor<string, []>("input_325_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_17_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_17_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(572329152)))];
tensor<fp16, [1, 256, 768]> linear_71_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_17_mlp_Wo_weight_to_fp16, x = input_325_cast_fp16)[name = tensor<string, []>("linear_71_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_329_cast_fp16 = add(x = input_319_cast_fp16, y = linear_71_cast_fp16)[name = tensor<string, []>("input_329_cast_fp16")];
tensor<int32, []> var_2872 = const()[name = tensor<string, []>("op_2872"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_37_axes_0 = const()[name = tensor<string, []>("hidden_states_37_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_18_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_18_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(574098688)))];
tensor<fp16, []> var_2883_to_fp16 = const()[name = tensor<string, []>("op_2883_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_37_cast_fp16 = layer_norm(axes = hidden_states_37_axes_0, epsilon = var_2883_to_fp16, gamma = model_encoder_layers_18_attn_norm_weight_to_fp16, x = input_329_cast_fp16)[name = tensor<string, []>("hidden_states_37_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_18_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_18_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(574100288)))];
tensor<fp16, [1, 256, 2304]> linear_72_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_18_attn_Wqkv_weight_to_fp16, x = hidden_states_37_cast_fp16)[name = tensor<string, []>("linear_72_cast_fp16")];
tensor<int32, [5]> var_2890 = const()[name = tensor<string, []>("op_2890"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_75_cast_fp16 = reshape(shape = var_2890, x = linear_72_cast_fp16)[name = tensor<string, []>("qkv_75_cast_fp16")];
tensor<int32, [3]> var_2892_split_sizes_0 = const()[name = tensor<string, []>("op_2892_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_2892_axis_0 = const()[name = tensor<string, []>("op_2892_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_2892_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_2892_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_2892_cast_fp16_2 = split(axis = var_2892_axis_0, split_sizes = var_2892_split_sizes_0, x = qkv_75_cast_fp16)[name = tensor<string, []>("op_2892_cast_fp16")];
tensor<int32, [1]> squeeze_54_axes_0 = const()[name = tensor<string, []>("squeeze_54_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_54_cast_fp16 = squeeze(axes = squeeze_54_axes_0, x = var_2892_cast_fp16_0)[name = tensor<string, []>("squeeze_54_cast_fp16")];
tensor<int32, [1]> squeeze_55_axes_0 = const()[name = tensor<string, []>("squeeze_55_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_55_cast_fp16 = squeeze(axes = squeeze_55_axes_0, x = var_2892_cast_fp16_1)[name = tensor<string, []>("squeeze_55_cast_fp16")];
tensor<int32, [1]> squeeze_56_axes_0 = const()[name = tensor<string, []>("squeeze_56_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_56_cast_fp16 = squeeze(axes = squeeze_56_axes_0, x = var_2892_cast_fp16_2)[name = tensor<string, []>("squeeze_56_cast_fp16")];
tensor<int32, [4]> q_73_perm_0 = const()[name = tensor<string, []>("q_73_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_73_perm_0 = const()[name = tensor<string, []>("k_73_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_37_perm_0 = const()[name = tensor<string, []>("value_37_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_73_cast_fp16 = transpose(perm = q_73_perm_0, x = squeeze_54_cast_fp16)[name = tensor<string, []>("transpose_34")];
tensor<fp16, [1, 12, 256, 64]> var_2902_cast_fp16 = mul(x = q_73_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2902_cast_fp16")];
tensor<int32, [4]> x1_73_begin_0 = const()[name = tensor<string, []>("x1_73_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_73_end_0 = const()[name = tensor<string, []>("x1_73_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_73_end_mask_0 = const()[name = tensor<string, []>("x1_73_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_73_cast_fp16 = slice_by_index(begin = x1_73_begin_0, end = x1_73_end_0, end_mask = x1_73_end_mask_0, x = q_73_cast_fp16)[name = tensor<string, []>("x1_73_cast_fp16")];
tensor<int32, [4]> x2_73_begin_0 = const()[name = tensor<string, []>("x2_73_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_73_end_0 = const()[name = tensor<string, []>("x2_73_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_73_end_mask_0 = const()[name = tensor<string, []>("x2_73_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_73_cast_fp16 = slice_by_index(begin = x2_73_begin_0, end = x2_73_end_0, end_mask = x2_73_end_mask_0, x = q_73_cast_fp16)[name = tensor<string, []>("x2_73_cast_fp16")];
tensor<fp16, []> const_148_promoted_to_fp16 = const()[name = tensor<string, []>("const_148_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2914_cast_fp16 = mul(x = x2_73_cast_fp16, y = const_148_promoted_to_fp16)[name = tensor<string, []>("op_2914_cast_fp16")];
tensor<bool, []> var_2916_interleave_0 = const()[name = tensor<string, []>("op_2916_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2916_cast_fp16 = concat(axis = var_2872, interleave = var_2916_interleave_0, values = (var_2914_cast_fp16, x1_73_cast_fp16))[name = tensor<string, []>("op_2916_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2917_cast_fp16 = mul(x = var_2916_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2917_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_37_cast_fp16 = add(x = var_2902_cast_fp16, y = var_2917_cast_fp16)[name = tensor<string, []>("q_embed_37_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_73_cast_fp16 = transpose(perm = k_73_perm_0, x = squeeze_55_cast_fp16)[name = tensor<string, []>("transpose_33")];
tensor<fp16, [1, 12, 256, 64]> var_2920_cast_fp16 = mul(x = k_73_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2920_cast_fp16")];
tensor<int32, [4]> x1_75_begin_0 = const()[name = tensor<string, []>("x1_75_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_75_end_0 = const()[name = tensor<string, []>("x1_75_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_75_end_mask_0 = const()[name = tensor<string, []>("x1_75_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_75_cast_fp16 = slice_by_index(begin = x1_75_begin_0, end = x1_75_end_0, end_mask = x1_75_end_mask_0, x = k_73_cast_fp16)[name = tensor<string, []>("x1_75_cast_fp16")];
tensor<int32, [4]> x2_75_begin_0 = const()[name = tensor<string, []>("x2_75_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_75_end_0 = const()[name = tensor<string, []>("x2_75_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_75_end_mask_0 = const()[name = tensor<string, []>("x2_75_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_75_cast_fp16 = slice_by_index(begin = x2_75_begin_0, end = x2_75_end_0, end_mask = x2_75_end_mask_0, x = k_73_cast_fp16)[name = tensor<string, []>("x2_75_cast_fp16")];
tensor<fp16, []> const_151_promoted_to_fp16 = const()[name = tensor<string, []>("const_151_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_2932_cast_fp16 = mul(x = x2_75_cast_fp16, y = const_151_promoted_to_fp16)[name = tensor<string, []>("op_2932_cast_fp16")];
tensor<bool, []> var_2934_interleave_0 = const()[name = tensor<string, []>("op_2934_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_2934_cast_fp16 = concat(axis = var_2872, interleave = var_2934_interleave_0, values = (var_2932_cast_fp16, x1_75_cast_fp16))[name = tensor<string, []>("op_2934_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_2935_cast_fp16 = mul(x = var_2934_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_2935_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_37_cast_fp16 = add(x = var_2920_cast_fp16, y = var_2935_cast_fp16)[name = tensor<string, []>("k_embed_37_cast_fp16")];
tensor<bool, []> var_2940_transpose_x_1 = const()[name = tensor<string, []>("op_2940_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_2940_transpose_y_1 = const()[name = tensor<string, []>("op_2940_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_2940_cast_fp16 = matmul(transpose_x = var_2940_transpose_x_1, transpose_y = var_2940_transpose_y_1, x = q_embed_37_cast_fp16, y = k_embed_37_cast_fp16)[name = tensor<string, []>("op_2940_cast_fp16")];
tensor<fp16, []> var_2941_to_fp16 = const()[name = tensor<string, []>("op_2941_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_73_cast_fp16 = mul(x = var_2940_cast_fp16, y = var_2941_to_fp16)[name = tensor<string, []>("attn_weights_73_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_331_cast_fp16 = add(x = attn_weights_73_cast_fp16, y = attention_mask_3_cast_fp16)[name = tensor<string, []>("input_331_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_2944_cast_fp16 = softmax(axis = var_2872, x = input_331_cast_fp16)[name = tensor<string, []>("op_2944_cast_fp16")];
tensor<bool, []> attn_output_109_transpose_x_0 = const()[name = tensor<string, []>("attn_output_109_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_109_transpose_y_0 = const()[name = tensor<string, []>("attn_output_109_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_37_cast_fp16 = transpose(perm = value_37_perm_0, x = squeeze_56_cast_fp16)[name = tensor<string, []>("transpose_32")];
tensor<fp16, [1, 12, 256, 64]> attn_output_109_cast_fp16 = matmul(transpose_x = attn_output_109_transpose_x_0, transpose_y = attn_output_109_transpose_y_0, x = var_2944_cast_fp16, y = value_37_cast_fp16)[name = tensor<string, []>("attn_output_109_cast_fp16")];
tensor<int32, [4]> var_2948_perm_0 = const()[name = tensor<string, []>("op_2948_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_2950 = const()[name = tensor<string, []>("op_2950"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_2948_cast_fp16 = transpose(perm = var_2948_perm_0, x = attn_output_109_cast_fp16)[name = tensor<string, []>("transpose_31")];
tensor<fp16, [1, 256, 768]> var_2951_cast_fp16 = reshape(shape = var_2950, x = var_2948_cast_fp16)[name = tensor<string, []>("op_2951_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_18_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_18_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(577639296)))];
tensor<fp16, [1, 256, 768]> linear_73_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_18_attn_Wo_weight_to_fp16, x = var_2951_cast_fp16)[name = tensor<string, []>("linear_73_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_337_cast_fp16 = add(x = input_329_cast_fp16, y = linear_73_cast_fp16)[name = tensor<string, []>("input_337_cast_fp16")];
tensor<int32, [1]> input_339_axes_0 = const()[name = tensor<string, []>("input_339_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_18_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_18_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(578819008)))];
tensor<fp16, [1, 256, 768]> input_339_cast_fp16 = layer_norm(axes = input_339_axes_0, epsilon = var_2883_to_fp16, gamma = model_encoder_layers_18_mlp_norm_weight_to_fp16, x = input_337_cast_fp16)[name = tensor<string, []>("input_339_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_18_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_18_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(578820608)))];
tensor<fp16, [1, 256, 2304]> linear_74_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_18_mlp_Wi_weight_to_fp16, x = input_339_cast_fp16)[name = tensor<string, []>("linear_74_cast_fp16")];
tensor<int32, [2]> var_2958_split_sizes_0 = const()[name = tensor<string, []>("op_2958_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_2958_axis_0 = const()[name = tensor<string, []>("op_2958_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_2958_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_2958_cast_fp16_1 = split(axis = var_2958_axis_0, split_sizes = var_2958_split_sizes_0, x = linear_74_cast_fp16)[name = tensor<string, []>("op_2958_cast_fp16")];
tensor<string, []> var_2960_mode_0 = const()[name = tensor<string, []>("op_2960_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_2960_cast_fp16 = gelu(mode = var_2960_mode_0, x = var_2958_cast_fp16_0)[name = tensor<string, []>("op_2960_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_343_cast_fp16 = mul(x = var_2960_cast_fp16, y = var_2958_cast_fp16_1)[name = tensor<string, []>("input_343_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_18_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_18_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(582359616)))];
tensor<fp16, [1, 256, 768]> linear_75_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_18_mlp_Wo_weight_to_fp16, x = input_343_cast_fp16)[name = tensor<string, []>("linear_75_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_347_cast_fp16 = add(x = input_337_cast_fp16, y = linear_75_cast_fp16)[name = tensor<string, []>("input_347_cast_fp16")];
tensor<int32, []> var_2969 = const()[name = tensor<string, []>("op_2969"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_39_axes_0 = const()[name = tensor<string, []>("hidden_states_39_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_19_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_19_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(584129152)))];
tensor<fp16, []> var_2980_to_fp16 = const()[name = tensor<string, []>("op_2980_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_39_cast_fp16 = layer_norm(axes = hidden_states_39_axes_0, epsilon = var_2980_to_fp16, gamma = model_encoder_layers_19_attn_norm_weight_to_fp16, x = input_347_cast_fp16)[name = tensor<string, []>("hidden_states_39_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_19_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_19_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(584130752)))];
tensor<fp16, [1, 256, 2304]> linear_76_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_19_attn_Wqkv_weight_to_fp16, x = hidden_states_39_cast_fp16)[name = tensor<string, []>("linear_76_cast_fp16")];
tensor<int32, [5]> var_2987 = const()[name = tensor<string, []>("op_2987"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_79_cast_fp16 = reshape(shape = var_2987, x = linear_76_cast_fp16)[name = tensor<string, []>("qkv_79_cast_fp16")];
tensor<int32, [3]> var_2989_split_sizes_0 = const()[name = tensor<string, []>("op_2989_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_2989_axis_0 = const()[name = tensor<string, []>("op_2989_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_2989_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_2989_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_2989_cast_fp16_2 = split(axis = var_2989_axis_0, split_sizes = var_2989_split_sizes_0, x = qkv_79_cast_fp16)[name = tensor<string, []>("op_2989_cast_fp16")];
tensor<int32, [1]> squeeze_57_axes_0 = const()[name = tensor<string, []>("squeeze_57_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_57_cast_fp16 = squeeze(axes = squeeze_57_axes_0, x = var_2989_cast_fp16_0)[name = tensor<string, []>("squeeze_57_cast_fp16")];
tensor<int32, [1]> squeeze_58_axes_0 = const()[name = tensor<string, []>("squeeze_58_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_58_cast_fp16 = squeeze(axes = squeeze_58_axes_0, x = var_2989_cast_fp16_1)[name = tensor<string, []>("squeeze_58_cast_fp16")];
tensor<int32, [1]> squeeze_59_axes_0 = const()[name = tensor<string, []>("squeeze_59_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_59_cast_fp16 = squeeze(axes = squeeze_59_axes_0, x = var_2989_cast_fp16_2)[name = tensor<string, []>("squeeze_59_cast_fp16")];
tensor<int32, [4]> q_77_perm_0 = const()[name = tensor<string, []>("q_77_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_77_perm_0 = const()[name = tensor<string, []>("k_77_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_39_perm_0 = const()[name = tensor<string, []>("value_39_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_77_cast_fp16 = transpose(perm = q_77_perm_0, x = squeeze_57_cast_fp16)[name = tensor<string, []>("transpose_30")];
tensor<fp16, [1, 12, 256, 64]> var_2999_cast_fp16 = mul(x = q_77_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_2999_cast_fp16")];
tensor<int32, [4]> x1_77_begin_0 = const()[name = tensor<string, []>("x1_77_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_77_end_0 = const()[name = tensor<string, []>("x1_77_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_77_end_mask_0 = const()[name = tensor<string, []>("x1_77_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_77_cast_fp16 = slice_by_index(begin = x1_77_begin_0, end = x1_77_end_0, end_mask = x1_77_end_mask_0, x = q_77_cast_fp16)[name = tensor<string, []>("x1_77_cast_fp16")];
tensor<int32, [4]> x2_77_begin_0 = const()[name = tensor<string, []>("x2_77_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_77_end_0 = const()[name = tensor<string, []>("x2_77_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_77_end_mask_0 = const()[name = tensor<string, []>("x2_77_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_77_cast_fp16 = slice_by_index(begin = x2_77_begin_0, end = x2_77_end_0, end_mask = x2_77_end_mask_0, x = q_77_cast_fp16)[name = tensor<string, []>("x2_77_cast_fp16")];
tensor<fp16, []> const_156_promoted_to_fp16 = const()[name = tensor<string, []>("const_156_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_3011_cast_fp16 = mul(x = x2_77_cast_fp16, y = const_156_promoted_to_fp16)[name = tensor<string, []>("op_3011_cast_fp16")];
tensor<bool, []> var_3013_interleave_0 = const()[name = tensor<string, []>("op_3013_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_3013_cast_fp16 = concat(axis = var_2969, interleave = var_3013_interleave_0, values = (var_3011_cast_fp16, x1_77_cast_fp16))[name = tensor<string, []>("op_3013_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_3014_cast_fp16 = mul(x = var_3013_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_3014_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_39_cast_fp16 = add(x = var_2999_cast_fp16, y = var_3014_cast_fp16)[name = tensor<string, []>("q_embed_39_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_77_cast_fp16 = transpose(perm = k_77_perm_0, x = squeeze_58_cast_fp16)[name = tensor<string, []>("transpose_29")];
tensor<fp16, [1, 12, 256, 64]> var_3017_cast_fp16 = mul(x = k_77_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_3017_cast_fp16")];
tensor<int32, [4]> x1_79_begin_0 = const()[name = tensor<string, []>("x1_79_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_79_end_0 = const()[name = tensor<string, []>("x1_79_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_79_end_mask_0 = const()[name = tensor<string, []>("x1_79_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_79_cast_fp16 = slice_by_index(begin = x1_79_begin_0, end = x1_79_end_0, end_mask = x1_79_end_mask_0, x = k_77_cast_fp16)[name = tensor<string, []>("x1_79_cast_fp16")];
tensor<int32, [4]> x2_79_begin_0 = const()[name = tensor<string, []>("x2_79_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_79_end_0 = const()[name = tensor<string, []>("x2_79_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_79_end_mask_0 = const()[name = tensor<string, []>("x2_79_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_79_cast_fp16 = slice_by_index(begin = x2_79_begin_0, end = x2_79_end_0, end_mask = x2_79_end_mask_0, x = k_77_cast_fp16)[name = tensor<string, []>("x2_79_cast_fp16")];
tensor<fp16, []> const_159_promoted_to_fp16 = const()[name = tensor<string, []>("const_159_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_3029_cast_fp16 = mul(x = x2_79_cast_fp16, y = const_159_promoted_to_fp16)[name = tensor<string, []>("op_3029_cast_fp16")];
tensor<bool, []> var_3031_interleave_0 = const()[name = tensor<string, []>("op_3031_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_3031_cast_fp16 = concat(axis = var_2969, interleave = var_3031_interleave_0, values = (var_3029_cast_fp16, x1_79_cast_fp16))[name = tensor<string, []>("op_3031_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_3032_cast_fp16 = mul(x = var_3031_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_3032_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_39_cast_fp16 = add(x = var_3017_cast_fp16, y = var_3032_cast_fp16)[name = tensor<string, []>("k_embed_39_cast_fp16")];
tensor<bool, []> var_3037_transpose_x_1 = const()[name = tensor<string, []>("op_3037_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_3037_transpose_y_1 = const()[name = tensor<string, []>("op_3037_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_3037_cast_fp16 = matmul(transpose_x = var_3037_transpose_x_1, transpose_y = var_3037_transpose_y_1, x = q_embed_39_cast_fp16, y = k_embed_39_cast_fp16)[name = tensor<string, []>("op_3037_cast_fp16")];
tensor<fp16, []> var_3038_to_fp16 = const()[name = tensor<string, []>("op_3038_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_77_cast_fp16 = mul(x = var_3037_cast_fp16, y = var_3038_to_fp16)[name = tensor<string, []>("attn_weights_77_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_349_cast_fp16 = add(x = attn_weights_77_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_349_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_3041_cast_fp16 = softmax(axis = var_2969, x = input_349_cast_fp16)[name = tensor<string, []>("op_3041_cast_fp16")];
tensor<bool, []> attn_output_115_transpose_x_0 = const()[name = tensor<string, []>("attn_output_115_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_115_transpose_y_0 = const()[name = tensor<string, []>("attn_output_115_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_39_cast_fp16 = transpose(perm = value_39_perm_0, x = squeeze_59_cast_fp16)[name = tensor<string, []>("transpose_28")];
tensor<fp16, [1, 12, 256, 64]> attn_output_115_cast_fp16 = matmul(transpose_x = attn_output_115_transpose_x_0, transpose_y = attn_output_115_transpose_y_0, x = var_3041_cast_fp16, y = value_39_cast_fp16)[name = tensor<string, []>("attn_output_115_cast_fp16")];
tensor<int32, [4]> var_3045_perm_0 = const()[name = tensor<string, []>("op_3045_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_3047 = const()[name = tensor<string, []>("op_3047"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_3045_cast_fp16 = transpose(perm = var_3045_perm_0, x = attn_output_115_cast_fp16)[name = tensor<string, []>("transpose_27")];
tensor<fp16, [1, 256, 768]> var_3048_cast_fp16 = reshape(shape = var_3047, x = var_3045_cast_fp16)[name = tensor<string, []>("op_3048_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_19_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_19_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(587669760)))];
tensor<fp16, [1, 256, 768]> linear_77_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_19_attn_Wo_weight_to_fp16, x = var_3048_cast_fp16)[name = tensor<string, []>("linear_77_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_355_cast_fp16 = add(x = input_347_cast_fp16, y = linear_77_cast_fp16)[name = tensor<string, []>("input_355_cast_fp16")];
tensor<int32, [1]> input_357_axes_0 = const()[name = tensor<string, []>("input_357_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_19_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_19_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(588849472)))];
tensor<fp16, [1, 256, 768]> input_357_cast_fp16 = layer_norm(axes = input_357_axes_0, epsilon = var_2980_to_fp16, gamma = model_encoder_layers_19_mlp_norm_weight_to_fp16, x = input_355_cast_fp16)[name = tensor<string, []>("input_357_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_19_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_19_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(588851072)))];
tensor<fp16, [1, 256, 2304]> linear_78_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_19_mlp_Wi_weight_to_fp16, x = input_357_cast_fp16)[name = tensor<string, []>("linear_78_cast_fp16")];
tensor<int32, [2]> var_3055_split_sizes_0 = const()[name = tensor<string, []>("op_3055_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_3055_axis_0 = const()[name = tensor<string, []>("op_3055_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_3055_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_3055_cast_fp16_1 = split(axis = var_3055_axis_0, split_sizes = var_3055_split_sizes_0, x = linear_78_cast_fp16)[name = tensor<string, []>("op_3055_cast_fp16")];
tensor<string, []> var_3057_mode_0 = const()[name = tensor<string, []>("op_3057_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_3057_cast_fp16 = gelu(mode = var_3057_mode_0, x = var_3055_cast_fp16_0)[name = tensor<string, []>("op_3057_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_361_cast_fp16 = mul(x = var_3057_cast_fp16, y = var_3055_cast_fp16_1)[name = tensor<string, []>("input_361_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_19_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_19_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(592390080)))];
tensor<fp16, [1, 256, 768]> linear_79_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_19_mlp_Wo_weight_to_fp16, x = input_361_cast_fp16)[name = tensor<string, []>("linear_79_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_365_cast_fp16 = add(x = input_355_cast_fp16, y = linear_79_cast_fp16)[name = tensor<string, []>("input_365_cast_fp16")];
tensor<int32, []> var_3066 = const()[name = tensor<string, []>("op_3066"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_41_axes_0 = const()[name = tensor<string, []>("hidden_states_41_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_20_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_20_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(594159616)))];
tensor<fp16, []> var_3077_to_fp16 = const()[name = tensor<string, []>("op_3077_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_41_cast_fp16 = layer_norm(axes = hidden_states_41_axes_0, epsilon = var_3077_to_fp16, gamma = model_encoder_layers_20_attn_norm_weight_to_fp16, x = input_365_cast_fp16)[name = tensor<string, []>("hidden_states_41_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_20_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_20_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(594161216)))];
tensor<fp16, [1, 256, 2304]> linear_80_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_20_attn_Wqkv_weight_to_fp16, x = hidden_states_41_cast_fp16)[name = tensor<string, []>("linear_80_cast_fp16")];
tensor<int32, [5]> var_3084 = const()[name = tensor<string, []>("op_3084"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_83_cast_fp16 = reshape(shape = var_3084, x = linear_80_cast_fp16)[name = tensor<string, []>("qkv_83_cast_fp16")];
tensor<int32, [3]> var_3086_split_sizes_0 = const()[name = tensor<string, []>("op_3086_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_3086_axis_0 = const()[name = tensor<string, []>("op_3086_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_3086_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_3086_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_3086_cast_fp16_2 = split(axis = var_3086_axis_0, split_sizes = var_3086_split_sizes_0, x = qkv_83_cast_fp16)[name = tensor<string, []>("op_3086_cast_fp16")];
tensor<int32, [1]> squeeze_60_axes_0 = const()[name = tensor<string, []>("squeeze_60_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_60_cast_fp16 = squeeze(axes = squeeze_60_axes_0, x = var_3086_cast_fp16_0)[name = tensor<string, []>("squeeze_60_cast_fp16")];
tensor<int32, [1]> squeeze_61_axes_0 = const()[name = tensor<string, []>("squeeze_61_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_61_cast_fp16 = squeeze(axes = squeeze_61_axes_0, x = var_3086_cast_fp16_1)[name = tensor<string, []>("squeeze_61_cast_fp16")];
tensor<int32, [1]> squeeze_62_axes_0 = const()[name = tensor<string, []>("squeeze_62_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_62_cast_fp16 = squeeze(axes = squeeze_62_axes_0, x = var_3086_cast_fp16_2)[name = tensor<string, []>("squeeze_62_cast_fp16")];
tensor<int32, [4]> q_81_perm_0 = const()[name = tensor<string, []>("q_81_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_81_perm_0 = const()[name = tensor<string, []>("k_81_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_41_perm_0 = const()[name = tensor<string, []>("value_41_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_81_cast_fp16 = transpose(perm = q_81_perm_0, x = squeeze_60_cast_fp16)[name = tensor<string, []>("transpose_26")];
tensor<fp16, [1, 12, 256, 64]> var_3096_cast_fp16 = mul(x = q_81_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_3096_cast_fp16")];
tensor<int32, [4]> x1_81_begin_0 = const()[name = tensor<string, []>("x1_81_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_81_end_0 = const()[name = tensor<string, []>("x1_81_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_81_end_mask_0 = const()[name = tensor<string, []>("x1_81_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_81_cast_fp16 = slice_by_index(begin = x1_81_begin_0, end = x1_81_end_0, end_mask = x1_81_end_mask_0, x = q_81_cast_fp16)[name = tensor<string, []>("x1_81_cast_fp16")];
tensor<int32, [4]> x2_81_begin_0 = const()[name = tensor<string, []>("x2_81_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_81_end_0 = const()[name = tensor<string, []>("x2_81_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_81_end_mask_0 = const()[name = tensor<string, []>("x2_81_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_81_cast_fp16 = slice_by_index(begin = x2_81_begin_0, end = x2_81_end_0, end_mask = x2_81_end_mask_0, x = q_81_cast_fp16)[name = tensor<string, []>("x2_81_cast_fp16")];
tensor<fp16, []> const_164_promoted_to_fp16 = const()[name = tensor<string, []>("const_164_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_3108_cast_fp16 = mul(x = x2_81_cast_fp16, y = const_164_promoted_to_fp16)[name = tensor<string, []>("op_3108_cast_fp16")];
tensor<bool, []> var_3110_interleave_0 = const()[name = tensor<string, []>("op_3110_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_3110_cast_fp16 = concat(axis = var_3066, interleave = var_3110_interleave_0, values = (var_3108_cast_fp16, x1_81_cast_fp16))[name = tensor<string, []>("op_3110_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_3111_cast_fp16 = mul(x = var_3110_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_3111_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_41_cast_fp16 = add(x = var_3096_cast_fp16, y = var_3111_cast_fp16)[name = tensor<string, []>("q_embed_41_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_81_cast_fp16 = transpose(perm = k_81_perm_0, x = squeeze_61_cast_fp16)[name = tensor<string, []>("transpose_25")];
tensor<fp16, [1, 12, 256, 64]> var_3114_cast_fp16 = mul(x = k_81_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_3114_cast_fp16")];
tensor<int32, [4]> x1_83_begin_0 = const()[name = tensor<string, []>("x1_83_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_83_end_0 = const()[name = tensor<string, []>("x1_83_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_83_end_mask_0 = const()[name = tensor<string, []>("x1_83_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_83_cast_fp16 = slice_by_index(begin = x1_83_begin_0, end = x1_83_end_0, end_mask = x1_83_end_mask_0, x = k_81_cast_fp16)[name = tensor<string, []>("x1_83_cast_fp16")];
tensor<int32, [4]> x2_83_begin_0 = const()[name = tensor<string, []>("x2_83_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_83_end_0 = const()[name = tensor<string, []>("x2_83_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_83_end_mask_0 = const()[name = tensor<string, []>("x2_83_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_83_cast_fp16 = slice_by_index(begin = x2_83_begin_0, end = x2_83_end_0, end_mask = x2_83_end_mask_0, x = k_81_cast_fp16)[name = tensor<string, []>("x2_83_cast_fp16")];
tensor<fp16, []> const_167_promoted_to_fp16 = const()[name = tensor<string, []>("const_167_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_3126_cast_fp16 = mul(x = x2_83_cast_fp16, y = const_167_promoted_to_fp16)[name = tensor<string, []>("op_3126_cast_fp16")];
tensor<bool, []> var_3128_interleave_0 = const()[name = tensor<string, []>("op_3128_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_3128_cast_fp16 = concat(axis = var_3066, interleave = var_3128_interleave_0, values = (var_3126_cast_fp16, x1_83_cast_fp16))[name = tensor<string, []>("op_3128_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_3129_cast_fp16 = mul(x = var_3128_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_3129_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_41_cast_fp16 = add(x = var_3114_cast_fp16, y = var_3129_cast_fp16)[name = tensor<string, []>("k_embed_41_cast_fp16")];
tensor<bool, []> var_3134_transpose_x_1 = const()[name = tensor<string, []>("op_3134_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_3134_transpose_y_1 = const()[name = tensor<string, []>("op_3134_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_3134_cast_fp16 = matmul(transpose_x = var_3134_transpose_x_1, transpose_y = var_3134_transpose_y_1, x = q_embed_41_cast_fp16, y = k_embed_41_cast_fp16)[name = tensor<string, []>("op_3134_cast_fp16")];
tensor<fp16, []> var_3135_to_fp16 = const()[name = tensor<string, []>("op_3135_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_81_cast_fp16 = mul(x = var_3134_cast_fp16, y = var_3135_to_fp16)[name = tensor<string, []>("attn_weights_81_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_367_cast_fp16 = add(x = attn_weights_81_cast_fp16, y = attention_mask_cast_fp16)[name = tensor<string, []>("input_367_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_3138_cast_fp16 = softmax(axis = var_3066, x = input_367_cast_fp16)[name = tensor<string, []>("op_3138_cast_fp16")];
tensor<bool, []> attn_output_121_transpose_x_0 = const()[name = tensor<string, []>("attn_output_121_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_121_transpose_y_0 = const()[name = tensor<string, []>("attn_output_121_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_41_cast_fp16 = transpose(perm = value_41_perm_0, x = squeeze_62_cast_fp16)[name = tensor<string, []>("transpose_24")];
tensor<fp16, [1, 12, 256, 64]> attn_output_121_cast_fp16 = matmul(transpose_x = attn_output_121_transpose_x_0, transpose_y = attn_output_121_transpose_y_0, x = var_3138_cast_fp16, y = value_41_cast_fp16)[name = tensor<string, []>("attn_output_121_cast_fp16")];
tensor<int32, [4]> var_3142_perm_0 = const()[name = tensor<string, []>("op_3142_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_3144 = const()[name = tensor<string, []>("op_3144"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_3142_cast_fp16 = transpose(perm = var_3142_perm_0, x = attn_output_121_cast_fp16)[name = tensor<string, []>("transpose_23")];
tensor<fp16, [1, 256, 768]> var_3145_cast_fp16 = reshape(shape = var_3144, x = var_3142_cast_fp16)[name = tensor<string, []>("op_3145_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_20_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_20_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(597700224)))];
tensor<fp16, [1, 256, 768]> linear_81_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_20_attn_Wo_weight_to_fp16, x = var_3145_cast_fp16)[name = tensor<string, []>("linear_81_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_373_cast_fp16 = add(x = input_365_cast_fp16, y = linear_81_cast_fp16)[name = tensor<string, []>("input_373_cast_fp16")];
tensor<int32, [1]> input_375_axes_0 = const()[name = tensor<string, []>("input_375_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_20_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_20_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(598879936)))];
tensor<fp16, [1, 256, 768]> input_375_cast_fp16 = layer_norm(axes = input_375_axes_0, epsilon = var_3077_to_fp16, gamma = model_encoder_layers_20_mlp_norm_weight_to_fp16, x = input_373_cast_fp16)[name = tensor<string, []>("input_375_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_20_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_20_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(598881536)))];
tensor<fp16, [1, 256, 2304]> linear_82_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_20_mlp_Wi_weight_to_fp16, x = input_375_cast_fp16)[name = tensor<string, []>("linear_82_cast_fp16")];
tensor<int32, [2]> var_3152_split_sizes_0 = const()[name = tensor<string, []>("op_3152_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_3152_axis_0 = const()[name = tensor<string, []>("op_3152_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_3152_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_3152_cast_fp16_1 = split(axis = var_3152_axis_0, split_sizes = var_3152_split_sizes_0, x = linear_82_cast_fp16)[name = tensor<string, []>("op_3152_cast_fp16")];
tensor<string, []> var_3154_mode_0 = const()[name = tensor<string, []>("op_3154_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_3154_cast_fp16 = gelu(mode = var_3154_mode_0, x = var_3152_cast_fp16_0)[name = tensor<string, []>("op_3154_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_379_cast_fp16 = mul(x = var_3154_cast_fp16, y = var_3152_cast_fp16_1)[name = tensor<string, []>("input_379_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_20_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_20_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(602420544)))];
tensor<fp16, [1, 256, 768]> linear_83_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_20_mlp_Wo_weight_to_fp16, x = input_379_cast_fp16)[name = tensor<string, []>("linear_83_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_383_cast_fp16 = add(x = input_373_cast_fp16, y = linear_83_cast_fp16)[name = tensor<string, []>("input_383_cast_fp16")];
tensor<int32, []> var_3163 = const()[name = tensor<string, []>("op_3163"), val = tensor<int32, []>(-1)];
tensor<int32, [1]> hidden_states_axes_0 = const()[name = tensor<string, []>("hidden_states_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_21_attn_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_21_attn_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(604190080)))];
tensor<fp16, []> var_3174_to_fp16 = const()[name = tensor<string, []>("op_3174_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> hidden_states_cast_fp16 = layer_norm(axes = hidden_states_axes_0, epsilon = var_3174_to_fp16, gamma = model_encoder_layers_21_attn_norm_weight_to_fp16, x = input_383_cast_fp16)[name = tensor<string, []>("hidden_states_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_21_attn_Wqkv_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_21_attn_Wqkv_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(604191680)))];
tensor<fp16, [1, 256, 2304]> linear_84_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_21_attn_Wqkv_weight_to_fp16, x = hidden_states_cast_fp16)[name = tensor<string, []>("linear_84_cast_fp16")];
tensor<int32, [5]> var_3181 = const()[name = tensor<string, []>("op_3181"), val = tensor<int32, [5]>([1, 256, 3, -1, 64])];
tensor<fp16, [1, 256, 3, 12, 64]> qkv_87_cast_fp16 = reshape(shape = var_3181, x = linear_84_cast_fp16)[name = tensor<string, []>("qkv_87_cast_fp16")];
tensor<int32, [3]> var_3183_split_sizes_0 = const()[name = tensor<string, []>("op_3183_split_sizes_0"), val = tensor<int32, [3]>([1, 1, 1])];
tensor<int32, []> var_3183_axis_0 = const()[name = tensor<string, []>("op_3183_axis_0"), val = tensor<int32, []>(-3)];
tensor<fp16, [1, 256, 1, 12, 64]> var_3183_cast_fp16_0, tensor<fp16, [1, 256, 1, 12, 64]> var_3183_cast_fp16_1, tensor<fp16, [1, 256, 1, 12, 64]> var_3183_cast_fp16_2 = split(axis = var_3183_axis_0, split_sizes = var_3183_split_sizes_0, x = qkv_87_cast_fp16)[name = tensor<string, []>("op_3183_cast_fp16")];
tensor<int32, [1]> squeeze_63_axes_0 = const()[name = tensor<string, []>("squeeze_63_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_63_cast_fp16 = squeeze(axes = squeeze_63_axes_0, x = var_3183_cast_fp16_0)[name = tensor<string, []>("squeeze_63_cast_fp16")];
tensor<int32, [1]> squeeze_64_axes_0 = const()[name = tensor<string, []>("squeeze_64_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_64_cast_fp16 = squeeze(axes = squeeze_64_axes_0, x = var_3183_cast_fp16_1)[name = tensor<string, []>("squeeze_64_cast_fp16")];
tensor<int32, [1]> squeeze_65_axes_0 = const()[name = tensor<string, []>("squeeze_65_axes_0"), val = tensor<int32, [1]>([-3])];
tensor<fp16, [1, 256, 12, 64]> squeeze_65_cast_fp16 = squeeze(axes = squeeze_65_axes_0, x = var_3183_cast_fp16_2)[name = tensor<string, []>("squeeze_65_cast_fp16")];
tensor<int32, [4]> q_85_perm_0 = const()[name = tensor<string, []>("q_85_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> k_85_perm_0 = const()[name = tensor<string, []>("k_85_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [4]> value_perm_0 = const()[name = tensor<string, []>("value_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<fp16, [1, 12, 256, 64]> q_85_cast_fp16 = transpose(perm = q_85_perm_0, x = squeeze_63_cast_fp16)[name = tensor<string, []>("transpose_22")];
tensor<fp16, [1, 12, 256, 64]> var_3193_cast_fp16 = mul(x = q_85_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_3193_cast_fp16")];
tensor<int32, [4]> x1_85_begin_0 = const()[name = tensor<string, []>("x1_85_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_85_end_0 = const()[name = tensor<string, []>("x1_85_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_85_end_mask_0 = const()[name = tensor<string, []>("x1_85_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_85_cast_fp16 = slice_by_index(begin = x1_85_begin_0, end = x1_85_end_0, end_mask = x1_85_end_mask_0, x = q_85_cast_fp16)[name = tensor<string, []>("x1_85_cast_fp16")];
tensor<int32, [4]> x2_85_begin_0 = const()[name = tensor<string, []>("x2_85_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_85_end_0 = const()[name = tensor<string, []>("x2_85_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_85_end_mask_0 = const()[name = tensor<string, []>("x2_85_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_85_cast_fp16 = slice_by_index(begin = x2_85_begin_0, end = x2_85_end_0, end_mask = x2_85_end_mask_0, x = q_85_cast_fp16)[name = tensor<string, []>("x2_85_cast_fp16")];
tensor<fp16, []> const_172_promoted_to_fp16 = const()[name = tensor<string, []>("const_172_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_3205_cast_fp16 = mul(x = x2_85_cast_fp16, y = const_172_promoted_to_fp16)[name = tensor<string, []>("op_3205_cast_fp16")];
tensor<bool, []> var_3207_interleave_0 = const()[name = tensor<string, []>("op_3207_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_3207_cast_fp16 = concat(axis = var_3163, interleave = var_3207_interleave_0, values = (var_3205_cast_fp16, x1_85_cast_fp16))[name = tensor<string, []>("op_3207_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_3208_cast_fp16 = mul(x = var_3207_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_3208_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> q_embed_cast_fp16 = add(x = var_3193_cast_fp16, y = var_3208_cast_fp16)[name = tensor<string, []>("q_embed_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_85_cast_fp16 = transpose(perm = k_85_perm_0, x = squeeze_64_cast_fp16)[name = tensor<string, []>("transpose_21")];
tensor<fp16, [1, 12, 256, 64]> var_3211_cast_fp16 = mul(x = k_85_cast_fp16, y = cos_3_to_fp16)[name = tensor<string, []>("op_3211_cast_fp16")];
tensor<int32, [4]> x1_begin_0 = const()[name = tensor<string, []>("x1_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
tensor<int32, [4]> x1_end_0 = const()[name = tensor<string, []>("x1_end_0"), val = tensor<int32, [4]>([1, 12, 256, 32])];
tensor<bool, [4]> x1_end_mask_0 = const()[name = tensor<string, []>("x1_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
tensor<fp16, [1, 12, 256, 32]> x1_cast_fp16 = slice_by_index(begin = x1_begin_0, end = x1_end_0, end_mask = x1_end_mask_0, x = k_85_cast_fp16)[name = tensor<string, []>("x1_cast_fp16")];
tensor<int32, [4]> x2_begin_0 = const()[name = tensor<string, []>("x2_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 32])];
tensor<int32, [4]> x2_end_0 = const()[name = tensor<string, []>("x2_end_0"), val = tensor<int32, [4]>([1, 12, 256, 64])];
tensor<bool, [4]> x2_end_mask_0 = const()[name = tensor<string, []>("x2_end_mask_0"), val = tensor<bool, [4]>([true, true, true, true])];
tensor<fp16, [1, 12, 256, 32]> x2_cast_fp16 = slice_by_index(begin = x2_begin_0, end = x2_end_0, end_mask = x2_end_mask_0, x = k_85_cast_fp16)[name = tensor<string, []>("x2_cast_fp16")];
tensor<fp16, []> const_175_promoted_to_fp16 = const()[name = tensor<string, []>("const_175_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 12, 256, 32]> var_3223_cast_fp16 = mul(x = x2_cast_fp16, y = const_175_promoted_to_fp16)[name = tensor<string, []>("op_3223_cast_fp16")];
tensor<bool, []> var_3225_interleave_0 = const()[name = tensor<string, []>("op_3225_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> var_3225_cast_fp16 = concat(axis = var_3163, interleave = var_3225_interleave_0, values = (var_3223_cast_fp16, x1_cast_fp16))[name = tensor<string, []>("op_3225_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> var_3226_cast_fp16 = mul(x = var_3225_cast_fp16, y = sin_3_to_fp16)[name = tensor<string, []>("op_3226_cast_fp16")];
tensor<fp16, [1, 12, 256, 64]> k_embed_cast_fp16 = add(x = var_3211_cast_fp16, y = var_3226_cast_fp16)[name = tensor<string, []>("k_embed_cast_fp16")];
tensor<bool, []> var_3231_transpose_x_1 = const()[name = tensor<string, []>("op_3231_transpose_x_1"), val = tensor<bool, []>(false)];
tensor<bool, []> var_3231_transpose_y_1 = const()[name = tensor<string, []>("op_3231_transpose_y_1"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 12, 256, 256]> var_3231_cast_fp16 = matmul(transpose_x = var_3231_transpose_x_1, transpose_y = var_3231_transpose_y_1, x = q_embed_cast_fp16, y = k_embed_cast_fp16)[name = tensor<string, []>("op_3231_cast_fp16")];
tensor<fp16, []> var_3232_to_fp16 = const()[name = tensor<string, []>("op_3232_to_fp16"), val = tensor<fp16, []>(0x1p-3)];
tensor<fp16, [1, 12, 256, 256]> attn_weights_85_cast_fp16 = mul(x = var_3231_cast_fp16, y = var_3232_to_fp16)[name = tensor<string, []>("attn_weights_85_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> input_385_cast_fp16 = add(x = attn_weights_85_cast_fp16, y = attention_mask_3_cast_fp16)[name = tensor<string, []>("input_385_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> var_3235_cast_fp16 = softmax(axis = var_3163, x = input_385_cast_fp16)[name = tensor<string, []>("op_3235_cast_fp16")];
tensor<bool, []> attn_output_127_transpose_x_0 = const()[name = tensor<string, []>("attn_output_127_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> attn_output_127_transpose_y_0 = const()[name = tensor<string, []>("attn_output_127_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> value_cast_fp16 = transpose(perm = value_perm_0, x = squeeze_65_cast_fp16)[name = tensor<string, []>("transpose_20")];
tensor<fp16, [1, 12, 256, 64]> attn_output_127_cast_fp16 = matmul(transpose_x = attn_output_127_transpose_x_0, transpose_y = attn_output_127_transpose_y_0, x = var_3235_cast_fp16, y = value_cast_fp16)[name = tensor<string, []>("attn_output_127_cast_fp16")];
tensor<int32, [4]> var_3239_perm_0 = const()[name = tensor<string, []>("op_3239_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_3241 = const()[name = tensor<string, []>("op_3241"), val = tensor<int32, [3]>([1, 256, -1])];
tensor<fp16, [1, 256, 12, 64]> var_3239_cast_fp16 = transpose(perm = var_3239_perm_0, x = attn_output_127_cast_fp16)[name = tensor<string, []>("transpose_19")];
tensor<fp16, [1, 256, 768]> var_3242_cast_fp16 = reshape(shape = var_3241, x = var_3239_cast_fp16)[name = tensor<string, []>("op_3242_cast_fp16")];
tensor<fp16, [768, 768]> model_encoder_layers_21_attn_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_21_attn_Wo_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(607730688)))];
tensor<fp16, [1, 256, 768]> linear_85_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_21_attn_Wo_weight_to_fp16, x = var_3242_cast_fp16)[name = tensor<string, []>("linear_85_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_391_cast_fp16 = add(x = input_383_cast_fp16, y = linear_85_cast_fp16)[name = tensor<string, []>("input_391_cast_fp16")];
tensor<int32, [1]> input_393_axes_0 = const()[name = tensor<string, []>("input_393_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_layers_21_mlp_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_21_mlp_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(608910400)))];
tensor<fp16, [1, 256, 768]> input_393_cast_fp16 = layer_norm(axes = input_393_axes_0, epsilon = var_3174_to_fp16, gamma = model_encoder_layers_21_mlp_norm_weight_to_fp16, x = input_391_cast_fp16)[name = tensor<string, []>("input_393_cast_fp16")];
tensor<fp16, [2304, 768]> model_encoder_layers_21_mlp_Wi_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_21_mlp_Wi_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(608912000)))];
tensor<fp16, [1, 256, 2304]> linear_86_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_encoder_layers_21_mlp_Wi_weight_to_fp16, x = input_393_cast_fp16)[name = tensor<string, []>("linear_86_cast_fp16")];
tensor<int32, [2]> var_3249_split_sizes_0 = const()[name = tensor<string, []>("op_3249_split_sizes_0"), val = tensor<int32, [2]>([1152, 1152])];
tensor<int32, []> var_3249_axis_0 = const()[name = tensor<string, []>("op_3249_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 1152]> var_3249_cast_fp16_0, tensor<fp16, [1, 256, 1152]> var_3249_cast_fp16_1 = split(axis = var_3249_axis_0, split_sizes = var_3249_split_sizes_0, x = linear_86_cast_fp16)[name = tensor<string, []>("op_3249_cast_fp16")];
tensor<string, []> var_3251_mode_0 = const()[name = tensor<string, []>("op_3251_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256, 1152]> var_3251_cast_fp16 = gelu(mode = var_3251_mode_0, x = var_3249_cast_fp16_0)[name = tensor<string, []>("op_3251_cast_fp16")];
tensor<fp16, [1, 256, 1152]> input_397_cast_fp16 = mul(x = var_3251_cast_fp16, y = var_3249_cast_fp16_1)[name = tensor<string, []>("input_397_cast_fp16")];
tensor<fp16, [768, 1152]> model_encoder_layers_21_mlp_Wo_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_layers_21_mlp_Wo_weight_to_fp16"), val = tensor<fp16, [768, 1152]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(612451008)))];
tensor<fp16, [1, 256, 768]> linear_87_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = model_encoder_layers_21_mlp_Wo_weight_to_fp16, x = input_397_cast_fp16)[name = tensor<string, []>("linear_87_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_401_cast_fp16 = add(x = input_391_cast_fp16, y = linear_87_cast_fp16)[name = tensor<string, []>("input_401_cast_fp16")];
tensor<int32, [1]> x_89_axes_0 = const()[name = tensor<string, []>("x_89_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_encoder_final_norm_weight_to_fp16 = const()[name = tensor<string, []>("model_encoder_final_norm_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(614220544)))];
tensor<fp16, []> var_3257_to_fp16 = const()[name = tensor<string, []>("op_3257_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> x_89_cast_fp16 = layer_norm(axes = x_89_axes_0, epsilon = var_3257_to_fp16, gamma = model_encoder_final_norm_weight_to_fp16, x = input_401_cast_fp16)[name = tensor<string, []>("x_89_cast_fp16")];
tensor<string, []> question_type_to_fp16_dtype_0 = const()[name = tensor<string, []>("question_type_to_fp16_dtype_0"), val = tensor<string, []>("fp16")];
tensor<fp16, [768, 3]> transpose_0_to_fp16 = const()[name = tensor<string, []>("transpose_0_to_fp16"), val = tensor<fp16, [768, 3]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(614222144)))];
tensor<fp16, [1, 3]> question_type_to_fp16 = cast(dtype = question_type_to_fp16_dtype_0, x = question_type)[name = tensor<string, []>("cast_62")];
tensor<fp16, [1, 768]> var_3262_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = transpose_0_to_fp16, x = question_type_to_fp16)[name = tensor<string, []>("op_3262_cast_fp16")];
tensor<int32, [1]> var_3264_axes_0 = const()[name = tensor<string, []>("op_3264_axes_0"), val = tensor<int32, [1]>([1])];
tensor<fp16, [1, 1, 768]> var_3264_cast_fp16 = expand_dims(axes = var_3264_axes_0, x = var_3262_cast_fp16)[name = tensor<string, []>("op_3264_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_403_cast_fp16 = add(x = x_89_cast_fp16, y = var_3264_cast_fp16)[name = tensor<string, []>("input_403_cast_fp16")];
tensor<int32, [1]> h_1_axes_0 = const()[name = tensor<string, []>("h_1_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_head_layers_0_norm1_weight_to_fp16 = const()[name = tensor<string, []>("model_head_layers_0_norm1_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(614226816)))];
tensor<fp16, [768]> model_head_layers_0_norm1_bias_to_fp16 = const()[name = tensor<string, []>("model_head_layers_0_norm1_bias_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(614228416)))];
tensor<fp16, []> var_3268_to_fp16 = const()[name = tensor<string, []>("op_3268_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> h_1_cast_fp16 = layer_norm(axes = h_1_axes_0, beta = model_head_layers_0_norm1_bias_to_fp16, epsilon = var_3268_to_fp16, gamma = model_head_layers_0_norm1_weight_to_fp16, x = input_403_cast_fp16)[name = tensor<string, []>("h_1_cast_fp16")];
tensor<fp16, [2304, 768]> model_head_layers_0_self_attn_in_proj_weight_to_fp16 = const()[name = tensor<string, []>("model_head_layers_0_self_attn_in_proj_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(614230016)))];
tensor<fp16, [2304]> model_head_layers_0_self_attn_in_proj_bias_to_fp16 = const()[name = tensor<string, []>("model_head_layers_0_self_attn_in_proj_bias_to_fp16"), val = tensor<fp16, [2304]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(617769024)))];
tensor<fp16, [1, 256, 2304]> linear_88_cast_fp16 = linear(bias = model_head_layers_0_self_attn_in_proj_bias_to_fp16, weight = model_head_layers_0_self_attn_in_proj_weight_to_fp16, x = h_1_cast_fp16)[name = tensor<string, []>("linear_88_cast_fp16")];
tensor<int32, [3]> var_3291_split_sizes_0 = const()[name = tensor<string, []>("op_3291_split_sizes_0"), val = tensor<int32, [3]>([768, 768, 768])];
tensor<int32, []> var_3291_axis_0 = const()[name = tensor<string, []>("op_3291_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 768]> var_3291_cast_fp16_0, tensor<fp16, [1, 256, 768]> var_3291_cast_fp16_1, tensor<fp16, [1, 256, 768]> var_3291_cast_fp16_2 = split(axis = var_3291_axis_0, split_sizes = var_3291_split_sizes_0, x = linear_88_cast_fp16)[name = tensor<string, []>("op_3291_cast_fp16")];
tensor<int32, [4]> var_3300 = const()[name = tensor<string, []>("op_3300"), val = tensor<int32, [4]>([1, 256, 12, 64])];
tensor<fp16, [1, 256, 12, 64]> var_3301_cast_fp16 = reshape(shape = var_3300, x = var_3291_cast_fp16_0)[name = tensor<string, []>("op_3301_cast_fp16")];
tensor<int32, [4]> var_3306 = const()[name = tensor<string, []>("op_3306"), val = tensor<int32, [4]>([1, 256, 12, 64])];
tensor<fp16, [1, 256, 12, 64]> var_3307_cast_fp16 = reshape(shape = var_3306, x = var_3291_cast_fp16_1)[name = tensor<string, []>("op_3307_cast_fp16")];
tensor<int32, [4]> var_3312 = const()[name = tensor<string, []>("op_3312"), val = tensor<int32, [4]>([1, 256, 12, 64])];
tensor<fp16, [1, 256, 12, 64]> var_3313_cast_fp16 = reshape(shape = var_3312, x = var_3291_cast_fp16_2)[name = tensor<string, []>("op_3313_cast_fp16")];
tensor<int32, [4]> v_3_perm_0 = const()[name = tensor<string, []>("v_3_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<bool, []> var_3320_transpose_x_0 = const()[name = tensor<string, []>("op_3320_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> var_3320_transpose_y_0 = const()[name = tensor<string, []>("op_3320_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<int32, [4]> transpose_7_perm_0 = const()[name = tensor<string, []>("transpose_7_perm_0"), val = tensor<int32, [4]>([0, 2, -3, -1])];
tensor<int32, [4]> transpose_8_perm_0 = const()[name = tensor<string, []>("transpose_8_perm_0"), val = tensor<int32, [4]>([0, 2, -1, -3])];
tensor<fp16, [1, 12, 64, 256]> transpose_8 = transpose(perm = transpose_8_perm_0, x = var_3307_cast_fp16)[name = tensor<string, []>("transpose_16")];
tensor<fp16, [1, 12, 256, 64]> transpose_7 = transpose(perm = transpose_7_perm_0, x = var_3301_cast_fp16)[name = tensor<string, []>("transpose_17")];
tensor<fp16, [1, 12, 256, 256]> var_3320_cast_fp16 = matmul(transpose_x = var_3320_transpose_x_0, transpose_y = var_3320_transpose_y_0, x = transpose_7, y = transpose_8)[name = tensor<string, []>("op_3320_cast_fp16")];
tensor<fp16, [1]> var_3322_to_fp16 = const()[name = tensor<string, []>("op_3322_to_fp16"), val = tensor<fp16, [1]>([0x1p-3])];
tensor<fp16, [1, 12, 256, 256]> var_3323_cast_fp16 = mul(x = var_3320_cast_fp16, y = var_3322_to_fp16)[name = tensor<string, []>("op_3323_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> scores_1_cast_fp16 = add(x = var_3323_cast_fp16, y = attention_mask_3_cast_fp16)[name = tensor<string, []>("scores_1_cast_fp16")];
tensor<int32, []> var_3326 = const()[name = tensor<string, []>("op_3326"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 12, 256, 256]> weights_1_cast_fp16 = softmax(axis = var_3326, x = scores_1_cast_fp16)[name = tensor<string, []>("weights_1_cast_fp16")];
tensor<bool, []> var_3329_transpose_x_0 = const()[name = tensor<string, []>("op_3329_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> var_3329_transpose_y_0 = const()[name = tensor<string, []>("op_3329_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> v_3_cast_fp16 = transpose(perm = v_3_perm_0, x = var_3313_cast_fp16)[name = tensor<string, []>("transpose_18")];
tensor<fp16, [1, 12, 256, 64]> var_3329_cast_fp16 = matmul(transpose_x = var_3329_transpose_x_0, transpose_y = var_3329_transpose_y_0, x = weights_1_cast_fp16, y = v_3_cast_fp16)[name = tensor<string, []>("op_3329_cast_fp16")];
tensor<int32, [4]> var_3332_perm_0 = const()[name = tensor<string, []>("op_3332_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_3333 = const()[name = tensor<string, []>("op_3333"), val = tensor<int32, [3]>([1, 256, 768])];
tensor<fp16, [1, 256, 12, 64]> var_3332_cast_fp16 = transpose(perm = var_3332_perm_0, x = var_3329_cast_fp16)[name = tensor<string, []>("transpose_15")];
tensor<fp16, [1, 256, 768]> input_405_cast_fp16 = reshape(shape = var_3333, x = var_3332_cast_fp16)[name = tensor<string, []>("input_405_cast_fp16")];
tensor<fp16, [768, 768]> model_head_layers_0_self_attn_out_proj_weight_to_fp16 = const()[name = tensor<string, []>("model_head_layers_0_self_attn_out_proj_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(617773696)))];
tensor<fp16, [768]> model_head_layers_0_self_attn_out_proj_bias_to_fp16 = const()[name = tensor<string, []>("model_head_layers_0_self_attn_out_proj_bias_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(618953408)))];
tensor<fp16, [1, 256, 768]> linear_89_cast_fp16 = linear(bias = model_head_layers_0_self_attn_out_proj_bias_to_fp16, weight = model_head_layers_0_self_attn_out_proj_weight_to_fp16, x = input_405_cast_fp16)[name = tensor<string, []>("linear_89_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_407_cast_fp16 = add(x = input_403_cast_fp16, y = linear_89_cast_fp16)[name = tensor<string, []>("input_407_cast_fp16")];
tensor<int32, [1]> input_409_axes_0 = const()[name = tensor<string, []>("input_409_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_head_layers_0_norm2_weight_to_fp16 = const()[name = tensor<string, []>("model_head_layers_0_norm2_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(618955008)))];
tensor<fp16, [768]> model_head_layers_0_norm2_bias_to_fp16 = const()[name = tensor<string, []>("model_head_layers_0_norm2_bias_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(618956608)))];
tensor<fp16, []> var_3339_to_fp16 = const()[name = tensor<string, []>("op_3339_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> input_409_cast_fp16 = layer_norm(axes = input_409_axes_0, beta = model_head_layers_0_norm2_bias_to_fp16, epsilon = var_3339_to_fp16, gamma = model_head_layers_0_norm2_weight_to_fp16, x = input_407_cast_fp16)[name = tensor<string, []>("input_409_cast_fp16")];
tensor<fp16, [3072, 768]> model_head_layers_0_linear1_weight_to_fp16 = const()[name = tensor<string, []>("model_head_layers_0_linear1_weight_to_fp16"), val = tensor<fp16, [3072, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(618958208)))];
tensor<fp16, [3072]> model_head_layers_0_linear1_bias_to_fp16 = const()[name = tensor<string, []>("model_head_layers_0_linear1_bias_to_fp16"), val = tensor<fp16, [3072]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(623676864)))];
tensor<fp16, [1, 256, 3072]> linear_90_cast_fp16 = linear(bias = model_head_layers_0_linear1_bias_to_fp16, weight = model_head_layers_0_linear1_weight_to_fp16, x = input_409_cast_fp16)[name = tensor<string, []>("linear_90_cast_fp16")];
tensor<fp16, [1, 256, 3072]> input_411_cast_fp16 = relu(x = linear_90_cast_fp16)[name = tensor<string, []>("input_411_cast_fp16")];
tensor<fp16, [768, 3072]> model_head_layers_0_linear2_weight_to_fp16 = const()[name = tensor<string, []>("model_head_layers_0_linear2_weight_to_fp16"), val = tensor<fp16, [768, 3072]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(623683072)))];
tensor<fp16, [768]> model_head_layers_0_linear2_bias_to_fp16 = const()[name = tensor<string, []>("model_head_layers_0_linear2_bias_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(628401728)))];
tensor<fp16, [1, 256, 768]> linear_91_cast_fp16 = linear(bias = model_head_layers_0_linear2_bias_to_fp16, weight = model_head_layers_0_linear2_weight_to_fp16, x = input_411_cast_fp16)[name = tensor<string, []>("linear_91_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_413_cast_fp16 = add(x = input_407_cast_fp16, y = linear_91_cast_fp16)[name = tensor<string, []>("input_413_cast_fp16")];
tensor<int32, [1]> h_axes_0 = const()[name = tensor<string, []>("h_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_head_layers_1_norm1_weight_to_fp16 = const()[name = tensor<string, []>("model_head_layers_1_norm1_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(628403328)))];
tensor<fp16, [768]> model_head_layers_1_norm1_bias_to_fp16 = const()[name = tensor<string, []>("model_head_layers_1_norm1_bias_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(628404928)))];
tensor<fp16, []> var_3349_to_fp16 = const()[name = tensor<string, []>("op_3349_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> h_cast_fp16 = layer_norm(axes = h_axes_0, beta = model_head_layers_1_norm1_bias_to_fp16, epsilon = var_3349_to_fp16, gamma = model_head_layers_1_norm1_weight_to_fp16, x = input_413_cast_fp16)[name = tensor<string, []>("h_cast_fp16")];
tensor<fp16, [2304, 768]> model_head_layers_1_self_attn_in_proj_weight_to_fp16 = const()[name = tensor<string, []>("model_head_layers_1_self_attn_in_proj_weight_to_fp16"), val = tensor<fp16, [2304, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(628406528)))];
tensor<fp16, [2304]> model_head_layers_1_self_attn_in_proj_bias_to_fp16 = const()[name = tensor<string, []>("model_head_layers_1_self_attn_in_proj_bias_to_fp16"), val = tensor<fp16, [2304]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(631945536)))];
tensor<fp16, [1, 256, 2304]> linear_92_cast_fp16 = linear(bias = model_head_layers_1_self_attn_in_proj_bias_to_fp16, weight = model_head_layers_1_self_attn_in_proj_weight_to_fp16, x = h_cast_fp16)[name = tensor<string, []>("linear_92_cast_fp16")];
tensor<int32, [3]> var_3372_split_sizes_0 = const()[name = tensor<string, []>("op_3372_split_sizes_0"), val = tensor<int32, [3]>([768, 768, 768])];
tensor<int32, []> var_3372_axis_0 = const()[name = tensor<string, []>("op_3372_axis_0"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 256, 768]> var_3372_cast_fp16_0, tensor<fp16, [1, 256, 768]> var_3372_cast_fp16_1, tensor<fp16, [1, 256, 768]> var_3372_cast_fp16_2 = split(axis = var_3372_axis_0, split_sizes = var_3372_split_sizes_0, x = linear_92_cast_fp16)[name = tensor<string, []>("op_3372_cast_fp16")];
tensor<int32, [4]> var_3381 = const()[name = tensor<string, []>("op_3381"), val = tensor<int32, [4]>([1, 256, 12, 64])];
tensor<fp16, [1, 256, 12, 64]> var_3382_cast_fp16 = reshape(shape = var_3381, x = var_3372_cast_fp16_0)[name = tensor<string, []>("op_3382_cast_fp16")];
tensor<int32, [4]> var_3387 = const()[name = tensor<string, []>("op_3387"), val = tensor<int32, [4]>([1, 256, 12, 64])];
tensor<fp16, [1, 256, 12, 64]> var_3388_cast_fp16 = reshape(shape = var_3387, x = var_3372_cast_fp16_1)[name = tensor<string, []>("op_3388_cast_fp16")];
tensor<int32, [4]> var_3393 = const()[name = tensor<string, []>("op_3393"), val = tensor<int32, [4]>([1, 256, 12, 64])];
tensor<fp16, [1, 256, 12, 64]> var_3394_cast_fp16 = reshape(shape = var_3393, x = var_3372_cast_fp16_2)[name = tensor<string, []>("op_3394_cast_fp16")];
tensor<int32, [4]> v_perm_0 = const()[name = tensor<string, []>("v_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<bool, []> var_3401_transpose_x_0 = const()[name = tensor<string, []>("op_3401_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> var_3401_transpose_y_0 = const()[name = tensor<string, []>("op_3401_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<int32, [4]> transpose_9_perm_0 = const()[name = tensor<string, []>("transpose_9_perm_0"), val = tensor<int32, [4]>([0, 2, -3, -1])];
tensor<int32, [4]> transpose_10_perm_0 = const()[name = tensor<string, []>("transpose_10_perm_0"), val = tensor<int32, [4]>([0, 2, -1, -3])];
tensor<fp16, [1, 12, 64, 256]> transpose_10 = transpose(perm = transpose_10_perm_0, x = var_3388_cast_fp16)[name = tensor<string, []>("transpose_12")];
tensor<fp16, [1, 12, 256, 64]> transpose_9 = transpose(perm = transpose_9_perm_0, x = var_3382_cast_fp16)[name = tensor<string, []>("transpose_13")];
tensor<fp16, [1, 12, 256, 256]> var_3401_cast_fp16 = matmul(transpose_x = var_3401_transpose_x_0, transpose_y = var_3401_transpose_y_0, x = transpose_9, y = transpose_10)[name = tensor<string, []>("op_3401_cast_fp16")];
tensor<fp16, [1]> var_3403_to_fp16 = const()[name = tensor<string, []>("op_3403_to_fp16"), val = tensor<fp16, [1]>([0x1p-3])];
tensor<fp16, [1, 12, 256, 256]> var_3404_cast_fp16 = mul(x = var_3401_cast_fp16, y = var_3403_to_fp16)[name = tensor<string, []>("op_3404_cast_fp16")];
tensor<fp16, [1, 12, 256, 256]> scores_cast_fp16 = add(x = var_3404_cast_fp16, y = attention_mask_3_cast_fp16)[name = tensor<string, []>("scores_cast_fp16")];
tensor<int32, []> var_3407 = const()[name = tensor<string, []>("op_3407"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 12, 256, 256]> weights_cast_fp16 = softmax(axis = var_3407, x = scores_cast_fp16)[name = tensor<string, []>("weights_cast_fp16")];
tensor<bool, []> var_3410_transpose_x_0 = const()[name = tensor<string, []>("op_3410_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> var_3410_transpose_y_0 = const()[name = tensor<string, []>("op_3410_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 12, 256, 64]> v_cast_fp16 = transpose(perm = v_perm_0, x = var_3394_cast_fp16)[name = tensor<string, []>("transpose_14")];
tensor<fp16, [1, 12, 256, 64]> var_3410_cast_fp16 = matmul(transpose_x = var_3410_transpose_x_0, transpose_y = var_3410_transpose_y_0, x = weights_cast_fp16, y = v_cast_fp16)[name = tensor<string, []>("op_3410_cast_fp16")];
tensor<int32, [4]> var_3413_perm_0 = const()[name = tensor<string, []>("op_3413_perm_0"), val = tensor<int32, [4]>([0, 2, 1, 3])];
tensor<int32, [3]> var_3414 = const()[name = tensor<string, []>("op_3414"), val = tensor<int32, [3]>([1, 256, 768])];
tensor<fp16, [1, 256, 12, 64]> var_3413_cast_fp16 = transpose(perm = var_3413_perm_0, x = var_3410_cast_fp16)[name = tensor<string, []>("transpose_11")];
tensor<fp16, [1, 256, 768]> input_415_cast_fp16 = reshape(shape = var_3414, x = var_3413_cast_fp16)[name = tensor<string, []>("input_415_cast_fp16")];
tensor<fp16, [768, 768]> model_head_layers_1_self_attn_out_proj_weight_to_fp16 = const()[name = tensor<string, []>("model_head_layers_1_self_attn_out_proj_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(631950208)))];
tensor<fp16, [768]> model_head_layers_1_self_attn_out_proj_bias_to_fp16 = const()[name = tensor<string, []>("model_head_layers_1_self_attn_out_proj_bias_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(633129920)))];
tensor<fp16, [1, 256, 768]> linear_93_cast_fp16 = linear(bias = model_head_layers_1_self_attn_out_proj_bias_to_fp16, weight = model_head_layers_1_self_attn_out_proj_weight_to_fp16, x = input_415_cast_fp16)[name = tensor<string, []>("linear_93_cast_fp16")];
tensor<fp16, [1, 256, 768]> input_417_cast_fp16 = add(x = input_413_cast_fp16, y = linear_93_cast_fp16)[name = tensor<string, []>("input_417_cast_fp16")];
tensor<int32, [1]> input_419_axes_0 = const()[name = tensor<string, []>("input_419_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_head_layers_1_norm2_weight_to_fp16 = const()[name = tensor<string, []>("model_head_layers_1_norm2_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(633131520)))];
tensor<fp16, [768]> model_head_layers_1_norm2_bias_to_fp16 = const()[name = tensor<string, []>("model_head_layers_1_norm2_bias_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(633133120)))];
tensor<fp16, []> var_3420_to_fp16 = const()[name = tensor<string, []>("op_3420_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 256, 768]> input_419_cast_fp16 = layer_norm(axes = input_419_axes_0, beta = model_head_layers_1_norm2_bias_to_fp16, epsilon = var_3420_to_fp16, gamma = model_head_layers_1_norm2_weight_to_fp16, x = input_417_cast_fp16)[name = tensor<string, []>("input_419_cast_fp16")];
tensor<fp16, [3072, 768]> model_head_layers_1_linear1_weight_to_fp16 = const()[name = tensor<string, []>("model_head_layers_1_linear1_weight_to_fp16"), val = tensor<fp16, [3072, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(633134720)))];
tensor<fp16, [3072]> model_head_layers_1_linear1_bias_to_fp16 = const()[name = tensor<string, []>("model_head_layers_1_linear1_bias_to_fp16"), val = tensor<fp16, [3072]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(637853376)))];
tensor<fp16, [1, 256, 3072]> linear_94_cast_fp16 = linear(bias = model_head_layers_1_linear1_bias_to_fp16, weight = model_head_layers_1_linear1_weight_to_fp16, x = input_419_cast_fp16)[name = tensor<string, []>("linear_94_cast_fp16")];
tensor<fp16, [1, 256, 3072]> input_421_cast_fp16 = relu(x = linear_94_cast_fp16)[name = tensor<string, []>("input_421_cast_fp16")];
tensor<fp16, [768, 3072]> model_head_layers_1_linear2_weight_to_fp16 = const()[name = tensor<string, []>("model_head_layers_1_linear2_weight_to_fp16"), val = tensor<fp16, [768, 3072]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(637859584)))];
tensor<fp16, [768]> model_head_layers_1_linear2_bias_to_fp16 = const()[name = tensor<string, []>("model_head_layers_1_linear2_bias_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(642578240)))];
tensor<fp16, [1, 256, 768]> linear_95_cast_fp16 = linear(bias = model_head_layers_1_linear2_bias_to_fp16, weight = model_head_layers_1_linear2_weight_to_fp16, x = input_421_cast_fp16)[name = tensor<string, []>("linear_95_cast_fp16")];
tensor<fp16, [1, 256, 768]> x_cast_fp16 = add(x = input_417_cast_fp16, y = linear_95_cast_fp16)[name = tensor<string, []>("x_cast_fp16")];
tensor<bool, []> input_423_transpose_x_0 = const()[name = tensor<string, []>("input_423_transpose_x_0"), val = tensor<bool, []>(false)];
tensor<bool, []> input_423_transpose_y_0 = const()[name = tensor<string, []>("input_423_transpose_y_0"), val = tensor<bool, []>(false)];
tensor<string, []> marker_map_to_fp16_dtype_0 = const()[name = tensor<string, []>("marker_map_to_fp16_dtype_0"), val = tensor<string, []>("fp16")];
tensor<fp16, [1, 32, 256]> marker_map_to_fp16 = cast(dtype = marker_map_to_fp16_dtype_0, x = marker_map)[name = tensor<string, []>("cast_61")];
tensor<fp16, [1, 32, 768]> input_423_cast_fp16 = matmul(transpose_x = input_423_transpose_x_0, transpose_y = input_423_transpose_y_0, x = marker_map_to_fp16, y = x_cast_fp16)[name = tensor<string, []>("input_423_cast_fp16")];
tensor<int32, [1]> input_425_axes_0 = const()[name = tensor<string, []>("input_425_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [768]> model_scorer_0_weight_to_fp16 = const()[name = tensor<string, []>("model_scorer_0_weight_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(642579840)))];
tensor<fp16, [768]> model_scorer_0_bias_to_fp16 = const()[name = tensor<string, []>("model_scorer_0_bias_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(642581440)))];
tensor<fp16, []> var_3432_to_fp16 = const()[name = tensor<string, []>("op_3432_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 32, 768]> input_425_cast_fp16 = layer_norm(axes = input_425_axes_0, beta = model_scorer_0_bias_to_fp16, epsilon = var_3432_to_fp16, gamma = model_scorer_0_weight_to_fp16, x = input_423_cast_fp16)[name = tensor<string, []>("input_425_cast_fp16")];
tensor<fp16, [768, 768]> model_scorer_1_weight_to_fp16 = const()[name = tensor<string, []>("model_scorer_1_weight_to_fp16"), val = tensor<fp16, [768, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(642583040)))];
tensor<fp16, [768]> model_scorer_1_bias_to_fp16 = const()[name = tensor<string, []>("model_scorer_1_bias_to_fp16"), val = tensor<fp16, [768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(643762752)))];
tensor<fp16, [1, 32, 768]> linear_96_cast_fp16 = linear(bias = model_scorer_1_bias_to_fp16, weight = model_scorer_1_weight_to_fp16, x = input_425_cast_fp16)[name = tensor<string, []>("linear_96_cast_fp16")];
tensor<string, []> input_429_mode_0 = const()[name = tensor<string, []>("input_429_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 32, 768]> input_429_cast_fp16 = gelu(mode = input_429_mode_0, x = linear_96_cast_fp16)[name = tensor<string, []>("input_429_cast_fp16")];
tensor<fp16, [1, 768]> model_scorer_3_weight_to_fp16 = const()[name = tensor<string, []>("model_scorer_3_weight_to_fp16"), val = tensor<fp16, [1, 768]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(643764352)))];
tensor<fp16, [1]> model_scorer_3_bias_to_fp16 = const()[name = tensor<string, []>("model_scorer_3_bias_to_fp16"), val = tensor<fp16, [1]>([0x1.438p-6])];
tensor<fp16, [1, 32, 1]> linear_97_cast_fp16 = linear(bias = model_scorer_3_bias_to_fp16, weight = model_scorer_3_weight_to_fp16, x = input_429_cast_fp16)[name = tensor<string, []>("linear_97_cast_fp16")];
tensor<int32, [1]> logits_1_axes_0 = const()[name = tensor<string, []>("logits_1_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<fp16, [1, 32]> logits_1_cast_fp16 = squeeze(axes = logits_1_axes_0, x = linear_97_cast_fp16)[name = tensor<string, []>("logits_1_cast_fp16")];
tensor<int32, [1]> option_mask_axes_0 = const()[name = tensor<string, []>("option_mask_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<bool, []> option_mask_keep_dims_0 = const()[name = tensor<string, []>("option_mask_keep_dims_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 32]> option_mask_cast_fp16 = reduce_sum(axes = option_mask_axes_0, keep_dims = option_mask_keep_dims_0, x = marker_map_to_fp16)[name = tensor<string, []>("option_mask_cast_fp16")];
tensor<fp16, [1, 32]> var_3446_cast_fp16 = mul(x = logits_1_cast_fp16, y = option_mask_cast_fp16)[name = tensor<string, []>("op_3446_cast_fp16")];
tensor<fp16, []> var_3447_to_fp16 = const()[name = tensor<string, []>("op_3447_to_fp16"), val = tensor<fp16, []>(0x1p+0)];
tensor<fp16, [1, 32]> var_3449_cast_fp16 = sub(x = var_3447_to_fp16, y = option_mask_cast_fp16)[name = tensor<string, []>("op_3449_cast_fp16")];
tensor<fp16, []> var_3450_to_fp16 = const()[name = tensor<string, []>("op_3450_to_fp16"), val = tensor<fp16, []>(-0x1.388p+13)];
tensor<fp16, [1, 32]> var_3451_cast_fp16 = mul(x = var_3449_cast_fp16, y = var_3450_to_fp16)[name = tensor<string, []>("op_3451_cast_fp16")];
tensor<fp16, [1, 32]> logits_cast_fp16 = add(x = var_3446_cast_fp16, y = var_3451_cast_fp16)[name = tensor<string, []>("logits_cast_fp16")];
tensor<string, []> logits_cast_fp16_to_fp32_dtype_0 = const()[name = tensor<string, []>("logits_cast_fp16_to_fp32_dtype_0"), val = tensor<string, []>("fp32")];
tensor<int32, []> var_3454 = const()[name = tensor<string, []>("op_3454"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 32]> probabilities_cast_fp16 = softmax(axis = var_3454, x = logits_cast_fp16)[name = tensor<string, []>("probabilities_cast_fp16")];
tensor<string, []> probabilities_cast_fp16_to_fp32_dtype_0 = const()[name = tensor<string, []>("probabilities_cast_fp16_to_fp32_dtype_0"), val = tensor<string, []>("fp32")];
tensor<int32, [1]> var_3461_axes_0 = const()[name = tensor<string, []>("op_3461_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<bool, []> var_3461_keep_dims_0 = const()[name = tensor<string, []>("op_3461_keep_dims_0"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 1]> var_3461_cast_fp16 = reduce_sum(axes = var_3461_axes_0, keep_dims = var_3461_keep_dims_0, x = option_mask_cast_fp16)[name = tensor<string, []>("op_3461_cast_fp16")];
tensor<fp16, []> var_3462_to_fp16 = const()[name = tensor<string, []>("op_3462_to_fp16"), val = tensor<fp16, []>(0x1p+1)];
tensor<fp16, []> const_182_to_fp16 = const()[name = tensor<string, []>("const_182_to_fp16"), val = tensor<fp16, []>(inf)];
tensor<fp16, [1, 1]> clip_0_cast_fp16 = clip(alpha = var_3462_to_fp16, beta = const_182_to_fp16, x = var_3461_cast_fp16)[name = tensor<string, []>("clip_0_cast_fp16")];
tensor<fp16, []> var_3465_to_fp16 = const()[name = tensor<string, []>("op_3465_to_fp16"), val = tensor<fp16, []>(0x1p-24)];
tensor<fp16, []> const_183_to_fp16 = const()[name = tensor<string, []>("const_183_to_fp16"), val = tensor<fp16, []>(inf)];
tensor<fp16, [1, 32]> clip_1_cast_fp16 = clip(alpha = var_3465_to_fp16, beta = const_183_to_fp16, x = probabilities_cast_fp16)[name = tensor<string, []>("clip_1_cast_fp16")];
tensor<fp32, []> var_3468_epsilon_0 = const()[name = tensor<string, []>("op_3468_epsilon_0"), val = tensor<fp32, []>(0x1p-149)];
tensor<fp16, [1, 32]> var_3468_cast_fp16 = log(epsilon = var_3468_epsilon_0, x = clip_1_cast_fp16)[name = tensor<string, []>("op_3468_cast_fp16")];
tensor<fp16, [1, 32]> var_3469_cast_fp16 = mul(x = probabilities_cast_fp16, y = var_3468_cast_fp16)[name = tensor<string, []>("op_3469_cast_fp16")];
tensor<int32, [1]> var_3474_axes_0 = const()[name = tensor<string, []>("op_3474_axes_0"), val = tensor<int32, [1]>([-1])];
tensor<bool, []> var_3474_keep_dims_0 = const()[name = tensor<string, []>("op_3474_keep_dims_0"), val = tensor<bool, []>(true)];
tensor<fp16, [1, 1]> var_3474_cast_fp16 = reduce_sum(axes = var_3474_axes_0, keep_dims = var_3474_keep_dims_0, x = var_3469_cast_fp16)[name = tensor<string, []>("op_3474_cast_fp16")];
tensor<fp16, []> const_184_promoted_to_fp16 = const()[name = tensor<string, []>("const_184_promoted_to_fp16"), val = tensor<fp16, []>(-0x1p+0)];
tensor<fp16, [1, 1]> var_3475_cast_fp16 = mul(x = var_3474_cast_fp16, y = const_184_promoted_to_fp16)[name = tensor<string, []>("op_3475_cast_fp16")];
tensor<fp32, []> var_3476_epsilon_0 = const()[name = tensor<string, []>("op_3476_epsilon_0"), val = tensor<fp32, []>(0x1p-149)];
tensor<fp16, [1, 1]> var_3476_cast_fp16 = log(epsilon = var_3476_epsilon_0, x = clip_0_cast_fp16)[name = tensor<string, []>("op_3476_cast_fp16")];
tensor<fp16, [1, 1]> entropy_cast_fp16 = real_div(x = var_3475_cast_fp16, y = var_3476_cast_fp16)[name = tensor<string, []>("entropy_cast_fp16")];
tensor<int32, []> var_3478 = const()[name = tensor<string, []>("op_3478"), val = tensor<int32, []>(2)];
tensor<int32, []> top2_axis_0 = const()[name = tensor<string, []>("top2_axis_0"), val = tensor<int32, []>(-1)];
tensor<bool, []> top2_ascending_0 = const()[name = tensor<string, []>("top2_ascending_0"), val = tensor<bool, []>(false)];
tensor<bool, []> top2_sort_0 = const()[name = tensor<string, []>("top2_sort_0"), val = tensor<bool, []>(true)];
tensor<bool, []> top2_return_indices_0 = const()[name = tensor<string, []>("top2_return_indices_0"), val = tensor<bool, []>(true)];
tensor<string, []> top2_cast_fp16_cast_int16_output_indices_dtype_0 = const()[name = tensor<string, []>("top2_cast_fp16_cast_int16_output_indices_dtype_0"), val = tensor<string, []>("uint16")];
tensor<fp16, [1, 2]> top2_cast_fp16_cast_int16_0, tensor<uint16, [1, 2]> top2_cast_fp16_cast_int16_1 = topk(ascending = top2_ascending_0, axis = top2_axis_0, k = var_3478, output_indices_dtype = top2_cast_fp16_cast_int16_output_indices_dtype_0, return_indices = top2_return_indices_0, sort = top2_sort_0, x = probabilities_cast_fp16)[name = tensor<string, []>("top2_cast_fp16_cast_int16")];
tensor<int32, [2]> var_3493_begin_0 = const()[name = tensor<string, []>("op_3493_begin_0"), val = tensor<int32, [2]>([0, 0])];
tensor<int32, [2]> var_3493_end_0 = const()[name = tensor<string, []>("op_3493_end_0"), val = tensor<int32, [2]>([1, 1])];
tensor<bool, [2]> var_3493_end_mask_0 = const()[name = tensor<string, []>("op_3493_end_mask_0"), val = tensor<bool, [2]>([true, false])];
tensor<fp16, [1, 1]> var_3493_cast_fp16 = slice_by_index(begin = var_3493_begin_0, end = var_3493_end_0, end_mask = var_3493_end_mask_0, x = top2_cast_fp16_cast_int16_0)[name = tensor<string, []>("op_3493_cast_fp16")];
tensor<int32, [2]> var_3513_begin_0 = const()[name = tensor<string, []>("op_3513_begin_0"), val = tensor<int32, [2]>([0, 1])];
tensor<int32, [2]> var_3513_end_0 = const()[name = tensor<string, []>("op_3513_end_0"), val = tensor<int32, [2]>([1, 1])];
tensor<bool, [2]> var_3513_end_mask_0 = const()[name = tensor<string, []>("op_3513_end_mask_0"), val = tensor<bool, [2]>([true, true])];
tensor<fp16, [1, 1]> var_3513_cast_fp16 = slice_by_index(begin = var_3513_begin_0, end = var_3513_end_0, end_mask = var_3513_end_mask_0, x = top2_cast_fp16_cast_int16_0)[name = tensor<string, []>("op_3513_cast_fp16")];
tensor<fp16, [1, 1]> var_3515_cast_fp16 = sub(x = var_3493_cast_fp16, y = var_3513_cast_fp16)[name = tensor<string, []>("op_3515_cast_fp16")];
tensor<fp16, []> _inversed_3517_y_0_to_fp16 = const()[name = tensor<string, []>("_inversed_3517_y_0_to_fp16"), val = tensor<fp16, []>(0x1.01p-8)];
tensor<fp16, [1, 1]> _inversed_3517_cast_fp16 = mul(x = clip_0_cast_fp16, y = _inversed_3517_y_0_to_fp16)[name = tensor<string, []>("_inversed_3517_cast_fp16")];
tensor<int32, []> var_3519 = const()[name = tensor<string, []>("op_3519"), val = tensor<int32, []>(-1)];
tensor<bool, []> feats_interleave_0 = const()[name = tensor<string, []>("feats_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 4]> feats_cast_fp16 = concat(axis = var_3519, interleave = feats_interleave_0, values = (var_3493_cast_fp16, var_3515_cast_fp16, entropy_cast_fp16, _inversed_3517_cast_fp16))[name = tensor<string, []>("feats_cast_fp16")];
tensor<int32, [3]> pooled_begin_0 = const()[name = tensor<string, []>("pooled_begin_0"), val = tensor<int32, [3]>([0, 0, 0])];
tensor<int32, [3]> pooled_end_0 = const()[name = tensor<string, []>("pooled_end_0"), val = tensor<int32, [3]>([1, 1, 768])];
tensor<bool, [3]> pooled_end_mask_0 = const()[name = tensor<string, []>("pooled_end_mask_0"), val = tensor<bool, [3]>([true, false, true])];
tensor<bool, [3]> pooled_squeeze_mask_0 = const()[name = tensor<string, []>("pooled_squeeze_mask_0"), val = tensor<bool, [3]>([false, true, false])];
tensor<fp16, [1, 768]> pooled_cast_fp16 = slice_by_index(begin = pooled_begin_0, end = pooled_end_0, end_mask = pooled_end_mask_0, squeeze_mask = pooled_squeeze_mask_0, x = x_cast_fp16)[name = tensor<string, []>("pooled_cast_fp16")];
tensor<int32, []> var_3530 = const()[name = tensor<string, []>("op_3530"), val = tensor<int32, []>(-1)];
tensor<bool, []> input_431_interleave_0 = const()[name = tensor<string, []>("input_431_interleave_0"), val = tensor<bool, []>(false)];
tensor<fp16, [1, 772]> input_431_cast_fp16 = concat(axis = var_3530, interleave = input_431_interleave_0, values = (pooled_cast_fp16, feats_cast_fp16))[name = tensor<string, []>("input_431_cast_fp16")];
tensor<fp16, [256, 772]> model_act_head_0_weight_to_fp16 = const()[name = tensor<string, []>("model_act_head_0_weight_to_fp16"), val = tensor<fp16, [256, 772]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(643765952)))];
tensor<fp16, [256]> model_act_head_0_bias_to_fp16 = const()[name = tensor<string, []>("model_act_head_0_bias_to_fp16"), val = tensor<fp16, [256]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(644161280)))];
tensor<fp16, [1, 256]> linear_98_cast_fp16 = linear(bias = model_act_head_0_bias_to_fp16, weight = model_act_head_0_weight_to_fp16, x = input_431_cast_fp16)[name = tensor<string, []>("linear_98_cast_fp16")];
tensor<string, []> input_mode_0 = const()[name = tensor<string, []>("input_mode_0"), val = tensor<string, []>("EXACT")];
tensor<fp16, [1, 256]> input_cast_fp16 = gelu(mode = input_mode_0, x = linear_98_cast_fp16)[name = tensor<string, []>("input_cast_fp16")];
tensor<fp16, [2, 256]> model_act_head_2_weight_to_fp16 = const()[name = tensor<string, []>("model_act_head_2_weight_to_fp16"), val = tensor<fp16, [2, 256]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(644161856)))];
tensor<fp16, [2]> model_act_head_2_bias_to_fp16 = const()[name = tensor<string, []>("model_act_head_2_bias_to_fp16"), val = tensor<fp16, [2]>([0x1.818p-6, -0x1.4fcp-6])];
tensor<fp16, [1, 2]> linear_99_cast_fp16 = linear(bias = model_act_head_2_bias_to_fp16, weight = model_act_head_2_weight_to_fp16, x = input_cast_fp16)[name = tensor<string, []>("linear_99_cast_fp16")];
tensor<int32, []> var_3536 = const()[name = tensor<string, []>("op_3536"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 2]> var_3538_cast_fp16 = softmax(axis = var_3536, x = linear_99_cast_fp16)[name = tensor<string, []>("op_3538_cast_fp16")];
tensor<string, []> var_3538_cast_fp16_to_fp32_dtype_0 = const()[name = tensor<string, []>("op_3538_cast_fp16_to_fp32_dtype_0"), val = tensor<string, []>("fp32")];
tensor<fp32, [1, 2]> action_probabilities = cast(dtype = var_3538_cast_fp16_to_fp32_dtype_0, x = var_3538_cast_fp16)[name = tensor<string, []>("cast_58")];
tensor<fp32, [1, 32]> probabilities = cast(dtype = probabilities_cast_fp16_to_fp32_dtype_0, x = probabilities_cast_fp16)[name = tensor<string, []>("cast_59")];
tensor<fp32, [1, 32]> logits = cast(dtype = logits_cast_fp16_to_fp32_dtype_0, x = logits_cast_fp16)[name = tensor<string, []>("cast_60")];
} -> (logits, probabilities, action_probabilities);
}