diff --git a/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/analytics/coremldata.bin b/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/analytics/coremldata.bin new file mode 100644 index 0000000000000000000000000000000000000000..5fb537fbab27a8f122f8cfd8fb69cd6fab6fb6d9 --- /dev/null +++ b/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/analytics/coremldata.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4b1f35bd2ae7d47f54a1d7ea1bc0365d0811cf1638d907d18fb12e4a39d5c76b +size 243 diff --git a/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/coremldata.bin b/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/coremldata.bin new file mode 100644 index 0000000000000000000000000000000000000000..3017bc7894fe9124407c78ea8c52139d98871768 --- /dev/null +++ b/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/coremldata.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:20a306ad1c06654b84a69e7b3653a75c77cb591dcc673715e1965dd4e09e22c4 +size 624 diff --git a/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/metadata.json b/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..31ca694b0b0633e117c4f863671937caecd92dee --- /dev/null +++ b/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/metadata.json @@ -0,0 +1,386 @@ +[ + { + "metadataOutputVersion" : "3.0", + "storagePrecision" : "Mixed (Float16, Palettized (8 bits), UInt8)", + "outputSchema" : [ + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 15 × 2048)", + "shortDescription" : "", + "shape" : "[1, 15, 2048]", + "name" : "all_logits", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 1]", + "name" : "key_cache_updates", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 1]", + "name" : "value_cache_updates", + "type" : "MultiArray" + } + ], + "modelParameters" : [ + + ], + "specificationVersion" : 9, + "functions" : [ + { + "inputSchema" : [ + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 1024 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 1024, 1, 1]", + "name" : "input_embeds", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Int32", + "formattedType" : "MultiArray (Int32 1)", + "shortDescription" : "", + "shape" : "[1]", + "name" : "cache_length", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 16]", + "name" : "key_cache", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 16]", + "name" : "value_cache", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 16]", + "name" : "kv_cache_update_mask", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 16]", + "name" : "key_padding_mask", + "type" : "MultiArray" + } + ], + "computePrecision" : "Mixed (Float16, Float32, Int16, Int32, UInt16)", + "storagePrecision" : "Mixed (Float16, Palettized (8 bits), UInt8)", + "stateSchema" : [ + + ], + "outputSchema" : [ + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 15 × 2048)", + "shortDescription" : "", + "shape" : "[1, 15, 2048]", + "name" : "all_logits", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 1]", + "name" : "key_cache_updates", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 1]", + "name" : "value_cache_updates", + "type" : "MultiArray" + } + ], + "name" : "stepped", + "mlProgramOperationTypeHistogram" : { + "Ios18.expandDims" : 8, + "Ios18.softmax" : 5, + "Ios18.mul" : 123, + "Ios18.matmul" : 10, + "Ios18.rsqrt" : 21, + "Ios16.reduceMean" : 21, + "Split" : 2, + "Ios18.greaterEqual" : 2, + "Select" : 2, + "Ios18.gather" : 2, + "Ios18.add" : 58, + "Ios18.reshape" : 40, + "Ios18.constexprLutToDense" : 50, + "Ios18.conv" : 50, + "Ios18.concat" : 23, + "Ios18.cast" : 5, + "Ios18.sub" : 1, + "Ios18.silu" : 5, + "Ios18.transpose" : 1, + "Ios18.sliceByIndex" : 100, + "Ios18.squeeze" : 15 + } + }, + { + "inputSchema" : [ + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 1024 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 1024, 1, 1]", + "name" : "code0_hidden_states", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 1024 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 1024, 1, 1]", + "name" : "code0_embed", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 15 × 2048)", + "shortDescription" : "", + "shape" : "[15, 2048]", + "name" : "gumbel", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1)", + "shortDescription" : "", + "shape" : "[1]", + "name" : "temperature", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Int32", + "formattedType" : "MultiArray (Int32 1)", + "shortDescription" : "", + "shape" : "[1]", + "name" : "top_k", + "type" : "MultiArray" + } + ], + "computePrecision" : "Mixed (Float16, Float32, Int32, UInt16)", + "storagePrecision" : "Mixed (Float16, Palettized (8 bits), UInt8)", + "stateSchema" : [ + + ], + "outputSchema" : [ + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Int32", + "formattedType" : "MultiArray (Int32 1 × 15)", + "shortDescription" : "", + "shape" : "[1, 15]", + "name" : "codes", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 1024 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 1024, 1, 1]", + "name" : "embed_sum", + "type" : "MultiArray" + } + ], + "name" : "fused", + "mlProgramOperationTypeHistogram" : { + "Ios18.softmax" : 79, + "Ios18.mul" : 1951, + "Ios18.matmul" : 158, + "Ios18.rsqrt" : 333, + "Ios16.reduceMean" : 333, + "Ios18.realDiv" : 15, + "Ios18.greaterEqual" : 15, + "Select" : 15, + "Ios16.reduceMin" : 15, + "Ios18.add" : 932, + "Tile" : 158, + "Ios18.reduceArgmax" : 15, + "Ios16.fillLike" : 15, + "Ios18.reshape" : 996, + "Ios18.gather" : 15, + "Ios18.constexprLutToDense" : 51, + "Ios18.conv" : 570, + "Ios18.concat" : 189, + "Ios18.topk" : 15, + "Ios18.sub" : 1, + "Ios18.transpose" : 474, + "Ios18.silu" : 79, + "Ios18.cast" : 17, + "Ios18.less" : 1, + "Stack" : 1, + "Ios18.sliceByIndex" : 483 + } + } + ], + "mlProgramOperationTypeHistogram" : { + "Ios18.expandDims" : 8, + "Ios18.softmax" : 5, + "Ios18.mul" : 123, + "Ios18.matmul" : 10, + "Ios18.rsqrt" : 21, + "Ios16.reduceMean" : 21, + "Split" : 2, + "Ios18.greaterEqual" : 2, + "Select" : 2, + "Ios18.gather" : 2, + "Ios18.add" : 58, + "Ios18.reshape" : 40, + "Ios18.constexprLutToDense" : 50, + "Ios18.conv" : 50, + "Ios18.concat" : 23, + "Ios18.cast" : 5, + "Ios18.sub" : 1, + "Ios18.silu" : 5, + "Ios18.transpose" : 1, + "Ios18.sliceByIndex" : 100, + "Ios18.squeeze" : 15 + }, + "isUpdatable" : "0", + "stateSchema" : [ + + ], + "availability" : { + "macOS" : "15.0", + "tvOS" : "18.0", + "visionOS" : "2.0", + "watchOS" : "11.0", + "iOS" : "18.0", + "macCatalyst" : "18.0" + }, + "computePrecision" : "Mixed (Float16, Float32, Int16, Int32, UInt16)", + "modelType" : { + "name" : "MLModelType_mlProgram" + }, + "inputSchema" : [ + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 1024 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 1024, 1, 1]", + "name" : "input_embeds", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Int32", + "formattedType" : "MultiArray (Int32 1)", + "shortDescription" : "", + "shape" : "[1]", + "name" : "cache_length", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 16]", + "name" : "key_cache", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 16]", + "name" : "value_cache", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 16]", + "name" : "kv_cache_update_mask", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 16]", + "name" : "key_padding_mask", + "type" : "MultiArray" + } + ], + "defaultFunctionName" : "stepped", + "generatedClassName" : "MultiCodeDecoder", + "userDefinedMetadata" : { + + }, + "method" : "predict" + } +] \ No newline at end of file diff --git a/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/model.mil b/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/model.mil new file mode 100644 index 0000000000000000000000000000000000000000..dd737013b69c6b6bc066e29670ef2d9b06f32d95 --- /dev/null +++ b/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/model.mil @@ -0,0 +1,17119 @@ +program(1.3) +[buildInfo = dict({{"coremlc-component-MIL", "3520.4.1"}, {"coremlc-version", "3520.5.1"}})] +{ + func fused(tensor code0_embed, tensor code0_hidden_states, tensor gumbel, tensor temperature, tensor top_k) { + int32 var_168 = const()[name = string("op_168"), val = int32(2)]; + int32 var_172 = const()[name = string("op_172"), val = int32(3)]; + tensor var_187_cast_fp16 = mul(x = code0_hidden_states, y = code0_hidden_states)[name = string("op_187_cast_fp16")]; + tensor variance_1_axes_0 = const()[name = string("variance_1_axes_0"), val = tensor([1])]; + bool variance_1_keep_dims_0 = const()[name = string("variance_1_keep_dims_0"), val = bool(true)]; + tensor variance_1_cast_fp16 = reduce_mean(axes = variance_1_axes_0, keep_dims = variance_1_keep_dims_0, x = var_187_cast_fp16)[name = string("variance_1_cast_fp16")]; + fp16 var_190_to_fp16 = const()[name = string("op_190_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_191_cast_fp16 = add(x = variance_1_cast_fp16, y = var_190_to_fp16)[name = string("op_191_cast_fp16")]; + fp32 var_192_epsilon_0 = const()[name = string("op_192_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_192_cast_fp16 = rsqrt(epsilon = var_192_epsilon_0, x = var_191_cast_fp16)[name = string("op_192_cast_fp16")]; + tensor var_193_cast_fp16 = mul(x = code0_hidden_states, y = var_192_cast_fp16)[name = string("op_193_cast_fp16")]; + tensor layers_0_input_layernorm_weight_to_fp16 = const()[name = string("layers_0_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8384)))]; + tensor input_1_cast_fp16 = mul(x = var_193_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_1_cast_fp16")]; + string q_1_pad_type_0 = const()[name = string("q_1_pad_type_0"), val = string("valid")]; + tensor q_1_strides_0 = const()[name = string("q_1_strides_0"), val = tensor([1, 1])]; + tensor q_1_pad_0 = const()[name = string("q_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_1_dilations_0 = const()[name = string("q_1_dilations_0"), val = tensor([1, 1])]; + int32 q_1_groups_0 = const()[name = string("q_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2107712))))[name = string("layers_0_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor q_1_cast_fp16 = conv(dilations = q_1_dilations_0, groups = q_1_groups_0, pad = q_1_pad_0, pad_type = q_1_pad_type_0, strides = q_1_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_1_cast_fp16)[name = string("q_1_cast_fp16")]; + string k_1_pad_type_0 = const()[name = string("k_1_pad_type_0"), val = string("valid")]; + tensor k_1_strides_0 = const()[name = string("k_1_strides_0"), val = tensor([1, 1])]; + tensor k_1_pad_0 = const()[name = string("k_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_1_dilations_0 = const()[name = string("k_1_dilations_0"), val = tensor([1, 1])]; + int32 k_1_groups_0 = const()[name = string("k_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2112448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3161088))))[name = string("layers_0_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor k_1_cast_fp16 = conv(dilations = k_1_dilations_0, groups = k_1_groups_0, pad = k_1_pad_0, pad_type = k_1_pad_type_0, strides = k_1_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_1_cast_fp16)[name = string("k_1_cast_fp16")]; + string v_1_pad_type_0 = const()[name = string("v_1_pad_type_0"), val = string("valid")]; + tensor v_1_strides_0 = const()[name = string("v_1_strides_0"), val = tensor([1, 1])]; + tensor v_1_pad_0 = const()[name = string("v_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_1_dilations_0 = const()[name = string("v_1_dilations_0"), val = tensor([1, 1])]; + int32 v_1_groups_0 = const()[name = string("v_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3161664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4210304))))[name = string("layers_0_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor v_1_cast_fp16 = conv(dilations = v_1_dilations_0, groups = v_1_groups_0, pad = v_1_pad_0, pad_type = v_1_pad_type_0, strides = v_1_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_1_cast_fp16)[name = string("v_1_cast_fp16")]; + tensor var_227 = const()[name = string("op_227"), val = tensor([16, 128, 1, 1])]; + tensor x_1_cast_fp16 = reshape(shape = var_227, x = q_1_cast_fp16)[name = string("x_1_cast_fp16")]; + tensor var_230_cast_fp16 = mul(x = x_1_cast_fp16, y = x_1_cast_fp16)[name = string("op_230_cast_fp16")]; + tensor variance_3_axes_0 = const()[name = string("variance_3_axes_0"), val = tensor([1])]; + bool variance_3_keep_dims_0 = const()[name = string("variance_3_keep_dims_0"), val = bool(true)]; + tensor variance_3_cast_fp16 = reduce_mean(axes = variance_3_axes_0, keep_dims = variance_3_keep_dims_0, x = var_230_cast_fp16)[name = string("variance_3_cast_fp16")]; + fp16 var_233_to_fp16 = const()[name = string("op_233_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_234_cast_fp16 = add(x = variance_3_cast_fp16, y = var_233_to_fp16)[name = string("op_234_cast_fp16")]; + fp32 var_235_epsilon_0 = const()[name = string("op_235_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_235_cast_fp16 = rsqrt(epsilon = var_235_epsilon_0, x = var_234_cast_fp16)[name = string("op_235_cast_fp16")]; + tensor var_236_cast_fp16 = mul(x = x_1_cast_fp16, y = var_235_cast_fp16)[name = string("op_236_cast_fp16")]; + tensor layers_0_self_attn_q_norm_weight_to_fp16 = const()[name = string("layers_0_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4212992)))]; + tensor q_3_cast_fp16 = mul(x = var_236_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_3_cast_fp16")]; + tensor var_238 = const()[name = string("op_238"), val = tensor([8, 128, 1, 1])]; + tensor x_3_cast_fp16 = reshape(shape = var_238, x = k_1_cast_fp16)[name = string("x_3_cast_fp16")]; + tensor var_241_cast_fp16 = mul(x = x_3_cast_fp16, y = x_3_cast_fp16)[name = string("op_241_cast_fp16")]; + tensor variance_5_axes_0 = const()[name = string("variance_5_axes_0"), val = tensor([1])]; + bool variance_5_keep_dims_0 = const()[name = string("variance_5_keep_dims_0"), val = bool(true)]; + tensor variance_5_cast_fp16 = reduce_mean(axes = variance_5_axes_0, keep_dims = variance_5_keep_dims_0, x = var_241_cast_fp16)[name = string("variance_5_cast_fp16")]; + fp16 var_244_to_fp16 = const()[name = string("op_244_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_245_cast_fp16 = add(x = variance_5_cast_fp16, y = var_244_to_fp16)[name = string("op_245_cast_fp16")]; + fp32 var_246_epsilon_0 = const()[name = string("op_246_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_246_cast_fp16 = rsqrt(epsilon = var_246_epsilon_0, x = var_245_cast_fp16)[name = string("op_246_cast_fp16")]; + tensor var_247_cast_fp16 = mul(x = x_3_cast_fp16, y = var_246_cast_fp16)[name = string("op_247_cast_fp16")]; + tensor layers_0_self_attn_k_norm_weight_to_fp16 = const()[name = string("layers_0_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4213312)))]; + tensor k_3_cast_fp16 = mul(x = var_247_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_3_cast_fp16")]; + tensor var_249 = const()[name = string("op_249"), val = tensor([1, 16, 128, 1])]; + tensor z_1_cast_fp16 = reshape(shape = var_249, x = q_3_cast_fp16)[name = string("z_1_cast_fp16")]; + tensor var_251 = const()[name = string("op_251"), val = tensor([1, 8, 128, 1])]; + tensor z_3_cast_fp16 = reshape(shape = var_251, x = k_3_cast_fp16)[name = string("z_3_cast_fp16")]; + tensor z1_1_begin_0 = const()[name = string("z1_1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_1_end_0 = const()[name = string("z1_1_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_1_end_mask_0 = const()[name = string("z1_1_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_1_cast_fp16 = slice_by_index(begin = z1_1_begin_0, end = z1_1_end_0, end_mask = z1_1_end_mask_0, x = z_1_cast_fp16)[name = string("z1_1_cast_fp16")]; + tensor z2_1_begin_0 = const()[name = string("z2_1_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_1_end_0 = const()[name = string("z2_1_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_1_end_mask_0 = const()[name = string("z2_1_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_1_cast_fp16 = slice_by_index(begin = z2_1_begin_0, end = z2_1_end_0, end_mask = z2_1_end_mask_0, x = z_1_cast_fp16)[name = string("z2_1_cast_fp16")]; + fp16 const_1_promoted_to_fp16 = const()[name = string("const_1_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_260_cast_fp16 = mul(x = z2_1_cast_fp16, y = const_1_promoted_to_fp16)[name = string("op_260_cast_fp16")]; + bool var_262_interleave_0 = const()[name = string("op_262_interleave_0"), val = bool(false)]; + tensor var_262_cast_fp16 = concat(axis = var_168, interleave = var_262_interleave_0, values = (var_260_cast_fp16, z1_1_cast_fp16))[name = string("op_262_cast_fp16")]; + tensor sin_1_to_fp16 = const()[name = string("sin_1_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110173568)))]; + tensor var_263_cast_fp16 = mul(x = var_262_cast_fp16, y = sin_1_to_fp16)[name = string("op_263_cast_fp16")]; + tensor q_5_cast_fp16 = add(x = z_1_cast_fp16, y = var_263_cast_fp16)[name = string("q_5_cast_fp16")]; + tensor z1_3_begin_0 = const()[name = string("z1_3_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_3_end_0 = const()[name = string("z1_3_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_3_end_mask_0 = const()[name = string("z1_3_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_3_cast_fp16 = slice_by_index(begin = z1_3_begin_0, end = z1_3_end_0, end_mask = z1_3_end_mask_0, x = z_3_cast_fp16)[name = string("z1_3_cast_fp16")]; + tensor z2_3_begin_0 = const()[name = string("z2_3_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_3_end_0 = const()[name = string("z2_3_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_3_end_mask_0 = const()[name = string("z2_3_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_3_cast_fp16 = slice_by_index(begin = z2_3_begin_0, end = z2_3_end_0, end_mask = z2_3_end_mask_0, x = z_3_cast_fp16)[name = string("z2_3_cast_fp16")]; + fp16 const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_272_cast_fp16 = mul(x = z2_3_cast_fp16, y = const_2_promoted_to_fp16)[name = string("op_272_cast_fp16")]; + bool var_274_interleave_0 = const()[name = string("op_274_interleave_0"), val = bool(false)]; + tensor var_274_cast_fp16 = concat(axis = var_168, interleave = var_274_interleave_0, values = (var_272_cast_fp16, z1_3_cast_fp16))[name = string("op_274_cast_fp16")]; + tensor var_275_cast_fp16 = mul(x = var_274_cast_fp16, y = sin_1_to_fp16)[name = string("op_275_cast_fp16")]; + tensor k_5_cast_fp16 = add(x = z_3_cast_fp16, y = var_275_cast_fp16)[name = string("k_5_cast_fp16")]; + tensor var_277 = const()[name = string("op_277"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_1_cast_fp16 = reshape(shape = var_277, x = k_5_cast_fp16)[name = string("cur_key_1_cast_fp16")]; + tensor upd_1_to_fp16 = const()[name = string("upd_1_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110173888)))]; + tensor var_281_cast_fp16 = mul(x = cur_key_1_cast_fp16, y = upd_1_to_fp16)[name = string("op_281_cast_fp16")]; + tensor var_285_cast_fp16 = mul(x = v_1_cast_fp16, y = upd_1_to_fp16)[name = string("op_285_cast_fp16")]; + tensor var_287 = const()[name = string("op_287"), val = tensor([1, 8, 128, 16])]; + tensor kh_1_cast_fp16 = reshape(shape = var_287, x = var_281_cast_fp16)[name = string("kh_1_cast_fp16")]; + tensor var_289 = const()[name = string("op_289"), val = tensor([1, 8, 128, 16])]; + tensor vh_1_cast_fp16 = reshape(shape = var_289, x = var_285_cast_fp16)[name = string("vh_1_cast_fp16")]; + tensor transpose_0_perm_0 = const()[name = string("transpose_0_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_0_reps_0 = const()[name = string("tile_0_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_0_cast_fp16 = transpose(perm = transpose_0_perm_0, x = kh_1_cast_fp16)[name = string("transpose_473")]; + tensor tile_0_cast_fp16 = tile(reps = tile_0_reps_0, x = transpose_0_cast_fp16)[name = string("tile_0_cast_fp16")]; + tensor concat_4 = const()[name = string("concat_4"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_0_cast_fp16 = reshape(shape = concat_4, x = tile_0_cast_fp16)[name = string("reshape_0_cast_fp16")]; + tensor transpose_1_perm_0 = const()[name = string("transpose_1_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_5 = const()[name = string("concat_5"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_1_cast_fp16 = transpose(perm = transpose_1_perm_0, x = reshape_0_cast_fp16)[name = string("transpose_472")]; + tensor reshape_1_cast_fp16 = reshape(shape = concat_5, x = transpose_1_cast_fp16)[name = string("reshape_1_cast_fp16")]; + tensor transpose_2_perm_0 = const()[name = string("transpose_2_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_1_reps_0 = const()[name = string("tile_1_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_2_cast_fp16 = transpose(perm = transpose_2_perm_0, x = vh_1_cast_fp16)[name = string("transpose_471")]; + tensor tile_1_cast_fp16 = tile(reps = tile_1_reps_0, x = transpose_2_cast_fp16)[name = string("tile_1_cast_fp16")]; + tensor concat_6 = const()[name = string("concat_6"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_2_cast_fp16 = reshape(shape = concat_6, x = tile_1_cast_fp16)[name = string("reshape_2_cast_fp16")]; + tensor transpose_3_perm_0 = const()[name = string("transpose_3_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_7 = const()[name = string("concat_7"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_3_cast_fp16 = transpose(perm = transpose_3_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_470")]; + tensor reshape_3_cast_fp16 = reshape(shape = concat_7, x = transpose_3_cast_fp16)[name = string("reshape_3_cast_fp16")]; + fp16 var_293_to_fp16 = const()[name = string("op_293_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_294_cast_fp16 = mul(x = q_5_cast_fp16, y = var_293_to_fp16)[name = string("op_294_cast_fp16")]; + tensor transpose_321_perm_0 = const()[name = string("transpose_321_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_1_transpose_x_1 = const()[name = string("w_1_transpose_x_1"), val = bool(true)]; + bool w_1_transpose_y_1 = const()[name = string("w_1_transpose_y_1"), val = bool(false)]; + tensor transpose_321_cast_fp16 = transpose(perm = transpose_321_perm_0, x = reshape_1_cast_fp16)[name = string("transpose_469")]; + tensor w_1_cast_fp16 = matmul(transpose_x = w_1_transpose_x_1, transpose_y = w_1_transpose_y_1, x = var_294_cast_fp16, y = transpose_321_cast_fp16)[name = string("w_1_cast_fp16")]; + tensor pad_1_to_fp16 = const()[name = string("pad_1_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174016)))]; + tensor var_297_cast_fp16 = add(x = w_1_cast_fp16, y = pad_1_to_fp16)[name = string("op_297_cast_fp16")]; + tensor w_3_cast_fp16 = softmax(axis = var_172, x = var_297_cast_fp16)[name = string("w_3_cast_fp16")]; + tensor transpose_322_perm_0 = const()[name = string("transpose_322_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_1_transpose_x_1 = const()[name = string("attn_1_transpose_x_1"), val = bool(false)]; + bool attn_1_transpose_y_1 = const()[name = string("attn_1_transpose_y_1"), val = bool(true)]; + tensor transpose_322_cast_fp16 = transpose(perm = transpose_322_perm_0, x = reshape_3_cast_fp16)[name = string("transpose_468")]; + tensor attn_1_cast_fp16 = matmul(transpose_x = attn_1_transpose_x_1, transpose_y = attn_1_transpose_y_1, x = transpose_322_cast_fp16, y = w_3_cast_fp16)[name = string("attn_1_cast_fp16")]; + tensor var_301 = const()[name = string("op_301"), val = tensor([1, 2048, 1, 1])]; + tensor input_3_cast_fp16 = reshape(shape = var_301, x = attn_1_cast_fp16)[name = string("input_3_cast_fp16")]; + string attn_output_1_pad_type_0 = const()[name = string("attn_output_1_pad_type_0"), val = string("valid")]; + tensor attn_output_1_strides_0 = const()[name = string("attn_output_1_strides_0"), val = tensor([1, 1])]; + tensor attn_output_1_pad_0 = const()[name = string("attn_output_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_1_dilations_0 = const()[name = string("attn_output_1_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_1_groups_0 = const()[name = string("attn_output_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4213632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6310848))))[name = string("layers_0_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor attn_output_1_cast_fp16 = conv(dilations = attn_output_1_dilations_0, groups = attn_output_1_groups_0, pad = attn_output_1_pad_0, pad_type = attn_output_1_pad_type_0, strides = attn_output_1_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_3_cast_fp16)[name = string("attn_output_1_cast_fp16")]; + tensor x_5_cast_fp16 = add(x = code0_hidden_states, y = attn_output_1_cast_fp16)[name = string("x_5_cast_fp16")]; + tensor var_315_cast_fp16 = mul(x = x_5_cast_fp16, y = x_5_cast_fp16)[name = string("op_315_cast_fp16")]; + tensor variance_7_axes_0 = const()[name = string("variance_7_axes_0"), val = tensor([1])]; + bool variance_7_keep_dims_0 = const()[name = string("variance_7_keep_dims_0"), val = bool(true)]; + tensor variance_7_cast_fp16 = reduce_mean(axes = variance_7_axes_0, keep_dims = variance_7_keep_dims_0, x = var_315_cast_fp16)[name = string("variance_7_cast_fp16")]; + fp16 var_318_to_fp16 = const()[name = string("op_318_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_319_cast_fp16 = add(x = variance_7_cast_fp16, y = var_318_to_fp16)[name = string("op_319_cast_fp16")]; + fp32 var_320_epsilon_0 = const()[name = string("op_320_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_320_cast_fp16 = rsqrt(epsilon = var_320_epsilon_0, x = var_319_cast_fp16)[name = string("op_320_cast_fp16")]; + tensor var_321_cast_fp16 = mul(x = x_5_cast_fp16, y = var_320_cast_fp16)[name = string("op_321_cast_fp16")]; + tensor layers_0_post_attention_layernorm_weight_to_fp16 = const()[name = string("layers_0_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6311424)))]; + tensor input_5_cast_fp16 = mul(x = var_321_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_5_cast_fp16")]; + string input_7_pad_type_0 = const()[name = string("input_7_pad_type_0"), val = string("valid")]; + tensor input_7_strides_0 = const()[name = string("input_7_strides_0"), val = tensor([1, 1])]; + tensor input_7_pad_0 = const()[name = string("input_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_7_dilations_0 = const()[name = string("input_7_dilations_0"), val = tensor([1, 1])]; + int32 input_7_groups_0 = const()[name = string("input_7_groups_0"), val = int32(1)]; + tensor layers_0_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6313536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9459328))))[name = string("layers_0_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_7_cast_fp16 = conv(dilations = input_7_dilations_0, groups = input_7_groups_0, pad = input_7_pad_0, pad_type = input_7_pad_type_0, strides = input_7_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_5_cast_fp16)[name = string("input_7_cast_fp16")]; + tensor var_329_cast_fp16 = silu(x = input_7_cast_fp16)[name = string("op_329_cast_fp16")]; + string var_335_pad_type_0 = const()[name = string("op_335_pad_type_0"), val = string("valid")]; + tensor var_335_strides_0 = const()[name = string("op_335_strides_0"), val = tensor([1, 1])]; + tensor var_335_pad_0 = const()[name = string("op_335_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_335_dilations_0 = const()[name = string("op_335_dilations_0"), val = tensor([1, 1])]; + int32 var_335_groups_0 = const()[name = string("op_335_groups_0"), val = int32(1)]; + tensor layers_0_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9459904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12605696))))[name = string("layers_0_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_335_cast_fp16 = conv(dilations = var_335_dilations_0, groups = var_335_groups_0, pad = var_335_pad_0, pad_type = var_335_pad_type_0, strides = var_335_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_5_cast_fp16)[name = string("op_335_cast_fp16")]; + tensor input_9_cast_fp16 = mul(x = var_329_cast_fp16, y = var_335_cast_fp16)[name = string("input_9_cast_fp16")]; + string h_1_pad_type_0 = const()[name = string("h_1_pad_type_0"), val = string("valid")]; + tensor h_1_strides_0 = const()[name = string("h_1_strides_0"), val = tensor([1, 1])]; + tensor h_1_pad_0 = const()[name = string("h_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_1_dilations_0 = const()[name = string("h_1_dilations_0"), val = tensor([1, 1])]; + int32 h_1_groups_0 = const()[name = string("h_1_groups_0"), val = int32(1)]; + tensor layers_0_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12606272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15752064))))[name = string("layers_0_mlp_down_proj_weight_to_fp16_palettized")]; + tensor h_1_cast_fp16 = conv(dilations = h_1_dilations_0, groups = h_1_groups_0, pad = h_1_pad_0, pad_type = h_1_pad_type_0, strides = h_1_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_9_cast_fp16)[name = string("h_1_cast_fp16")]; + tensor x_7_cast_fp16 = add(x = x_5_cast_fp16, y = h_1_cast_fp16)[name = string("x_7_cast_fp16")]; + int32 var_388 = const()[name = string("op_388"), val = int32(2)]; + int32 var_392 = const()[name = string("op_392"), val = int32(3)]; + tensor var_407_cast_fp16 = mul(x = x_7_cast_fp16, y = x_7_cast_fp16)[name = string("op_407_cast_fp16")]; + tensor variance_9_axes_0 = const()[name = string("variance_9_axes_0"), val = tensor([1])]; + bool variance_9_keep_dims_0 = const()[name = string("variance_9_keep_dims_0"), val = bool(true)]; + tensor variance_9_cast_fp16 = reduce_mean(axes = variance_9_axes_0, keep_dims = variance_9_keep_dims_0, x = var_407_cast_fp16)[name = string("variance_9_cast_fp16")]; + fp16 var_410_to_fp16 = const()[name = string("op_410_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_411_cast_fp16 = add(x = variance_9_cast_fp16, y = var_410_to_fp16)[name = string("op_411_cast_fp16")]; + fp32 var_412_epsilon_0 = const()[name = string("op_412_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_412_cast_fp16 = rsqrt(epsilon = var_412_epsilon_0, x = var_411_cast_fp16)[name = string("op_412_cast_fp16")]; + tensor var_413_cast_fp16 = mul(x = x_7_cast_fp16, y = var_412_cast_fp16)[name = string("op_413_cast_fp16")]; + tensor layers_1_input_layernorm_weight_to_fp16 = const()[name = string("layers_1_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15752640)))]; + tensor input_11_cast_fp16 = mul(x = var_413_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_11_cast_fp16")]; + string q_7_pad_type_0 = const()[name = string("q_7_pad_type_0"), val = string("valid")]; + tensor q_7_strides_0 = const()[name = string("q_7_strides_0"), val = tensor([1, 1])]; + tensor q_7_pad_0 = const()[name = string("q_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_7_dilations_0 = const()[name = string("q_7_dilations_0"), val = tensor([1, 1])]; + int32 q_7_groups_0 = const()[name = string("q_7_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15754752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17851968))))[name = string("layers_1_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor q_7_cast_fp16 = conv(dilations = q_7_dilations_0, groups = q_7_groups_0, pad = q_7_pad_0, pad_type = q_7_pad_type_0, strides = q_7_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_11_cast_fp16)[name = string("q_7_cast_fp16")]; + string k_7_pad_type_0 = const()[name = string("k_7_pad_type_0"), val = string("valid")]; + tensor k_7_strides_0 = const()[name = string("k_7_strides_0"), val = tensor([1, 1])]; + tensor k_7_pad_0 = const()[name = string("k_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_7_dilations_0 = const()[name = string("k_7_dilations_0"), val = tensor([1, 1])]; + int32 k_7_groups_0 = const()[name = string("k_7_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17852544))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18901184))))[name = string("layers_1_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor k_7_cast_fp16 = conv(dilations = k_7_dilations_0, groups = k_7_groups_0, pad = k_7_pad_0, pad_type = k_7_pad_type_0, strides = k_7_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_11_cast_fp16)[name = string("k_7_cast_fp16")]; + string v_3_pad_type_0 = const()[name = string("v_3_pad_type_0"), val = string("valid")]; + tensor v_3_strides_0 = const()[name = string("v_3_strides_0"), val = tensor([1, 1])]; + tensor v_3_pad_0 = const()[name = string("v_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_3_dilations_0 = const()[name = string("v_3_dilations_0"), val = tensor([1, 1])]; + int32 v_3_groups_0 = const()[name = string("v_3_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18901760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19950400))))[name = string("layers_1_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor v_3_cast_fp16 = conv(dilations = v_3_dilations_0, groups = v_3_groups_0, pad = v_3_pad_0, pad_type = v_3_pad_type_0, strides = v_3_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_11_cast_fp16)[name = string("v_3_cast_fp16")]; + tensor var_447 = const()[name = string("op_447"), val = tensor([16, 128, 1, 1])]; + tensor x_9_cast_fp16 = reshape(shape = var_447, x = q_7_cast_fp16)[name = string("x_9_cast_fp16")]; + tensor var_450_cast_fp16 = mul(x = x_9_cast_fp16, y = x_9_cast_fp16)[name = string("op_450_cast_fp16")]; + tensor variance_11_axes_0 = const()[name = string("variance_11_axes_0"), val = tensor([1])]; + bool variance_11_keep_dims_0 = const()[name = string("variance_11_keep_dims_0"), val = bool(true)]; + tensor variance_11_cast_fp16 = reduce_mean(axes = variance_11_axes_0, keep_dims = variance_11_keep_dims_0, x = var_450_cast_fp16)[name = string("variance_11_cast_fp16")]; + fp16 var_453_to_fp16 = const()[name = string("op_453_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_454_cast_fp16 = add(x = variance_11_cast_fp16, y = var_453_to_fp16)[name = string("op_454_cast_fp16")]; + fp32 var_455_epsilon_0 = const()[name = string("op_455_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_455_cast_fp16 = rsqrt(epsilon = var_455_epsilon_0, x = var_454_cast_fp16)[name = string("op_455_cast_fp16")]; + tensor var_456_cast_fp16 = mul(x = x_9_cast_fp16, y = var_455_cast_fp16)[name = string("op_456_cast_fp16")]; + tensor layers_1_self_attn_q_norm_weight_to_fp16 = const()[name = string("layers_1_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19950976)))]; + tensor q_9_cast_fp16 = mul(x = var_456_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_9_cast_fp16")]; + tensor var_458 = const()[name = string("op_458"), val = tensor([8, 128, 1, 1])]; + tensor x_11_cast_fp16 = reshape(shape = var_458, x = k_7_cast_fp16)[name = string("x_11_cast_fp16")]; + tensor var_461_cast_fp16 = mul(x = x_11_cast_fp16, y = x_11_cast_fp16)[name = string("op_461_cast_fp16")]; + tensor variance_13_axes_0 = const()[name = string("variance_13_axes_0"), val = tensor([1])]; + bool variance_13_keep_dims_0 = const()[name = string("variance_13_keep_dims_0"), val = bool(true)]; + tensor variance_13_cast_fp16 = reduce_mean(axes = variance_13_axes_0, keep_dims = variance_13_keep_dims_0, x = var_461_cast_fp16)[name = string("variance_13_cast_fp16")]; + fp16 var_464_to_fp16 = const()[name = string("op_464_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_465_cast_fp16 = add(x = variance_13_cast_fp16, y = var_464_to_fp16)[name = string("op_465_cast_fp16")]; + fp32 var_466_epsilon_0 = const()[name = string("op_466_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_466_cast_fp16 = rsqrt(epsilon = var_466_epsilon_0, x = var_465_cast_fp16)[name = string("op_466_cast_fp16")]; + tensor var_467_cast_fp16 = mul(x = x_11_cast_fp16, y = var_466_cast_fp16)[name = string("op_467_cast_fp16")]; + tensor layers_1_self_attn_k_norm_weight_to_fp16 = const()[name = string("layers_1_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19951296)))]; + tensor k_9_cast_fp16 = mul(x = var_467_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_9_cast_fp16")]; + tensor var_469 = const()[name = string("op_469"), val = tensor([1, 16, 128, 1])]; + tensor z_5_cast_fp16 = reshape(shape = var_469, x = q_9_cast_fp16)[name = string("z_5_cast_fp16")]; + tensor var_471 = const()[name = string("op_471"), val = tensor([1, 8, 128, 1])]; + tensor z_7_cast_fp16 = reshape(shape = var_471, x = k_9_cast_fp16)[name = string("z_7_cast_fp16")]; + tensor z1_5_begin_0 = const()[name = string("z1_5_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_5_end_0 = const()[name = string("z1_5_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_5_end_mask_0 = const()[name = string("z1_5_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_5_cast_fp16 = slice_by_index(begin = z1_5_begin_0, end = z1_5_end_0, end_mask = z1_5_end_mask_0, x = z_5_cast_fp16)[name = string("z1_5_cast_fp16")]; + tensor z2_5_begin_0 = const()[name = string("z2_5_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_5_end_0 = const()[name = string("z2_5_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_5_end_mask_0 = const()[name = string("z2_5_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_5_cast_fp16 = slice_by_index(begin = z2_5_begin_0, end = z2_5_end_0, end_mask = z2_5_end_mask_0, x = z_5_cast_fp16)[name = string("z2_5_cast_fp16")]; + fp16 const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_480_cast_fp16 = mul(x = z2_5_cast_fp16, y = const_3_promoted_to_fp16)[name = string("op_480_cast_fp16")]; + bool var_482_interleave_0 = const()[name = string("op_482_interleave_0"), val = bool(false)]; + tensor var_482_cast_fp16 = concat(axis = var_388, interleave = var_482_interleave_0, values = (var_480_cast_fp16, z1_5_cast_fp16))[name = string("op_482_cast_fp16")]; + tensor var_483_cast_fp16 = mul(x = var_482_cast_fp16, y = sin_1_to_fp16)[name = string("op_483_cast_fp16")]; + tensor q_11_cast_fp16 = add(x = z_5_cast_fp16, y = var_483_cast_fp16)[name = string("q_11_cast_fp16")]; + tensor z1_7_begin_0 = const()[name = string("z1_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_7_end_0 = const()[name = string("z1_7_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_7_end_mask_0 = const()[name = string("z1_7_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_7_cast_fp16 = slice_by_index(begin = z1_7_begin_0, end = z1_7_end_0, end_mask = z1_7_end_mask_0, x = z_7_cast_fp16)[name = string("z1_7_cast_fp16")]; + tensor z2_7_begin_0 = const()[name = string("z2_7_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_7_end_0 = const()[name = string("z2_7_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_7_end_mask_0 = const()[name = string("z2_7_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_7_cast_fp16 = slice_by_index(begin = z2_7_begin_0, end = z2_7_end_0, end_mask = z2_7_end_mask_0, x = z_7_cast_fp16)[name = string("z2_7_cast_fp16")]; + fp16 const_4_promoted_to_fp16 = const()[name = string("const_4_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_492_cast_fp16 = mul(x = z2_7_cast_fp16, y = const_4_promoted_to_fp16)[name = string("op_492_cast_fp16")]; + bool var_494_interleave_0 = const()[name = string("op_494_interleave_0"), val = bool(false)]; + tensor var_494_cast_fp16 = concat(axis = var_388, interleave = var_494_interleave_0, values = (var_492_cast_fp16, z1_7_cast_fp16))[name = string("op_494_cast_fp16")]; + tensor var_495_cast_fp16 = mul(x = var_494_cast_fp16, y = sin_1_to_fp16)[name = string("op_495_cast_fp16")]; + tensor k_11_cast_fp16 = add(x = z_7_cast_fp16, y = var_495_cast_fp16)[name = string("k_11_cast_fp16")]; + tensor var_497 = const()[name = string("op_497"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_3_cast_fp16 = reshape(shape = var_497, x = k_11_cast_fp16)[name = string("cur_key_3_cast_fp16")]; + tensor upd_3_to_fp16 = const()[name = string("upd_3_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110173888)))]; + tensor var_501_cast_fp16 = mul(x = cur_key_3_cast_fp16, y = upd_3_to_fp16)[name = string("op_501_cast_fp16")]; + tensor var_505_cast_fp16 = mul(x = v_3_cast_fp16, y = upd_3_to_fp16)[name = string("op_505_cast_fp16")]; + tensor var_507 = const()[name = string("op_507"), val = tensor([1, 8, 128, 16])]; + tensor kh_5_cast_fp16 = reshape(shape = var_507, x = var_501_cast_fp16)[name = string("kh_5_cast_fp16")]; + tensor var_509 = const()[name = string("op_509"), val = tensor([1, 8, 128, 16])]; + tensor vh_5_cast_fp16 = reshape(shape = var_509, x = var_505_cast_fp16)[name = string("vh_5_cast_fp16")]; + tensor transpose_4_perm_0 = const()[name = string("transpose_4_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_2_reps_0 = const()[name = string("tile_2_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_4_cast_fp16 = transpose(perm = transpose_4_perm_0, x = kh_5_cast_fp16)[name = string("transpose_467")]; + tensor tile_2_cast_fp16 = tile(reps = tile_2_reps_0, x = transpose_4_cast_fp16)[name = string("tile_2_cast_fp16")]; + tensor concat_8 = const()[name = string("concat_8"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_4_cast_fp16 = reshape(shape = concat_8, x = tile_2_cast_fp16)[name = string("reshape_4_cast_fp16")]; + tensor transpose_5_perm_0 = const()[name = string("transpose_5_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_9 = const()[name = string("concat_9"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_5_cast_fp16 = transpose(perm = transpose_5_perm_0, x = reshape_4_cast_fp16)[name = string("transpose_466")]; + tensor reshape_5_cast_fp16 = reshape(shape = concat_9, x = transpose_5_cast_fp16)[name = string("reshape_5_cast_fp16")]; + tensor transpose_6_perm_0 = const()[name = string("transpose_6_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_3_reps_0 = const()[name = string("tile_3_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_6_cast_fp16 = transpose(perm = transpose_6_perm_0, x = vh_5_cast_fp16)[name = string("transpose_465")]; + tensor tile_3_cast_fp16 = tile(reps = tile_3_reps_0, x = transpose_6_cast_fp16)[name = string("tile_3_cast_fp16")]; + tensor concat_10 = const()[name = string("concat_10"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_6_cast_fp16 = reshape(shape = concat_10, x = tile_3_cast_fp16)[name = string("reshape_6_cast_fp16")]; + tensor transpose_7_perm_0 = const()[name = string("transpose_7_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_11 = const()[name = string("concat_11"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_7_cast_fp16 = transpose(perm = transpose_7_perm_0, x = reshape_6_cast_fp16)[name = string("transpose_464")]; + tensor reshape_7_cast_fp16 = reshape(shape = concat_11, x = transpose_7_cast_fp16)[name = string("reshape_7_cast_fp16")]; + fp16 var_513_to_fp16 = const()[name = string("op_513_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_514_cast_fp16 = mul(x = q_11_cast_fp16, y = var_513_to_fp16)[name = string("op_514_cast_fp16")]; + tensor transpose_325_perm_0 = const()[name = string("transpose_325_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_5_transpose_x_1 = const()[name = string("w_5_transpose_x_1"), val = bool(true)]; + bool w_5_transpose_y_1 = const()[name = string("w_5_transpose_y_1"), val = bool(false)]; + tensor transpose_325_cast_fp16 = transpose(perm = transpose_325_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_463")]; + tensor w_5_cast_fp16 = matmul(transpose_x = w_5_transpose_x_1, transpose_y = w_5_transpose_y_1, x = var_514_cast_fp16, y = transpose_325_cast_fp16)[name = string("w_5_cast_fp16")]; + tensor pad_3_to_fp16 = const()[name = string("pad_3_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174016)))]; + tensor var_517_cast_fp16 = add(x = w_5_cast_fp16, y = pad_3_to_fp16)[name = string("op_517_cast_fp16")]; + tensor w_7_cast_fp16 = softmax(axis = var_392, x = var_517_cast_fp16)[name = string("w_7_cast_fp16")]; + tensor transpose_326_perm_0 = const()[name = string("transpose_326_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_3_transpose_x_1 = const()[name = string("attn_3_transpose_x_1"), val = bool(false)]; + bool attn_3_transpose_y_1 = const()[name = string("attn_3_transpose_y_1"), val = bool(true)]; + tensor transpose_326_cast_fp16 = transpose(perm = transpose_326_perm_0, x = reshape_7_cast_fp16)[name = string("transpose_462")]; + tensor attn_3_cast_fp16 = matmul(transpose_x = attn_3_transpose_x_1, transpose_y = attn_3_transpose_y_1, x = transpose_326_cast_fp16, y = w_7_cast_fp16)[name = string("attn_3_cast_fp16")]; + tensor var_521 = const()[name = string("op_521"), val = tensor([1, 2048, 1, 1])]; + tensor input_13_cast_fp16 = reshape(shape = var_521, x = attn_3_cast_fp16)[name = string("input_13_cast_fp16")]; + string attn_output_3_pad_type_0 = const()[name = string("attn_output_3_pad_type_0"), val = string("valid")]; + tensor attn_output_3_strides_0 = const()[name = string("attn_output_3_strides_0"), val = tensor([1, 1])]; + tensor attn_output_3_pad_0 = const()[name = string("attn_output_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_3_dilations_0 = const()[name = string("attn_output_3_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_3_groups_0 = const()[name = string("attn_output_3_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19951616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22048832))))[name = string("layers_1_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor attn_output_3_cast_fp16 = conv(dilations = attn_output_3_dilations_0, groups = attn_output_3_groups_0, pad = attn_output_3_pad_0, pad_type = attn_output_3_pad_type_0, strides = attn_output_3_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_13_cast_fp16)[name = string("attn_output_3_cast_fp16")]; + tensor x_13_cast_fp16 = add(x = x_7_cast_fp16, y = attn_output_3_cast_fp16)[name = string("x_13_cast_fp16")]; + tensor var_535_cast_fp16 = mul(x = x_13_cast_fp16, y = x_13_cast_fp16)[name = string("op_535_cast_fp16")]; + tensor variance_15_axes_0 = const()[name = string("variance_15_axes_0"), val = tensor([1])]; + bool variance_15_keep_dims_0 = const()[name = string("variance_15_keep_dims_0"), val = bool(true)]; + tensor variance_15_cast_fp16 = reduce_mean(axes = variance_15_axes_0, keep_dims = variance_15_keep_dims_0, x = var_535_cast_fp16)[name = string("variance_15_cast_fp16")]; + fp16 var_538_to_fp16 = const()[name = string("op_538_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_539_cast_fp16 = add(x = variance_15_cast_fp16, y = var_538_to_fp16)[name = string("op_539_cast_fp16")]; + fp32 var_540_epsilon_0 = const()[name = string("op_540_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_540_cast_fp16 = rsqrt(epsilon = var_540_epsilon_0, x = var_539_cast_fp16)[name = string("op_540_cast_fp16")]; + tensor var_541_cast_fp16 = mul(x = x_13_cast_fp16, y = var_540_cast_fp16)[name = string("op_541_cast_fp16")]; + tensor layers_1_post_attention_layernorm_weight_to_fp16 = const()[name = string("layers_1_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22049408)))]; + tensor input_15_cast_fp16 = mul(x = var_541_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_15_cast_fp16")]; + string input_17_pad_type_0 = const()[name = string("input_17_pad_type_0"), val = string("valid")]; + tensor input_17_strides_0 = const()[name = string("input_17_strides_0"), val = tensor([1, 1])]; + tensor input_17_pad_0 = const()[name = string("input_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_17_dilations_0 = const()[name = string("input_17_dilations_0"), val = tensor([1, 1])]; + int32 input_17_groups_0 = const()[name = string("input_17_groups_0"), val = int32(1)]; + tensor layers_1_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22051520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25197312))))[name = string("layers_1_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_17_cast_fp16 = conv(dilations = input_17_dilations_0, groups = input_17_groups_0, pad = input_17_pad_0, pad_type = input_17_pad_type_0, strides = input_17_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_15_cast_fp16)[name = string("input_17_cast_fp16")]; + tensor var_549_cast_fp16 = silu(x = input_17_cast_fp16)[name = string("op_549_cast_fp16")]; + string var_555_pad_type_0 = const()[name = string("op_555_pad_type_0"), val = string("valid")]; + tensor var_555_strides_0 = const()[name = string("op_555_strides_0"), val = tensor([1, 1])]; + tensor var_555_pad_0 = const()[name = string("op_555_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_555_dilations_0 = const()[name = string("op_555_dilations_0"), val = tensor([1, 1])]; + int32 var_555_groups_0 = const()[name = string("op_555_groups_0"), val = int32(1)]; + tensor layers_1_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25197888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28343680))))[name = string("layers_1_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_555_cast_fp16 = conv(dilations = var_555_dilations_0, groups = var_555_groups_0, pad = var_555_pad_0, pad_type = var_555_pad_type_0, strides = var_555_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_15_cast_fp16)[name = string("op_555_cast_fp16")]; + tensor input_19_cast_fp16 = mul(x = var_549_cast_fp16, y = var_555_cast_fp16)[name = string("input_19_cast_fp16")]; + string h_3_pad_type_0 = const()[name = string("h_3_pad_type_0"), val = string("valid")]; + tensor h_3_strides_0 = const()[name = string("h_3_strides_0"), val = tensor([1, 1])]; + tensor h_3_pad_0 = const()[name = string("h_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_3_dilations_0 = const()[name = string("h_3_dilations_0"), val = tensor([1, 1])]; + int32 h_3_groups_0 = const()[name = string("h_3_groups_0"), val = int32(1)]; + tensor layers_1_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28344256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31490048))))[name = string("layers_1_mlp_down_proj_weight_to_fp16_palettized")]; + tensor h_3_cast_fp16 = conv(dilations = h_3_dilations_0, groups = h_3_groups_0, pad = h_3_pad_0, pad_type = h_3_pad_type_0, strides = h_3_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_19_cast_fp16)[name = string("h_3_cast_fp16")]; + tensor x_15_cast_fp16 = add(x = x_13_cast_fp16, y = h_3_cast_fp16)[name = string("x_15_cast_fp16")]; + int32 var_608 = const()[name = string("op_608"), val = int32(2)]; + int32 var_612 = const()[name = string("op_612"), val = int32(3)]; + tensor var_627_cast_fp16 = mul(x = x_15_cast_fp16, y = x_15_cast_fp16)[name = string("op_627_cast_fp16")]; + tensor variance_17_axes_0 = const()[name = string("variance_17_axes_0"), val = tensor([1])]; + bool variance_17_keep_dims_0 = const()[name = string("variance_17_keep_dims_0"), val = bool(true)]; + tensor variance_17_cast_fp16 = reduce_mean(axes = variance_17_axes_0, keep_dims = variance_17_keep_dims_0, x = var_627_cast_fp16)[name = string("variance_17_cast_fp16")]; + fp16 var_630_to_fp16 = const()[name = string("op_630_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_631_cast_fp16 = add(x = variance_17_cast_fp16, y = var_630_to_fp16)[name = string("op_631_cast_fp16")]; + fp32 var_632_epsilon_0 = const()[name = string("op_632_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_632_cast_fp16 = rsqrt(epsilon = var_632_epsilon_0, x = var_631_cast_fp16)[name = string("op_632_cast_fp16")]; + tensor var_633_cast_fp16 = mul(x = x_15_cast_fp16, y = var_632_cast_fp16)[name = string("op_633_cast_fp16")]; + tensor layers_2_input_layernorm_weight_to_fp16 = const()[name = string("layers_2_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31490624)))]; + tensor input_21_cast_fp16 = mul(x = var_633_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_21_cast_fp16")]; + string q_13_pad_type_0 = const()[name = string("q_13_pad_type_0"), val = string("valid")]; + tensor q_13_strides_0 = const()[name = string("q_13_strides_0"), val = tensor([1, 1])]; + tensor q_13_pad_0 = const()[name = string("q_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_13_dilations_0 = const()[name = string("q_13_dilations_0"), val = tensor([1, 1])]; + int32 q_13_groups_0 = const()[name = string("q_13_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31492736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33589952))))[name = string("layers_2_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor q_13_cast_fp16 = conv(dilations = q_13_dilations_0, groups = q_13_groups_0, pad = q_13_pad_0, pad_type = q_13_pad_type_0, strides = q_13_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_21_cast_fp16)[name = string("q_13_cast_fp16")]; + string k_13_pad_type_0 = const()[name = string("k_13_pad_type_0"), val = string("valid")]; + tensor k_13_strides_0 = const()[name = string("k_13_strides_0"), val = tensor([1, 1])]; + tensor k_13_pad_0 = const()[name = string("k_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_13_dilations_0 = const()[name = string("k_13_dilations_0"), val = tensor([1, 1])]; + int32 k_13_groups_0 = const()[name = string("k_13_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33590528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34639168))))[name = string("layers_2_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor k_13_cast_fp16 = conv(dilations = k_13_dilations_0, groups = k_13_groups_0, pad = k_13_pad_0, pad_type = k_13_pad_type_0, strides = k_13_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_21_cast_fp16)[name = string("k_13_cast_fp16")]; + string v_5_pad_type_0 = const()[name = string("v_5_pad_type_0"), val = string("valid")]; + tensor v_5_strides_0 = const()[name = string("v_5_strides_0"), val = tensor([1, 1])]; + tensor v_5_pad_0 = const()[name = string("v_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_5_dilations_0 = const()[name = string("v_5_dilations_0"), val = tensor([1, 1])]; + int32 v_5_groups_0 = const()[name = string("v_5_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34639744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35688384))))[name = string("layers_2_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor v_5_cast_fp16 = conv(dilations = v_5_dilations_0, groups = v_5_groups_0, pad = v_5_pad_0, pad_type = v_5_pad_type_0, strides = v_5_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_21_cast_fp16)[name = string("v_5_cast_fp16")]; + tensor var_667 = const()[name = string("op_667"), val = tensor([16, 128, 1, 1])]; + tensor x_17_cast_fp16 = reshape(shape = var_667, x = q_13_cast_fp16)[name = string("x_17_cast_fp16")]; + tensor var_670_cast_fp16 = mul(x = x_17_cast_fp16, y = x_17_cast_fp16)[name = string("op_670_cast_fp16")]; + tensor variance_19_axes_0 = const()[name = string("variance_19_axes_0"), val = tensor([1])]; + bool variance_19_keep_dims_0 = const()[name = string("variance_19_keep_dims_0"), val = bool(true)]; + tensor variance_19_cast_fp16 = reduce_mean(axes = variance_19_axes_0, keep_dims = variance_19_keep_dims_0, x = var_670_cast_fp16)[name = string("variance_19_cast_fp16")]; + fp16 var_673_to_fp16 = const()[name = string("op_673_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_674_cast_fp16 = add(x = variance_19_cast_fp16, y = var_673_to_fp16)[name = string("op_674_cast_fp16")]; + fp32 var_675_epsilon_0 = const()[name = string("op_675_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_675_cast_fp16 = rsqrt(epsilon = var_675_epsilon_0, x = var_674_cast_fp16)[name = string("op_675_cast_fp16")]; + tensor var_676_cast_fp16 = mul(x = x_17_cast_fp16, y = var_675_cast_fp16)[name = string("op_676_cast_fp16")]; + tensor layers_2_self_attn_q_norm_weight_to_fp16 = const()[name = string("layers_2_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35688960)))]; + tensor q_15_cast_fp16 = mul(x = var_676_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_15_cast_fp16")]; + tensor var_678 = const()[name = string("op_678"), val = tensor([8, 128, 1, 1])]; + tensor x_19_cast_fp16 = reshape(shape = var_678, x = k_13_cast_fp16)[name = string("x_19_cast_fp16")]; + tensor var_681_cast_fp16 = mul(x = x_19_cast_fp16, y = x_19_cast_fp16)[name = string("op_681_cast_fp16")]; + tensor variance_21_axes_0 = const()[name = string("variance_21_axes_0"), val = tensor([1])]; + bool variance_21_keep_dims_0 = const()[name = string("variance_21_keep_dims_0"), val = bool(true)]; + tensor variance_21_cast_fp16 = reduce_mean(axes = variance_21_axes_0, keep_dims = variance_21_keep_dims_0, x = var_681_cast_fp16)[name = string("variance_21_cast_fp16")]; + fp16 var_684_to_fp16 = const()[name = string("op_684_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_685_cast_fp16 = add(x = variance_21_cast_fp16, y = var_684_to_fp16)[name = string("op_685_cast_fp16")]; + fp32 var_686_epsilon_0 = const()[name = string("op_686_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_686_cast_fp16 = rsqrt(epsilon = var_686_epsilon_0, x = var_685_cast_fp16)[name = string("op_686_cast_fp16")]; + tensor var_687_cast_fp16 = mul(x = x_19_cast_fp16, y = var_686_cast_fp16)[name = string("op_687_cast_fp16")]; + tensor layers_2_self_attn_k_norm_weight_to_fp16 = const()[name = string("layers_2_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35689280)))]; + tensor k_15_cast_fp16 = mul(x = var_687_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_15_cast_fp16")]; + tensor var_689 = const()[name = string("op_689"), val = tensor([1, 16, 128, 1])]; + tensor z_9_cast_fp16 = reshape(shape = var_689, x = q_15_cast_fp16)[name = string("z_9_cast_fp16")]; + tensor var_691 = const()[name = string("op_691"), val = tensor([1, 8, 128, 1])]; + tensor z_11_cast_fp16 = reshape(shape = var_691, x = k_15_cast_fp16)[name = string("z_11_cast_fp16")]; + tensor z1_9_begin_0 = const()[name = string("z1_9_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_9_end_0 = const()[name = string("z1_9_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_9_end_mask_0 = const()[name = string("z1_9_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_9_cast_fp16 = slice_by_index(begin = z1_9_begin_0, end = z1_9_end_0, end_mask = z1_9_end_mask_0, x = z_9_cast_fp16)[name = string("z1_9_cast_fp16")]; + tensor z2_9_begin_0 = const()[name = string("z2_9_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_9_end_0 = const()[name = string("z2_9_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_9_end_mask_0 = const()[name = string("z2_9_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_9_cast_fp16 = slice_by_index(begin = z2_9_begin_0, end = z2_9_end_0, end_mask = z2_9_end_mask_0, x = z_9_cast_fp16)[name = string("z2_9_cast_fp16")]; + fp16 const_5_promoted_to_fp16 = const()[name = string("const_5_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_700_cast_fp16 = mul(x = z2_9_cast_fp16, y = const_5_promoted_to_fp16)[name = string("op_700_cast_fp16")]; + bool var_702_interleave_0 = const()[name = string("op_702_interleave_0"), val = bool(false)]; + tensor var_702_cast_fp16 = concat(axis = var_608, interleave = var_702_interleave_0, values = (var_700_cast_fp16, z1_9_cast_fp16))[name = string("op_702_cast_fp16")]; + tensor var_703_cast_fp16 = mul(x = var_702_cast_fp16, y = sin_1_to_fp16)[name = string("op_703_cast_fp16")]; + tensor q_17_cast_fp16 = add(x = z_9_cast_fp16, y = var_703_cast_fp16)[name = string("q_17_cast_fp16")]; + tensor z1_11_begin_0 = const()[name = string("z1_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_11_end_0 = const()[name = string("z1_11_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_11_end_mask_0 = const()[name = string("z1_11_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_11_cast_fp16 = slice_by_index(begin = z1_11_begin_0, end = z1_11_end_0, end_mask = z1_11_end_mask_0, x = z_11_cast_fp16)[name = string("z1_11_cast_fp16")]; + tensor z2_11_begin_0 = const()[name = string("z2_11_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_11_end_0 = const()[name = string("z2_11_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_11_end_mask_0 = const()[name = string("z2_11_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_11_cast_fp16 = slice_by_index(begin = z2_11_begin_0, end = z2_11_end_0, end_mask = z2_11_end_mask_0, x = z_11_cast_fp16)[name = string("z2_11_cast_fp16")]; + fp16 const_6_promoted_to_fp16 = const()[name = string("const_6_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_712_cast_fp16 = mul(x = z2_11_cast_fp16, y = const_6_promoted_to_fp16)[name = string("op_712_cast_fp16")]; + bool var_714_interleave_0 = const()[name = string("op_714_interleave_0"), val = bool(false)]; + tensor var_714_cast_fp16 = concat(axis = var_608, interleave = var_714_interleave_0, values = (var_712_cast_fp16, z1_11_cast_fp16))[name = string("op_714_cast_fp16")]; + tensor var_715_cast_fp16 = mul(x = var_714_cast_fp16, y = sin_1_to_fp16)[name = string("op_715_cast_fp16")]; + tensor k_17_cast_fp16 = add(x = z_11_cast_fp16, y = var_715_cast_fp16)[name = string("k_17_cast_fp16")]; + tensor var_717 = const()[name = string("op_717"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_5_cast_fp16 = reshape(shape = var_717, x = k_17_cast_fp16)[name = string("cur_key_5_cast_fp16")]; + tensor upd_5_to_fp16 = const()[name = string("upd_5_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110173888)))]; + tensor var_721_cast_fp16 = mul(x = cur_key_5_cast_fp16, y = upd_5_to_fp16)[name = string("op_721_cast_fp16")]; + tensor var_725_cast_fp16 = mul(x = v_5_cast_fp16, y = upd_5_to_fp16)[name = string("op_725_cast_fp16")]; + tensor var_727 = const()[name = string("op_727"), val = tensor([1, 8, 128, 16])]; + tensor kh_9_cast_fp16 = reshape(shape = var_727, x = var_721_cast_fp16)[name = string("kh_9_cast_fp16")]; + tensor var_729 = const()[name = string("op_729"), val = tensor([1, 8, 128, 16])]; + tensor vh_9_cast_fp16 = reshape(shape = var_729, x = var_725_cast_fp16)[name = string("vh_9_cast_fp16")]; + tensor transpose_8_perm_0 = const()[name = string("transpose_8_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_4_reps_0 = const()[name = string("tile_4_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_8_cast_fp16 = transpose(perm = transpose_8_perm_0, x = kh_9_cast_fp16)[name = string("transpose_461")]; + tensor tile_4_cast_fp16 = tile(reps = tile_4_reps_0, x = transpose_8_cast_fp16)[name = string("tile_4_cast_fp16")]; + tensor concat_12 = const()[name = string("concat_12"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_8_cast_fp16 = reshape(shape = concat_12, x = tile_4_cast_fp16)[name = string("reshape_8_cast_fp16")]; + tensor transpose_9_perm_0 = const()[name = string("transpose_9_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_13 = const()[name = string("concat_13"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_9_cast_fp16 = transpose(perm = transpose_9_perm_0, x = reshape_8_cast_fp16)[name = string("transpose_460")]; + tensor reshape_9_cast_fp16 = reshape(shape = concat_13, x = transpose_9_cast_fp16)[name = string("reshape_9_cast_fp16")]; + tensor transpose_10_perm_0 = const()[name = string("transpose_10_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_5_reps_0 = const()[name = string("tile_5_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_10_cast_fp16 = transpose(perm = transpose_10_perm_0, x = vh_9_cast_fp16)[name = string("transpose_459")]; + tensor tile_5_cast_fp16 = tile(reps = tile_5_reps_0, x = transpose_10_cast_fp16)[name = string("tile_5_cast_fp16")]; + tensor concat_14 = const()[name = string("concat_14"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_10_cast_fp16 = reshape(shape = concat_14, x = tile_5_cast_fp16)[name = string("reshape_10_cast_fp16")]; + tensor transpose_11_perm_0 = const()[name = string("transpose_11_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_15 = const()[name = string("concat_15"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_11_cast_fp16 = transpose(perm = transpose_11_perm_0, x = reshape_10_cast_fp16)[name = string("transpose_458")]; + tensor reshape_11_cast_fp16 = reshape(shape = concat_15, x = transpose_11_cast_fp16)[name = string("reshape_11_cast_fp16")]; + fp16 var_733_to_fp16 = const()[name = string("op_733_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_734_cast_fp16 = mul(x = q_17_cast_fp16, y = var_733_to_fp16)[name = string("op_734_cast_fp16")]; + tensor transpose_329_perm_0 = const()[name = string("transpose_329_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_9_transpose_x_1 = const()[name = string("w_9_transpose_x_1"), val = bool(true)]; + bool w_9_transpose_y_1 = const()[name = string("w_9_transpose_y_1"), val = bool(false)]; + tensor transpose_329_cast_fp16 = transpose(perm = transpose_329_perm_0, x = reshape_9_cast_fp16)[name = string("transpose_457")]; + tensor w_9_cast_fp16 = matmul(transpose_x = w_9_transpose_x_1, transpose_y = w_9_transpose_y_1, x = var_734_cast_fp16, y = transpose_329_cast_fp16)[name = string("w_9_cast_fp16")]; + tensor pad_5_to_fp16 = const()[name = string("pad_5_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174016)))]; + tensor var_737_cast_fp16 = add(x = w_9_cast_fp16, y = pad_5_to_fp16)[name = string("op_737_cast_fp16")]; + tensor w_11_cast_fp16 = softmax(axis = var_612, x = var_737_cast_fp16)[name = string("w_11_cast_fp16")]; + tensor transpose_330_perm_0 = const()[name = string("transpose_330_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_5_transpose_x_1 = const()[name = string("attn_5_transpose_x_1"), val = bool(false)]; + bool attn_5_transpose_y_1 = const()[name = string("attn_5_transpose_y_1"), val = bool(true)]; + tensor transpose_330_cast_fp16 = transpose(perm = transpose_330_perm_0, x = reshape_11_cast_fp16)[name = string("transpose_456")]; + tensor attn_5_cast_fp16 = matmul(transpose_x = attn_5_transpose_x_1, transpose_y = attn_5_transpose_y_1, x = transpose_330_cast_fp16, y = w_11_cast_fp16)[name = string("attn_5_cast_fp16")]; + tensor var_741 = const()[name = string("op_741"), val = tensor([1, 2048, 1, 1])]; + tensor input_23_cast_fp16 = reshape(shape = var_741, x = attn_5_cast_fp16)[name = string("input_23_cast_fp16")]; + string attn_output_5_pad_type_0 = const()[name = string("attn_output_5_pad_type_0"), val = string("valid")]; + tensor attn_output_5_strides_0 = const()[name = string("attn_output_5_strides_0"), val = tensor([1, 1])]; + tensor attn_output_5_pad_0 = const()[name = string("attn_output_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_5_dilations_0 = const()[name = string("attn_output_5_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_5_groups_0 = const()[name = string("attn_output_5_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35689600))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37786816))))[name = string("layers_2_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor attn_output_5_cast_fp16 = conv(dilations = attn_output_5_dilations_0, groups = attn_output_5_groups_0, pad = attn_output_5_pad_0, pad_type = attn_output_5_pad_type_0, strides = attn_output_5_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_23_cast_fp16)[name = string("attn_output_5_cast_fp16")]; + tensor x_21_cast_fp16 = add(x = x_15_cast_fp16, y = attn_output_5_cast_fp16)[name = string("x_21_cast_fp16")]; + tensor var_755_cast_fp16 = mul(x = x_21_cast_fp16, y = x_21_cast_fp16)[name = string("op_755_cast_fp16")]; + tensor variance_23_axes_0 = const()[name = string("variance_23_axes_0"), val = tensor([1])]; + bool variance_23_keep_dims_0 = const()[name = string("variance_23_keep_dims_0"), val = bool(true)]; + tensor variance_23_cast_fp16 = reduce_mean(axes = variance_23_axes_0, keep_dims = variance_23_keep_dims_0, x = var_755_cast_fp16)[name = string("variance_23_cast_fp16")]; + fp16 var_758_to_fp16 = const()[name = string("op_758_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_759_cast_fp16 = add(x = variance_23_cast_fp16, y = var_758_to_fp16)[name = string("op_759_cast_fp16")]; + fp32 var_760_epsilon_0 = const()[name = string("op_760_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_760_cast_fp16 = rsqrt(epsilon = var_760_epsilon_0, x = var_759_cast_fp16)[name = string("op_760_cast_fp16")]; + tensor var_761_cast_fp16 = mul(x = x_21_cast_fp16, y = var_760_cast_fp16)[name = string("op_761_cast_fp16")]; + tensor layers_2_post_attention_layernorm_weight_to_fp16 = const()[name = string("layers_2_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37787392)))]; + tensor input_25_cast_fp16 = mul(x = var_761_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_25_cast_fp16")]; + string input_27_pad_type_0 = const()[name = string("input_27_pad_type_0"), val = string("valid")]; + tensor input_27_strides_0 = const()[name = string("input_27_strides_0"), val = tensor([1, 1])]; + tensor input_27_pad_0 = const()[name = string("input_27_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_27_dilations_0 = const()[name = string("input_27_dilations_0"), val = tensor([1, 1])]; + int32 input_27_groups_0 = const()[name = string("input_27_groups_0"), val = int32(1)]; + tensor layers_2_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37789504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40935296))))[name = string("layers_2_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_27_cast_fp16 = conv(dilations = input_27_dilations_0, groups = input_27_groups_0, pad = input_27_pad_0, pad_type = input_27_pad_type_0, strides = input_27_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_25_cast_fp16)[name = string("input_27_cast_fp16")]; + tensor var_769_cast_fp16 = silu(x = input_27_cast_fp16)[name = string("op_769_cast_fp16")]; + string var_775_pad_type_0 = const()[name = string("op_775_pad_type_0"), val = string("valid")]; + tensor var_775_strides_0 = const()[name = string("op_775_strides_0"), val = tensor([1, 1])]; + tensor var_775_pad_0 = const()[name = string("op_775_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_775_dilations_0 = const()[name = string("op_775_dilations_0"), val = tensor([1, 1])]; + int32 var_775_groups_0 = const()[name = string("op_775_groups_0"), val = int32(1)]; + tensor layers_2_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40935872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44081664))))[name = string("layers_2_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_775_cast_fp16 = conv(dilations = var_775_dilations_0, groups = var_775_groups_0, pad = var_775_pad_0, pad_type = var_775_pad_type_0, strides = var_775_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_25_cast_fp16)[name = string("op_775_cast_fp16")]; + tensor input_29_cast_fp16 = mul(x = var_769_cast_fp16, y = var_775_cast_fp16)[name = string("input_29_cast_fp16")]; + string h_5_pad_type_0 = const()[name = string("h_5_pad_type_0"), val = string("valid")]; + tensor h_5_strides_0 = const()[name = string("h_5_strides_0"), val = tensor([1, 1])]; + tensor h_5_pad_0 = const()[name = string("h_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_5_dilations_0 = const()[name = string("h_5_dilations_0"), val = tensor([1, 1])]; + int32 h_5_groups_0 = const()[name = string("h_5_groups_0"), val = int32(1)]; + tensor layers_2_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44082240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47228032))))[name = string("layers_2_mlp_down_proj_weight_to_fp16_palettized")]; + tensor h_5_cast_fp16 = conv(dilations = h_5_dilations_0, groups = h_5_groups_0, pad = h_5_pad_0, pad_type = h_5_pad_type_0, strides = h_5_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_29_cast_fp16)[name = string("h_5_cast_fp16")]; + tensor x_23_cast_fp16 = add(x = x_21_cast_fp16, y = h_5_cast_fp16)[name = string("x_23_cast_fp16")]; + int32 var_828 = const()[name = string("op_828"), val = int32(2)]; + int32 var_832 = const()[name = string("op_832"), val = int32(3)]; + tensor var_847_cast_fp16 = mul(x = x_23_cast_fp16, y = x_23_cast_fp16)[name = string("op_847_cast_fp16")]; + tensor variance_25_axes_0 = const()[name = string("variance_25_axes_0"), val = tensor([1])]; + bool variance_25_keep_dims_0 = const()[name = string("variance_25_keep_dims_0"), val = bool(true)]; + tensor variance_25_cast_fp16 = reduce_mean(axes = variance_25_axes_0, keep_dims = variance_25_keep_dims_0, x = var_847_cast_fp16)[name = string("variance_25_cast_fp16")]; + fp16 var_850_to_fp16 = const()[name = string("op_850_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_851_cast_fp16 = add(x = variance_25_cast_fp16, y = var_850_to_fp16)[name = string("op_851_cast_fp16")]; + fp32 var_852_epsilon_0 = const()[name = string("op_852_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_852_cast_fp16 = rsqrt(epsilon = var_852_epsilon_0, x = var_851_cast_fp16)[name = string("op_852_cast_fp16")]; + tensor var_853_cast_fp16 = mul(x = x_23_cast_fp16, y = var_852_cast_fp16)[name = string("op_853_cast_fp16")]; + tensor layers_3_input_layernorm_weight_to_fp16 = const()[name = string("layers_3_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47228608)))]; + tensor input_31_cast_fp16 = mul(x = var_853_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_31_cast_fp16")]; + string q_19_pad_type_0 = const()[name = string("q_19_pad_type_0"), val = string("valid")]; + tensor q_19_strides_0 = const()[name = string("q_19_strides_0"), val = tensor([1, 1])]; + tensor q_19_pad_0 = const()[name = string("q_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_19_dilations_0 = const()[name = string("q_19_dilations_0"), val = tensor([1, 1])]; + int32 q_19_groups_0 = const()[name = string("q_19_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47230720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49327936))))[name = string("layers_3_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor q_19_cast_fp16 = conv(dilations = q_19_dilations_0, groups = q_19_groups_0, pad = q_19_pad_0, pad_type = q_19_pad_type_0, strides = q_19_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_31_cast_fp16)[name = string("q_19_cast_fp16")]; + string k_19_pad_type_0 = const()[name = string("k_19_pad_type_0"), val = string("valid")]; + tensor k_19_strides_0 = const()[name = string("k_19_strides_0"), val = tensor([1, 1])]; + tensor k_19_pad_0 = const()[name = string("k_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_19_dilations_0 = const()[name = string("k_19_dilations_0"), val = tensor([1, 1])]; + int32 k_19_groups_0 = const()[name = string("k_19_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49328512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50377152))))[name = string("layers_3_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor k_19_cast_fp16 = conv(dilations = k_19_dilations_0, groups = k_19_groups_0, pad = k_19_pad_0, pad_type = k_19_pad_type_0, strides = k_19_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_31_cast_fp16)[name = string("k_19_cast_fp16")]; + string v_7_pad_type_0 = const()[name = string("v_7_pad_type_0"), val = string("valid")]; + tensor v_7_strides_0 = const()[name = string("v_7_strides_0"), val = tensor([1, 1])]; + tensor v_7_pad_0 = const()[name = string("v_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_7_dilations_0 = const()[name = string("v_7_dilations_0"), val = tensor([1, 1])]; + int32 v_7_groups_0 = const()[name = string("v_7_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50377728))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51426368))))[name = string("layers_3_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor v_7_cast_fp16 = conv(dilations = v_7_dilations_0, groups = v_7_groups_0, pad = v_7_pad_0, pad_type = v_7_pad_type_0, strides = v_7_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_31_cast_fp16)[name = string("v_7_cast_fp16")]; + tensor var_887 = const()[name = string("op_887"), val = tensor([16, 128, 1, 1])]; + tensor x_25_cast_fp16 = reshape(shape = var_887, x = q_19_cast_fp16)[name = string("x_25_cast_fp16")]; + tensor var_890_cast_fp16 = mul(x = x_25_cast_fp16, y = x_25_cast_fp16)[name = string("op_890_cast_fp16")]; + tensor variance_27_axes_0 = const()[name = string("variance_27_axes_0"), val = tensor([1])]; + bool variance_27_keep_dims_0 = const()[name = string("variance_27_keep_dims_0"), val = bool(true)]; + tensor variance_27_cast_fp16 = reduce_mean(axes = variance_27_axes_0, keep_dims = variance_27_keep_dims_0, x = var_890_cast_fp16)[name = string("variance_27_cast_fp16")]; + fp16 var_893_to_fp16 = const()[name = string("op_893_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_894_cast_fp16 = add(x = variance_27_cast_fp16, y = var_893_to_fp16)[name = string("op_894_cast_fp16")]; + fp32 var_895_epsilon_0 = const()[name = string("op_895_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_895_cast_fp16 = rsqrt(epsilon = var_895_epsilon_0, x = var_894_cast_fp16)[name = string("op_895_cast_fp16")]; + tensor var_896_cast_fp16 = mul(x = x_25_cast_fp16, y = var_895_cast_fp16)[name = string("op_896_cast_fp16")]; + tensor layers_3_self_attn_q_norm_weight_to_fp16 = const()[name = string("layers_3_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51426944)))]; + tensor q_21_cast_fp16 = mul(x = var_896_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_21_cast_fp16")]; + tensor var_898 = const()[name = string("op_898"), val = tensor([8, 128, 1, 1])]; + tensor x_27_cast_fp16 = reshape(shape = var_898, x = k_19_cast_fp16)[name = string("x_27_cast_fp16")]; + tensor var_901_cast_fp16 = mul(x = x_27_cast_fp16, y = x_27_cast_fp16)[name = string("op_901_cast_fp16")]; + tensor variance_29_axes_0 = const()[name = string("variance_29_axes_0"), val = tensor([1])]; + bool variance_29_keep_dims_0 = const()[name = string("variance_29_keep_dims_0"), val = bool(true)]; + tensor variance_29_cast_fp16 = reduce_mean(axes = variance_29_axes_0, keep_dims = variance_29_keep_dims_0, x = var_901_cast_fp16)[name = string("variance_29_cast_fp16")]; + fp16 var_904_to_fp16 = const()[name = string("op_904_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_905_cast_fp16 = add(x = variance_29_cast_fp16, y = var_904_to_fp16)[name = string("op_905_cast_fp16")]; + fp32 var_906_epsilon_0 = const()[name = string("op_906_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_906_cast_fp16 = rsqrt(epsilon = var_906_epsilon_0, x = var_905_cast_fp16)[name = string("op_906_cast_fp16")]; + tensor var_907_cast_fp16 = mul(x = x_27_cast_fp16, y = var_906_cast_fp16)[name = string("op_907_cast_fp16")]; + tensor layers_3_self_attn_k_norm_weight_to_fp16 = const()[name = string("layers_3_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51427264)))]; + tensor k_21_cast_fp16 = mul(x = var_907_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_21_cast_fp16")]; + tensor var_909 = const()[name = string("op_909"), val = tensor([1, 16, 128, 1])]; + tensor z_13_cast_fp16 = reshape(shape = var_909, x = q_21_cast_fp16)[name = string("z_13_cast_fp16")]; + tensor var_911 = const()[name = string("op_911"), val = tensor([1, 8, 128, 1])]; + tensor z_15_cast_fp16 = reshape(shape = var_911, x = k_21_cast_fp16)[name = string("z_15_cast_fp16")]; + tensor z1_13_begin_0 = const()[name = string("z1_13_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_13_end_0 = const()[name = string("z1_13_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_13_end_mask_0 = const()[name = string("z1_13_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_13_cast_fp16 = slice_by_index(begin = z1_13_begin_0, end = z1_13_end_0, end_mask = z1_13_end_mask_0, x = z_13_cast_fp16)[name = string("z1_13_cast_fp16")]; + tensor z2_13_begin_0 = const()[name = string("z2_13_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_13_end_0 = const()[name = string("z2_13_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_13_end_mask_0 = const()[name = string("z2_13_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_13_cast_fp16 = slice_by_index(begin = z2_13_begin_0, end = z2_13_end_0, end_mask = z2_13_end_mask_0, x = z_13_cast_fp16)[name = string("z2_13_cast_fp16")]; + fp16 const_7_promoted_to_fp16 = const()[name = string("const_7_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_920_cast_fp16 = mul(x = z2_13_cast_fp16, y = const_7_promoted_to_fp16)[name = string("op_920_cast_fp16")]; + bool var_922_interleave_0 = const()[name = string("op_922_interleave_0"), val = bool(false)]; + tensor var_922_cast_fp16 = concat(axis = var_828, interleave = var_922_interleave_0, values = (var_920_cast_fp16, z1_13_cast_fp16))[name = string("op_922_cast_fp16")]; + tensor var_923_cast_fp16 = mul(x = var_922_cast_fp16, y = sin_1_to_fp16)[name = string("op_923_cast_fp16")]; + tensor q_23_cast_fp16 = add(x = z_13_cast_fp16, y = var_923_cast_fp16)[name = string("q_23_cast_fp16")]; + tensor z1_15_begin_0 = const()[name = string("z1_15_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_15_end_0 = const()[name = string("z1_15_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_15_end_mask_0 = const()[name = string("z1_15_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_15_cast_fp16 = slice_by_index(begin = z1_15_begin_0, end = z1_15_end_0, end_mask = z1_15_end_mask_0, x = z_15_cast_fp16)[name = string("z1_15_cast_fp16")]; + tensor z2_15_begin_0 = const()[name = string("z2_15_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_15_end_0 = const()[name = string("z2_15_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_15_end_mask_0 = const()[name = string("z2_15_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_15_cast_fp16 = slice_by_index(begin = z2_15_begin_0, end = z2_15_end_0, end_mask = z2_15_end_mask_0, x = z_15_cast_fp16)[name = string("z2_15_cast_fp16")]; + fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_932_cast_fp16 = mul(x = z2_15_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_932_cast_fp16")]; + bool var_934_interleave_0 = const()[name = string("op_934_interleave_0"), val = bool(false)]; + tensor var_934_cast_fp16 = concat(axis = var_828, interleave = var_934_interleave_0, values = (var_932_cast_fp16, z1_15_cast_fp16))[name = string("op_934_cast_fp16")]; + tensor var_935_cast_fp16 = mul(x = var_934_cast_fp16, y = sin_1_to_fp16)[name = string("op_935_cast_fp16")]; + tensor k_23_cast_fp16 = add(x = z_15_cast_fp16, y = var_935_cast_fp16)[name = string("k_23_cast_fp16")]; + tensor var_937 = const()[name = string("op_937"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_7_cast_fp16 = reshape(shape = var_937, x = k_23_cast_fp16)[name = string("cur_key_7_cast_fp16")]; + tensor upd_7_to_fp16 = const()[name = string("upd_7_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110173888)))]; + tensor var_941_cast_fp16 = mul(x = cur_key_7_cast_fp16, y = upd_7_to_fp16)[name = string("op_941_cast_fp16")]; + tensor var_945_cast_fp16 = mul(x = v_7_cast_fp16, y = upd_7_to_fp16)[name = string("op_945_cast_fp16")]; + tensor var_947 = const()[name = string("op_947"), val = tensor([1, 8, 128, 16])]; + tensor kh_13_cast_fp16 = reshape(shape = var_947, x = var_941_cast_fp16)[name = string("kh_13_cast_fp16")]; + tensor var_949 = const()[name = string("op_949"), val = tensor([1, 8, 128, 16])]; + tensor vh_13_cast_fp16 = reshape(shape = var_949, x = var_945_cast_fp16)[name = string("vh_13_cast_fp16")]; + tensor transpose_12_perm_0 = const()[name = string("transpose_12_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_6_reps_0 = const()[name = string("tile_6_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_12_cast_fp16 = transpose(perm = transpose_12_perm_0, x = kh_13_cast_fp16)[name = string("transpose_455")]; + tensor tile_6_cast_fp16 = tile(reps = tile_6_reps_0, x = transpose_12_cast_fp16)[name = string("tile_6_cast_fp16")]; + tensor concat_16 = const()[name = string("concat_16"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_12_cast_fp16 = reshape(shape = concat_16, x = tile_6_cast_fp16)[name = string("reshape_12_cast_fp16")]; + tensor transpose_13_perm_0 = const()[name = string("transpose_13_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_17 = const()[name = string("concat_17"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_13_cast_fp16 = transpose(perm = transpose_13_perm_0, x = reshape_12_cast_fp16)[name = string("transpose_454")]; + tensor reshape_13_cast_fp16 = reshape(shape = concat_17, x = transpose_13_cast_fp16)[name = string("reshape_13_cast_fp16")]; + tensor transpose_14_perm_0 = const()[name = string("transpose_14_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_7_reps_0 = const()[name = string("tile_7_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_14_cast_fp16 = transpose(perm = transpose_14_perm_0, x = vh_13_cast_fp16)[name = string("transpose_453")]; + tensor tile_7_cast_fp16 = tile(reps = tile_7_reps_0, x = transpose_14_cast_fp16)[name = string("tile_7_cast_fp16")]; + tensor concat_18 = const()[name = string("concat_18"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_14_cast_fp16 = reshape(shape = concat_18, x = tile_7_cast_fp16)[name = string("reshape_14_cast_fp16")]; + tensor transpose_15_perm_0 = const()[name = string("transpose_15_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_19 = const()[name = string("concat_19"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_15_cast_fp16 = transpose(perm = transpose_15_perm_0, x = reshape_14_cast_fp16)[name = string("transpose_452")]; + tensor reshape_15_cast_fp16 = reshape(shape = concat_19, x = transpose_15_cast_fp16)[name = string("reshape_15_cast_fp16")]; + fp16 var_953_to_fp16 = const()[name = string("op_953_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_954_cast_fp16 = mul(x = q_23_cast_fp16, y = var_953_to_fp16)[name = string("op_954_cast_fp16")]; + tensor transpose_333_perm_0 = const()[name = string("transpose_333_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_13_transpose_x_1 = const()[name = string("w_13_transpose_x_1"), val = bool(true)]; + bool w_13_transpose_y_1 = const()[name = string("w_13_transpose_y_1"), val = bool(false)]; + tensor transpose_333_cast_fp16 = transpose(perm = transpose_333_perm_0, x = reshape_13_cast_fp16)[name = string("transpose_451")]; + tensor w_13_cast_fp16 = matmul(transpose_x = w_13_transpose_x_1, transpose_y = w_13_transpose_y_1, x = var_954_cast_fp16, y = transpose_333_cast_fp16)[name = string("w_13_cast_fp16")]; + tensor pad_7_to_fp16 = const()[name = string("pad_7_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174016)))]; + tensor var_957_cast_fp16 = add(x = w_13_cast_fp16, y = pad_7_to_fp16)[name = string("op_957_cast_fp16")]; + tensor w_15_cast_fp16 = softmax(axis = var_832, x = var_957_cast_fp16)[name = string("w_15_cast_fp16")]; + tensor transpose_334_perm_0 = const()[name = string("transpose_334_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_7_transpose_x_1 = const()[name = string("attn_7_transpose_x_1"), val = bool(false)]; + bool attn_7_transpose_y_1 = const()[name = string("attn_7_transpose_y_1"), val = bool(true)]; + tensor transpose_334_cast_fp16 = transpose(perm = transpose_334_perm_0, x = reshape_15_cast_fp16)[name = string("transpose_450")]; + tensor attn_7_cast_fp16 = matmul(transpose_x = attn_7_transpose_x_1, transpose_y = attn_7_transpose_y_1, x = transpose_334_cast_fp16, y = w_15_cast_fp16)[name = string("attn_7_cast_fp16")]; + tensor var_961 = const()[name = string("op_961"), val = tensor([1, 2048, 1, 1])]; + tensor input_33_cast_fp16 = reshape(shape = var_961, x = attn_7_cast_fp16)[name = string("input_33_cast_fp16")]; + string attn_output_7_pad_type_0 = const()[name = string("attn_output_7_pad_type_0"), val = string("valid")]; + tensor attn_output_7_strides_0 = const()[name = string("attn_output_7_strides_0"), val = tensor([1, 1])]; + tensor attn_output_7_pad_0 = const()[name = string("attn_output_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_7_dilations_0 = const()[name = string("attn_output_7_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_7_groups_0 = const()[name = string("attn_output_7_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51427584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53524800))))[name = string("layers_3_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor attn_output_7_cast_fp16 = conv(dilations = attn_output_7_dilations_0, groups = attn_output_7_groups_0, pad = attn_output_7_pad_0, pad_type = attn_output_7_pad_type_0, strides = attn_output_7_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_33_cast_fp16)[name = string("attn_output_7_cast_fp16")]; + tensor x_29_cast_fp16 = add(x = x_23_cast_fp16, y = attn_output_7_cast_fp16)[name = string("x_29_cast_fp16")]; + tensor var_975_cast_fp16 = mul(x = x_29_cast_fp16, y = x_29_cast_fp16)[name = string("op_975_cast_fp16")]; + tensor variance_31_axes_0 = const()[name = string("variance_31_axes_0"), val = tensor([1])]; + bool variance_31_keep_dims_0 = const()[name = string("variance_31_keep_dims_0"), val = bool(true)]; + tensor variance_31_cast_fp16 = reduce_mean(axes = variance_31_axes_0, keep_dims = variance_31_keep_dims_0, x = var_975_cast_fp16)[name = string("variance_31_cast_fp16")]; + fp16 var_978_to_fp16 = const()[name = string("op_978_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_979_cast_fp16 = add(x = variance_31_cast_fp16, y = var_978_to_fp16)[name = string("op_979_cast_fp16")]; + fp32 var_980_epsilon_0 = const()[name = string("op_980_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_980_cast_fp16 = rsqrt(epsilon = var_980_epsilon_0, x = var_979_cast_fp16)[name = string("op_980_cast_fp16")]; + tensor var_981_cast_fp16 = mul(x = x_29_cast_fp16, y = var_980_cast_fp16)[name = string("op_981_cast_fp16")]; + tensor layers_3_post_attention_layernorm_weight_to_fp16 = const()[name = string("layers_3_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53525376)))]; + tensor input_35_cast_fp16 = mul(x = var_981_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_35_cast_fp16")]; + string input_37_pad_type_0 = const()[name = string("input_37_pad_type_0"), val = string("valid")]; + tensor input_37_strides_0 = const()[name = string("input_37_strides_0"), val = tensor([1, 1])]; + tensor input_37_pad_0 = const()[name = string("input_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_37_dilations_0 = const()[name = string("input_37_dilations_0"), val = tensor([1, 1])]; + int32 input_37_groups_0 = const()[name = string("input_37_groups_0"), val = int32(1)]; + tensor layers_3_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53527488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(56673280))))[name = string("layers_3_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_37_cast_fp16 = conv(dilations = input_37_dilations_0, groups = input_37_groups_0, pad = input_37_pad_0, pad_type = input_37_pad_type_0, strides = input_37_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_35_cast_fp16)[name = string("input_37_cast_fp16")]; + tensor var_989_cast_fp16 = silu(x = input_37_cast_fp16)[name = string("op_989_cast_fp16")]; + string var_995_pad_type_0 = const()[name = string("op_995_pad_type_0"), val = string("valid")]; + tensor var_995_strides_0 = const()[name = string("op_995_strides_0"), val = tensor([1, 1])]; + tensor var_995_pad_0 = const()[name = string("op_995_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_995_dilations_0 = const()[name = string("op_995_dilations_0"), val = tensor([1, 1])]; + int32 var_995_groups_0 = const()[name = string("op_995_groups_0"), val = int32(1)]; + tensor layers_3_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(56673856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59819648))))[name = string("layers_3_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_995_cast_fp16 = conv(dilations = var_995_dilations_0, groups = var_995_groups_0, pad = var_995_pad_0, pad_type = var_995_pad_type_0, strides = var_995_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_35_cast_fp16)[name = string("op_995_cast_fp16")]; + tensor input_39_cast_fp16 = mul(x = var_989_cast_fp16, y = var_995_cast_fp16)[name = string("input_39_cast_fp16")]; + string h_7_pad_type_0 = const()[name = string("h_7_pad_type_0"), val = string("valid")]; + tensor h_7_strides_0 = const()[name = string("h_7_strides_0"), val = tensor([1, 1])]; + tensor h_7_pad_0 = const()[name = string("h_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_7_dilations_0 = const()[name = string("h_7_dilations_0"), val = tensor([1, 1])]; + int32 h_7_groups_0 = const()[name = string("h_7_groups_0"), val = int32(1)]; + tensor layers_3_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59820224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62966016))))[name = string("layers_3_mlp_down_proj_weight_to_fp16_palettized")]; + tensor h_7_cast_fp16 = conv(dilations = h_7_dilations_0, groups = h_7_groups_0, pad = h_7_pad_0, pad_type = h_7_pad_type_0, strides = h_7_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_39_cast_fp16)[name = string("h_7_cast_fp16")]; + tensor x_31_cast_fp16 = add(x = x_29_cast_fp16, y = h_7_cast_fp16)[name = string("x_31_cast_fp16")]; + int32 var_1048 = const()[name = string("op_1048"), val = int32(2)]; + tensor var_1060_cast_fp16 = mul(x = x_31_cast_fp16, y = x_31_cast_fp16)[name = string("op_1060_cast_fp16")]; + tensor variance_33_axes_0 = const()[name = string("variance_33_axes_0"), val = tensor([1])]; + bool variance_33_keep_dims_0 = const()[name = string("variance_33_keep_dims_0"), val = bool(true)]; + tensor variance_33_cast_fp16 = reduce_mean(axes = variance_33_axes_0, keep_dims = variance_33_keep_dims_0, x = var_1060_cast_fp16)[name = string("variance_33_cast_fp16")]; + fp16 var_1063_to_fp16 = const()[name = string("op_1063_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1064_cast_fp16 = add(x = variance_33_cast_fp16, y = var_1063_to_fp16)[name = string("op_1064_cast_fp16")]; + fp32 var_1065_epsilon_0 = const()[name = string("op_1065_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1065_cast_fp16 = rsqrt(epsilon = var_1065_epsilon_0, x = var_1064_cast_fp16)[name = string("op_1065_cast_fp16")]; + tensor var_1066_cast_fp16 = mul(x = x_31_cast_fp16, y = var_1065_cast_fp16)[name = string("op_1066_cast_fp16")]; + tensor layers_4_input_layernorm_weight_to_fp16 = const()[name = string("layers_4_input_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62966592)))]; + tensor input_41_cast_fp16 = mul(x = var_1066_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_41_cast_fp16")]; + string k_25_pad_type_0 = const()[name = string("k_25_pad_type_0"), val = string("valid")]; + tensor k_25_strides_0 = const()[name = string("k_25_strides_0"), val = tensor([1, 1])]; + tensor k_25_pad_0 = const()[name = string("k_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_25_dilations_0 = const()[name = string("k_25_dilations_0"), val = tensor([1, 1])]; + int32 k_25_groups_0 = const()[name = string("k_25_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65066496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(66115136))))[name = string("layers_4_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor k_25_cast_fp16 = conv(dilations = k_25_dilations_0, groups = k_25_groups_0, pad = k_25_pad_0, pad_type = k_25_pad_type_0, strides = k_25_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_41_cast_fp16)[name = string("k_25_cast_fp16")]; + string v_9_pad_type_0 = const()[name = string("v_9_pad_type_0"), val = string("valid")]; + tensor v_9_strides_0 = const()[name = string("v_9_strides_0"), val = tensor([1, 1])]; + tensor v_9_pad_0 = const()[name = string("v_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_9_dilations_0 = const()[name = string("v_9_dilations_0"), val = tensor([1, 1])]; + int32 v_9_groups_0 = const()[name = string("v_9_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(66115712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67164352))))[name = string("layers_4_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor v_9_cast_fp16 = conv(dilations = v_9_dilations_0, groups = v_9_groups_0, pad = v_9_pad_0, pad_type = v_9_pad_type_0, strides = v_9_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_41_cast_fp16)[name = string("v_9_cast_fp16")]; + tensor var_1111 = const()[name = string("op_1111"), val = tensor([8, 128, 1, 1])]; + tensor x_35_cast_fp16 = reshape(shape = var_1111, x = k_25_cast_fp16)[name = string("x_35_cast_fp16")]; + tensor var_1114_cast_fp16 = mul(x = x_35_cast_fp16, y = x_35_cast_fp16)[name = string("op_1114_cast_fp16")]; + tensor variance_37_axes_0 = const()[name = string("variance_37_axes_0"), val = tensor([1])]; + bool variance_37_keep_dims_0 = const()[name = string("variance_37_keep_dims_0"), val = bool(true)]; + tensor variance_37_cast_fp16 = reduce_mean(axes = variance_37_axes_0, keep_dims = variance_37_keep_dims_0, x = var_1114_cast_fp16)[name = string("variance_37_cast_fp16")]; + fp16 var_1117_to_fp16 = const()[name = string("op_1117_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1118_cast_fp16 = add(x = variance_37_cast_fp16, y = var_1117_to_fp16)[name = string("op_1118_cast_fp16")]; + fp32 var_1119_epsilon_0 = const()[name = string("op_1119_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1119_cast_fp16 = rsqrt(epsilon = var_1119_epsilon_0, x = var_1118_cast_fp16)[name = string("op_1119_cast_fp16")]; + tensor var_1120_cast_fp16 = mul(x = x_35_cast_fp16, y = var_1119_cast_fp16)[name = string("op_1120_cast_fp16")]; + tensor layers_4_self_attn_k_norm_weight_to_fp16 = const()[name = string("layers_4_self_attn_k_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67165248)))]; + tensor k_27_cast_fp16 = mul(x = var_1120_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_27_cast_fp16")]; + tensor var_1124 = const()[name = string("op_1124"), val = tensor([1, 8, 128, 1])]; + tensor z_19_cast_fp16 = reshape(shape = var_1124, x = k_27_cast_fp16)[name = string("z_19_cast_fp16")]; + tensor z1_19_begin_0 = const()[name = string("z1_19_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_19_end_0 = const()[name = string("z1_19_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_19_end_mask_0 = const()[name = string("z1_19_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_19_cast_fp16 = slice_by_index(begin = z1_19_begin_0, end = z1_19_end_0, end_mask = z1_19_end_mask_0, x = z_19_cast_fp16)[name = string("z1_19_cast_fp16")]; + tensor z2_19_begin_0 = const()[name = string("z2_19_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_19_end_0 = const()[name = string("z2_19_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_19_end_mask_0 = const()[name = string("z2_19_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_19_cast_fp16 = slice_by_index(begin = z2_19_begin_0, end = z2_19_end_0, end_mask = z2_19_end_mask_0, x = z_19_cast_fp16)[name = string("z2_19_cast_fp16")]; + fp16 const_10_promoted_to_fp16 = const()[name = string("const_10_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1145_cast_fp16 = mul(x = z2_19_cast_fp16, y = const_10_promoted_to_fp16)[name = string("op_1145_cast_fp16")]; + bool var_1147_interleave_0 = const()[name = string("op_1147_interleave_0"), val = bool(false)]; + tensor var_1147_cast_fp16 = concat(axis = var_1048, interleave = var_1147_interleave_0, values = (var_1145_cast_fp16, z1_19_cast_fp16))[name = string("op_1147_cast_fp16")]; + tensor var_1148_cast_fp16 = mul(x = var_1147_cast_fp16, y = sin_1_to_fp16)[name = string("op_1148_cast_fp16")]; + tensor k_29_cast_fp16 = add(x = z_19_cast_fp16, y = var_1148_cast_fp16)[name = string("k_29_cast_fp16")]; + tensor var_1150 = const()[name = string("op_1150"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_9_cast_fp16 = reshape(shape = var_1150, x = k_29_cast_fp16)[name = string("cur_key_9_cast_fp16")]; + tensor var_136_to_fp16 = const()[name = string("op_136_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110173888)))]; + tensor var_1193_cast_fp16 = mul(x = cur_key_9_cast_fp16, y = var_136_to_fp16)[name = string("op_1193_cast_fp16")]; + tensor var_1200_cast_fp16 = mul(x = v_9_cast_fp16, y = var_136_to_fp16)[name = string("op_1200_cast_fp16")]; + int32 var_1204 = const()[name = string("op_1204"), val = int32(1)]; + bool layer_key_caches_3_interleave_0 = const()[name = string("layer_key_caches_3_interleave_0"), val = bool(false)]; + tensor layer_key_caches_3_cast_fp16 = concat(axis = var_1204, interleave = layer_key_caches_3_interleave_0, values = (var_281_cast_fp16, var_501_cast_fp16, var_721_cast_fp16, var_941_cast_fp16, var_1193_cast_fp16))[name = string("layer_key_caches_3_cast_fp16")]; + int32 var_1207 = const()[name = string("op_1207"), val = int32(1)]; + bool layer_value_caches_3_interleave_0 = const()[name = string("layer_value_caches_3_interleave_0"), val = bool(false)]; + tensor layer_value_caches_3_cast_fp16 = concat(axis = var_1207, interleave = layer_value_caches_3_interleave_0, values = (var_285_cast_fp16, var_505_cast_fp16, var_725_cast_fp16, var_945_cast_fp16, var_1200_cast_fp16))[name = string("layer_value_caches_3_cast_fp16")]; + tensor key_cache_11_begin_0 = const()[name = string("key_cache_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_11_end_0 = const()[name = string("key_cache_11_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_11_end_mask_0 = const()[name = string("key_cache_11_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_11_cast_fp16 = slice_by_index(begin = key_cache_11_begin_0, end = key_cache_11_end_0, end_mask = key_cache_11_end_mask_0, x = layer_key_caches_3_cast_fp16)[name = string("key_cache_11_cast_fp16")]; + tensor value_cache_11_begin_0 = const()[name = string("value_cache_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_11_end_0 = const()[name = string("value_cache_11_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_11_end_mask_0 = const()[name = string("value_cache_11_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_11_cast_fp16 = slice_by_index(begin = value_cache_11_begin_0, end = value_cache_11_end_0, end_mask = value_cache_11_end_mask_0, x = layer_value_caches_3_cast_fp16)[name = string("value_cache_11_cast_fp16")]; + int32 var_1301 = const()[name = string("op_1301"), val = int32(2)]; + int32 var_1305 = const()[name = string("op_1305"), val = int32(3)]; + tensor var_1320_cast_fp16 = mul(x = code0_embed, y = code0_embed)[name = string("op_1320_cast_fp16")]; + tensor variance_41_axes_0 = const()[name = string("variance_41_axes_0"), val = tensor([1])]; + bool variance_41_keep_dims_0 = const()[name = string("variance_41_keep_dims_0"), val = bool(true)]; + tensor variance_41_cast_fp16 = reduce_mean(axes = variance_41_axes_0, keep_dims = variance_41_keep_dims_0, x = var_1320_cast_fp16)[name = string("variance_41_cast_fp16")]; + fp16 var_1323_to_fp16 = const()[name = string("op_1323_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1324_cast_fp16 = add(x = variance_41_cast_fp16, y = var_1323_to_fp16)[name = string("op_1324_cast_fp16")]; + fp32 var_1325_epsilon_0 = const()[name = string("op_1325_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1325_cast_fp16 = rsqrt(epsilon = var_1325_epsilon_0, x = var_1324_cast_fp16)[name = string("op_1325_cast_fp16")]; + tensor var_1326_cast_fp16 = mul(x = code0_embed, y = var_1325_cast_fp16)[name = string("op_1326_cast_fp16")]; + tensor input_51_cast_fp16 = mul(x = var_1326_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_51_cast_fp16")]; + string q_31_pad_type_0 = const()[name = string("q_31_pad_type_0"), val = string("valid")]; + tensor q_31_strides_0 = const()[name = string("q_31_strides_0"), val = tensor([1, 1])]; + tensor q_31_pad_0 = const()[name = string("q_31_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_31_dilations_0 = const()[name = string("q_31_dilations_0"), val = tensor([1, 1])]; + int32 q_31_groups_0 = const()[name = string("q_31_groups_0"), val = int32(1)]; + tensor q_31_cast_fp16 = conv(dilations = q_31_dilations_0, groups = q_31_groups_0, pad = q_31_pad_0, pad_type = q_31_pad_type_0, strides = q_31_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_51_cast_fp16)[name = string("q_31_cast_fp16")]; + string k_31_pad_type_0 = const()[name = string("k_31_pad_type_0"), val = string("valid")]; + tensor k_31_strides_0 = const()[name = string("k_31_strides_0"), val = tensor([1, 1])]; + tensor k_31_pad_0 = const()[name = string("k_31_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_31_dilations_0 = const()[name = string("k_31_dilations_0"), val = tensor([1, 1])]; + int32 k_31_groups_0 = const()[name = string("k_31_groups_0"), val = int32(1)]; + tensor k_31_cast_fp16 = conv(dilations = k_31_dilations_0, groups = k_31_groups_0, pad = k_31_pad_0, pad_type = k_31_pad_type_0, strides = k_31_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_51_cast_fp16)[name = string("k_31_cast_fp16")]; + string v_11_pad_type_0 = const()[name = string("v_11_pad_type_0"), val = string("valid")]; + tensor v_11_strides_0 = const()[name = string("v_11_strides_0"), val = tensor([1, 1])]; + tensor v_11_pad_0 = const()[name = string("v_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_11_dilations_0 = const()[name = string("v_11_dilations_0"), val = tensor([1, 1])]; + int32 v_11_groups_0 = const()[name = string("v_11_groups_0"), val = int32(1)]; + tensor v_11_cast_fp16 = conv(dilations = v_11_dilations_0, groups = v_11_groups_0, pad = v_11_pad_0, pad_type = v_11_pad_type_0, strides = v_11_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_51_cast_fp16)[name = string("v_11_cast_fp16")]; + tensor var_1360 = const()[name = string("op_1360"), val = tensor([16, 128, 1, 1])]; + tensor x_39_cast_fp16 = reshape(shape = var_1360, x = q_31_cast_fp16)[name = string("x_39_cast_fp16")]; + tensor var_1363_cast_fp16 = mul(x = x_39_cast_fp16, y = x_39_cast_fp16)[name = string("op_1363_cast_fp16")]; + tensor variance_43_axes_0 = const()[name = string("variance_43_axes_0"), val = tensor([1])]; + bool variance_43_keep_dims_0 = const()[name = string("variance_43_keep_dims_0"), val = bool(true)]; + tensor variance_43_cast_fp16 = reduce_mean(axes = variance_43_axes_0, keep_dims = variance_43_keep_dims_0, x = var_1363_cast_fp16)[name = string("variance_43_cast_fp16")]; + fp16 var_1366_to_fp16 = const()[name = string("op_1366_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1367_cast_fp16 = add(x = variance_43_cast_fp16, y = var_1366_to_fp16)[name = string("op_1367_cast_fp16")]; + fp32 var_1368_epsilon_0 = const()[name = string("op_1368_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1368_cast_fp16 = rsqrt(epsilon = var_1368_epsilon_0, x = var_1367_cast_fp16)[name = string("op_1368_cast_fp16")]; + tensor var_1369_cast_fp16 = mul(x = x_39_cast_fp16, y = var_1368_cast_fp16)[name = string("op_1369_cast_fp16")]; + tensor q_33_cast_fp16 = mul(x = var_1369_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_33_cast_fp16")]; + tensor var_1371 = const()[name = string("op_1371"), val = tensor([8, 128, 1, 1])]; + tensor x_41_cast_fp16 = reshape(shape = var_1371, x = k_31_cast_fp16)[name = string("x_41_cast_fp16")]; + tensor var_1374_cast_fp16 = mul(x = x_41_cast_fp16, y = x_41_cast_fp16)[name = string("op_1374_cast_fp16")]; + tensor variance_45_axes_0 = const()[name = string("variance_45_axes_0"), val = tensor([1])]; + bool variance_45_keep_dims_0 = const()[name = string("variance_45_keep_dims_0"), val = bool(true)]; + tensor variance_45_cast_fp16 = reduce_mean(axes = variance_45_axes_0, keep_dims = variance_45_keep_dims_0, x = var_1374_cast_fp16)[name = string("variance_45_cast_fp16")]; + fp16 var_1377_to_fp16 = const()[name = string("op_1377_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1378_cast_fp16 = add(x = variance_45_cast_fp16, y = var_1377_to_fp16)[name = string("op_1378_cast_fp16")]; + fp32 var_1379_epsilon_0 = const()[name = string("op_1379_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1379_cast_fp16 = rsqrt(epsilon = var_1379_epsilon_0, x = var_1378_cast_fp16)[name = string("op_1379_cast_fp16")]; + tensor var_1380_cast_fp16 = mul(x = x_41_cast_fp16, y = var_1379_cast_fp16)[name = string("op_1380_cast_fp16")]; + tensor k_33_cast_fp16 = mul(x = var_1380_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_33_cast_fp16")]; + tensor var_1382 = const()[name = string("op_1382"), val = tensor([1, 16, 128, 1])]; + tensor z_21_cast_fp16 = reshape(shape = var_1382, x = q_33_cast_fp16)[name = string("z_21_cast_fp16")]; + tensor var_1384 = const()[name = string("op_1384"), val = tensor([1, 8, 128, 1])]; + tensor z_23_cast_fp16 = reshape(shape = var_1384, x = k_33_cast_fp16)[name = string("z_23_cast_fp16")]; + tensor z1_21_begin_0 = const()[name = string("z1_21_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_21_end_0 = const()[name = string("z1_21_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_21_end_mask_0 = const()[name = string("z1_21_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_21_cast_fp16 = slice_by_index(begin = z1_21_begin_0, end = z1_21_end_0, end_mask = z1_21_end_mask_0, x = z_21_cast_fp16)[name = string("z1_21_cast_fp16")]; + tensor z2_21_begin_0 = const()[name = string("z2_21_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_21_end_0 = const()[name = string("z2_21_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_21_end_mask_0 = const()[name = string("z2_21_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_21_cast_fp16 = slice_by_index(begin = z2_21_begin_0, end = z2_21_end_0, end_mask = z2_21_end_mask_0, x = z_21_cast_fp16)[name = string("z2_21_cast_fp16")]; + tensor cos_11_to_fp16 = const()[name = string("cos_11_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174144)))]; + tensor var_1392_cast_fp16 = mul(x = z_21_cast_fp16, y = cos_11_to_fp16)[name = string("op_1392_cast_fp16")]; + fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1393_cast_fp16 = mul(x = z2_21_cast_fp16, y = const_12_promoted_to_fp16)[name = string("op_1393_cast_fp16")]; + bool var_1395_interleave_0 = const()[name = string("op_1395_interleave_0"), val = bool(false)]; + tensor var_1395_cast_fp16 = concat(axis = var_1301, interleave = var_1395_interleave_0, values = (var_1393_cast_fp16, z1_21_cast_fp16))[name = string("op_1395_cast_fp16")]; + tensor sin_11_to_fp16 = const()[name = string("sin_11_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174464)))]; + tensor var_1396_cast_fp16 = mul(x = var_1395_cast_fp16, y = sin_11_to_fp16)[name = string("op_1396_cast_fp16")]; + tensor q_35_cast_fp16 = add(x = var_1392_cast_fp16, y = var_1396_cast_fp16)[name = string("q_35_cast_fp16")]; + tensor z1_23_begin_0 = const()[name = string("z1_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_23_end_0 = const()[name = string("z1_23_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_23_end_mask_0 = const()[name = string("z1_23_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_23_cast_fp16 = slice_by_index(begin = z1_23_begin_0, end = z1_23_end_0, end_mask = z1_23_end_mask_0, x = z_23_cast_fp16)[name = string("z1_23_cast_fp16")]; + tensor z2_23_begin_0 = const()[name = string("z2_23_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_23_end_0 = const()[name = string("z2_23_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_23_end_mask_0 = const()[name = string("z2_23_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_23_cast_fp16 = slice_by_index(begin = z2_23_begin_0, end = z2_23_end_0, end_mask = z2_23_end_mask_0, x = z_23_cast_fp16)[name = string("z2_23_cast_fp16")]; + tensor var_1404_cast_fp16 = mul(x = z_23_cast_fp16, y = cos_11_to_fp16)[name = string("op_1404_cast_fp16")]; + fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1405_cast_fp16 = mul(x = z2_23_cast_fp16, y = const_13_promoted_to_fp16)[name = string("op_1405_cast_fp16")]; + bool var_1407_interleave_0 = const()[name = string("op_1407_interleave_0"), val = bool(false)]; + tensor var_1407_cast_fp16 = concat(axis = var_1301, interleave = var_1407_interleave_0, values = (var_1405_cast_fp16, z1_23_cast_fp16))[name = string("op_1407_cast_fp16")]; + tensor var_1408_cast_fp16 = mul(x = var_1407_cast_fp16, y = sin_11_to_fp16)[name = string("op_1408_cast_fp16")]; + tensor k_35_cast_fp16 = add(x = var_1404_cast_fp16, y = var_1408_cast_fp16)[name = string("k_35_cast_fp16")]; + tensor var_1410 = const()[name = string("op_1410"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_11_cast_fp16 = reshape(shape = var_1410, x = k_35_cast_fp16)[name = string("cur_key_11_cast_fp16")]; + tensor var_1412_to_fp16 = const()[name = string("op_1412_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174784)))]; + tensor var_1413_cast_fp16 = mul(x = key_cache_11_cast_fp16, y = var_1412_to_fp16)[name = string("op_1413_cast_fp16")]; + tensor upd_11_to_fp16 = const()[name = string("upd_11_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174912)))]; + tensor var_1414_cast_fp16 = mul(x = cur_key_11_cast_fp16, y = upd_11_to_fp16)[name = string("op_1414_cast_fp16")]; + tensor key_11_cast_fp16 = add(x = var_1413_cast_fp16, y = var_1414_cast_fp16)[name = string("key_11_cast_fp16")]; + tensor var_1416_to_fp16 = const()[name = string("op_1416_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174784)))]; + tensor var_1417_cast_fp16 = mul(x = value_cache_11_cast_fp16, y = var_1416_to_fp16)[name = string("op_1417_cast_fp16")]; + tensor var_1418_cast_fp16 = mul(x = v_11_cast_fp16, y = upd_11_to_fp16)[name = string("op_1418_cast_fp16")]; + tensor value_11_cast_fp16 = add(x = var_1417_cast_fp16, y = var_1418_cast_fp16)[name = string("value_11_cast_fp16")]; + tensor var_1420 = const()[name = string("op_1420"), val = tensor([1, 8, 128, 16])]; + tensor kh_21_cast_fp16 = reshape(shape = var_1420, x = key_11_cast_fp16)[name = string("kh_21_cast_fp16")]; + tensor var_1422 = const()[name = string("op_1422"), val = tensor([1, 8, 128, 16])]; + tensor vh_21_cast_fp16 = reshape(shape = var_1422, x = value_11_cast_fp16)[name = string("vh_21_cast_fp16")]; + tensor transpose_20_perm_0 = const()[name = string("transpose_20_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_10_reps_0 = const()[name = string("tile_10_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_20_cast_fp16 = transpose(perm = transpose_20_perm_0, x = kh_21_cast_fp16)[name = string("transpose_449")]; + tensor tile_10_cast_fp16 = tile(reps = tile_10_reps_0, x = transpose_20_cast_fp16)[name = string("tile_10_cast_fp16")]; + tensor concat_28 = const()[name = string("concat_28"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_20_cast_fp16 = reshape(shape = concat_28, x = tile_10_cast_fp16)[name = string("reshape_20_cast_fp16")]; + tensor transpose_21_perm_0 = const()[name = string("transpose_21_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_29 = const()[name = string("concat_29"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_21_cast_fp16 = transpose(perm = transpose_21_perm_0, x = reshape_20_cast_fp16)[name = string("transpose_448")]; + tensor reshape_21_cast_fp16 = reshape(shape = concat_29, x = transpose_21_cast_fp16)[name = string("reshape_21_cast_fp16")]; + tensor transpose_22_perm_0 = const()[name = string("transpose_22_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_11_reps_0 = const()[name = string("tile_11_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_22_cast_fp16 = transpose(perm = transpose_22_perm_0, x = vh_21_cast_fp16)[name = string("transpose_447")]; + tensor tile_11_cast_fp16 = tile(reps = tile_11_reps_0, x = transpose_22_cast_fp16)[name = string("tile_11_cast_fp16")]; + tensor concat_30 = const()[name = string("concat_30"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_22_cast_fp16 = reshape(shape = concat_30, x = tile_11_cast_fp16)[name = string("reshape_22_cast_fp16")]; + tensor transpose_23_perm_0 = const()[name = string("transpose_23_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_31 = const()[name = string("concat_31"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_23_cast_fp16 = transpose(perm = transpose_23_perm_0, x = reshape_22_cast_fp16)[name = string("transpose_446")]; + tensor reshape_23_cast_fp16 = reshape(shape = concat_31, x = transpose_23_cast_fp16)[name = string("reshape_23_cast_fp16")]; + fp16 var_1426_to_fp16 = const()[name = string("op_1426_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_1427_cast_fp16 = mul(x = q_35_cast_fp16, y = var_1426_to_fp16)[name = string("op_1427_cast_fp16")]; + tensor transpose_337_perm_0 = const()[name = string("transpose_337_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_21_transpose_x_1 = const()[name = string("w_21_transpose_x_1"), val = bool(true)]; + bool w_21_transpose_y_1 = const()[name = string("w_21_transpose_y_1"), val = bool(false)]; + tensor transpose_337_cast_fp16 = transpose(perm = transpose_337_perm_0, x = reshape_21_cast_fp16)[name = string("transpose_445")]; + tensor w_21_cast_fp16 = matmul(transpose_x = w_21_transpose_x_1, transpose_y = w_21_transpose_y_1, x = var_1427_cast_fp16, y = transpose_337_cast_fp16)[name = string("w_21_cast_fp16")]; + tensor pad_11_to_fp16 = const()[name = string("pad_11_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110175040)))]; + tensor var_1430_cast_fp16 = add(x = w_21_cast_fp16, y = pad_11_to_fp16)[name = string("op_1430_cast_fp16")]; + tensor w_23_cast_fp16 = softmax(axis = var_1305, x = var_1430_cast_fp16)[name = string("w_23_cast_fp16")]; + tensor transpose_338_perm_0 = const()[name = string("transpose_338_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_11_transpose_x_1 = const()[name = string("attn_11_transpose_x_1"), val = bool(false)]; + bool attn_11_transpose_y_1 = const()[name = string("attn_11_transpose_y_1"), val = bool(true)]; + tensor transpose_338_cast_fp16 = transpose(perm = transpose_338_perm_0, x = reshape_23_cast_fp16)[name = string("transpose_444")]; + tensor attn_11_cast_fp16 = matmul(transpose_x = attn_11_transpose_x_1, transpose_y = attn_11_transpose_y_1, x = transpose_338_cast_fp16, y = w_23_cast_fp16)[name = string("attn_11_cast_fp16")]; + tensor var_1434 = const()[name = string("op_1434"), val = tensor([1, 2048, 1, 1])]; + tensor input_53_cast_fp16 = reshape(shape = var_1434, x = attn_11_cast_fp16)[name = string("input_53_cast_fp16")]; + string attn_output_11_pad_type_0 = const()[name = string("attn_output_11_pad_type_0"), val = string("valid")]; + tensor attn_output_11_strides_0 = const()[name = string("attn_output_11_strides_0"), val = tensor([1, 1])]; + tensor attn_output_11_pad_0 = const()[name = string("attn_output_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_11_dilations_0 = const()[name = string("attn_output_11_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_11_groups_0 = const()[name = string("attn_output_11_groups_0"), val = int32(1)]; + tensor attn_output_11_cast_fp16 = conv(dilations = attn_output_11_dilations_0, groups = attn_output_11_groups_0, pad = attn_output_11_pad_0, pad_type = attn_output_11_pad_type_0, strides = attn_output_11_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_53_cast_fp16)[name = string("attn_output_11_cast_fp16")]; + tensor x_43_cast_fp16 = add(x = code0_embed, y = attn_output_11_cast_fp16)[name = string("x_43_cast_fp16")]; + tensor var_1448_cast_fp16 = mul(x = x_43_cast_fp16, y = x_43_cast_fp16)[name = string("op_1448_cast_fp16")]; + tensor variance_47_axes_0 = const()[name = string("variance_47_axes_0"), val = tensor([1])]; + bool variance_47_keep_dims_0 = const()[name = string("variance_47_keep_dims_0"), val = bool(true)]; + tensor variance_47_cast_fp16 = reduce_mean(axes = variance_47_axes_0, keep_dims = variance_47_keep_dims_0, x = var_1448_cast_fp16)[name = string("variance_47_cast_fp16")]; + fp16 var_1451_to_fp16 = const()[name = string("op_1451_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1452_cast_fp16 = add(x = variance_47_cast_fp16, y = var_1451_to_fp16)[name = string("op_1452_cast_fp16")]; + fp32 var_1453_epsilon_0 = const()[name = string("op_1453_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1453_cast_fp16 = rsqrt(epsilon = var_1453_epsilon_0, x = var_1452_cast_fp16)[name = string("op_1453_cast_fp16")]; + tensor var_1454_cast_fp16 = mul(x = x_43_cast_fp16, y = var_1453_cast_fp16)[name = string("op_1454_cast_fp16")]; + tensor input_55_cast_fp16 = mul(x = var_1454_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_55_cast_fp16")]; + string input_57_pad_type_0 = const()[name = string("input_57_pad_type_0"), val = string("valid")]; + tensor input_57_strides_0 = const()[name = string("input_57_strides_0"), val = tensor([1, 1])]; + tensor input_57_pad_0 = const()[name = string("input_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_57_dilations_0 = const()[name = string("input_57_dilations_0"), val = tensor([1, 1])]; + int32 input_57_groups_0 = const()[name = string("input_57_groups_0"), val = int32(1)]; + tensor input_57_cast_fp16 = conv(dilations = input_57_dilations_0, groups = input_57_groups_0, pad = input_57_pad_0, pad_type = input_57_pad_type_0, strides = input_57_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_55_cast_fp16)[name = string("input_57_cast_fp16")]; + tensor var_1462_cast_fp16 = silu(x = input_57_cast_fp16)[name = string("op_1462_cast_fp16")]; + string var_1468_pad_type_0 = const()[name = string("op_1468_pad_type_0"), val = string("valid")]; + tensor var_1468_strides_0 = const()[name = string("op_1468_strides_0"), val = tensor([1, 1])]; + tensor var_1468_pad_0 = const()[name = string("op_1468_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1468_dilations_0 = const()[name = string("op_1468_dilations_0"), val = tensor([1, 1])]; + int32 var_1468_groups_0 = const()[name = string("op_1468_groups_0"), val = int32(1)]; + tensor var_1468_cast_fp16 = conv(dilations = var_1468_dilations_0, groups = var_1468_groups_0, pad = var_1468_pad_0, pad_type = var_1468_pad_type_0, strides = var_1468_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_55_cast_fp16)[name = string("op_1468_cast_fp16")]; + tensor input_59_cast_fp16 = mul(x = var_1462_cast_fp16, y = var_1468_cast_fp16)[name = string("input_59_cast_fp16")]; + string h_11_pad_type_0 = const()[name = string("h_11_pad_type_0"), val = string("valid")]; + tensor h_11_strides_0 = const()[name = string("h_11_strides_0"), val = tensor([1, 1])]; + tensor h_11_pad_0 = const()[name = string("h_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_11_dilations_0 = const()[name = string("h_11_dilations_0"), val = tensor([1, 1])]; + int32 h_11_groups_0 = const()[name = string("h_11_groups_0"), val = int32(1)]; + tensor h_11_cast_fp16 = conv(dilations = h_11_dilations_0, groups = h_11_groups_0, pad = h_11_pad_0, pad_type = h_11_pad_type_0, strides = h_11_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_59_cast_fp16)[name = string("h_11_cast_fp16")]; + tensor x_45_cast_fp16 = add(x = x_43_cast_fp16, y = h_11_cast_fp16)[name = string("x_45_cast_fp16")]; + tensor key_cache_13_begin_0 = const()[name = string("key_cache_13_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_13_end_0 = const()[name = string("key_cache_13_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_13_end_mask_0 = const()[name = string("key_cache_13_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_13_cast_fp16 = slice_by_index(begin = key_cache_13_begin_0, end = key_cache_13_end_0, end_mask = key_cache_13_end_mask_0, x = layer_key_caches_3_cast_fp16)[name = string("key_cache_13_cast_fp16")]; + tensor value_cache_13_begin_0 = const()[name = string("value_cache_13_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_13_end_0 = const()[name = string("value_cache_13_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_13_end_mask_0 = const()[name = string("value_cache_13_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_13_cast_fp16 = slice_by_index(begin = value_cache_13_begin_0, end = value_cache_13_end_0, end_mask = value_cache_13_end_mask_0, x = layer_value_caches_3_cast_fp16)[name = string("value_cache_13_cast_fp16")]; + int32 var_1521 = const()[name = string("op_1521"), val = int32(2)]; + int32 var_1525 = const()[name = string("op_1525"), val = int32(3)]; + tensor var_1540_cast_fp16 = mul(x = x_45_cast_fp16, y = x_45_cast_fp16)[name = string("op_1540_cast_fp16")]; + tensor variance_49_axes_0 = const()[name = string("variance_49_axes_0"), val = tensor([1])]; + bool variance_49_keep_dims_0 = const()[name = string("variance_49_keep_dims_0"), val = bool(true)]; + tensor variance_49_cast_fp16 = reduce_mean(axes = variance_49_axes_0, keep_dims = variance_49_keep_dims_0, x = var_1540_cast_fp16)[name = string("variance_49_cast_fp16")]; + fp16 var_1543_to_fp16 = const()[name = string("op_1543_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1544_cast_fp16 = add(x = variance_49_cast_fp16, y = var_1543_to_fp16)[name = string("op_1544_cast_fp16")]; + fp32 var_1545_epsilon_0 = const()[name = string("op_1545_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1545_cast_fp16 = rsqrt(epsilon = var_1545_epsilon_0, x = var_1544_cast_fp16)[name = string("op_1545_cast_fp16")]; + tensor var_1546_cast_fp16 = mul(x = x_45_cast_fp16, y = var_1545_cast_fp16)[name = string("op_1546_cast_fp16")]; + tensor input_61_cast_fp16 = mul(x = var_1546_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_61_cast_fp16")]; + string q_37_pad_type_0 = const()[name = string("q_37_pad_type_0"), val = string("valid")]; + tensor q_37_strides_0 = const()[name = string("q_37_strides_0"), val = tensor([1, 1])]; + tensor q_37_pad_0 = const()[name = string("q_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_37_dilations_0 = const()[name = string("q_37_dilations_0"), val = tensor([1, 1])]; + int32 q_37_groups_0 = const()[name = string("q_37_groups_0"), val = int32(1)]; + tensor q_37_cast_fp16 = conv(dilations = q_37_dilations_0, groups = q_37_groups_0, pad = q_37_pad_0, pad_type = q_37_pad_type_0, strides = q_37_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_61_cast_fp16)[name = string("q_37_cast_fp16")]; + string k_37_pad_type_0 = const()[name = string("k_37_pad_type_0"), val = string("valid")]; + tensor k_37_strides_0 = const()[name = string("k_37_strides_0"), val = tensor([1, 1])]; + tensor k_37_pad_0 = const()[name = string("k_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_37_dilations_0 = const()[name = string("k_37_dilations_0"), val = tensor([1, 1])]; + int32 k_37_groups_0 = const()[name = string("k_37_groups_0"), val = int32(1)]; + tensor k_37_cast_fp16 = conv(dilations = k_37_dilations_0, groups = k_37_groups_0, pad = k_37_pad_0, pad_type = k_37_pad_type_0, strides = k_37_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_61_cast_fp16)[name = string("k_37_cast_fp16")]; + string v_13_pad_type_0 = const()[name = string("v_13_pad_type_0"), val = string("valid")]; + tensor v_13_strides_0 = const()[name = string("v_13_strides_0"), val = tensor([1, 1])]; + tensor v_13_pad_0 = const()[name = string("v_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_13_dilations_0 = const()[name = string("v_13_dilations_0"), val = tensor([1, 1])]; + int32 v_13_groups_0 = const()[name = string("v_13_groups_0"), val = int32(1)]; + tensor v_13_cast_fp16 = conv(dilations = v_13_dilations_0, groups = v_13_groups_0, pad = v_13_pad_0, pad_type = v_13_pad_type_0, strides = v_13_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_61_cast_fp16)[name = string("v_13_cast_fp16")]; + tensor var_1580 = const()[name = string("op_1580"), val = tensor([16, 128, 1, 1])]; + tensor x_47_cast_fp16 = reshape(shape = var_1580, x = q_37_cast_fp16)[name = string("x_47_cast_fp16")]; + tensor var_1583_cast_fp16 = mul(x = x_47_cast_fp16, y = x_47_cast_fp16)[name = string("op_1583_cast_fp16")]; + tensor variance_51_axes_0 = const()[name = string("variance_51_axes_0"), val = tensor([1])]; + bool variance_51_keep_dims_0 = const()[name = string("variance_51_keep_dims_0"), val = bool(true)]; + tensor variance_51_cast_fp16 = reduce_mean(axes = variance_51_axes_0, keep_dims = variance_51_keep_dims_0, x = var_1583_cast_fp16)[name = string("variance_51_cast_fp16")]; + fp16 var_1586_to_fp16 = const()[name = string("op_1586_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1587_cast_fp16 = add(x = variance_51_cast_fp16, y = var_1586_to_fp16)[name = string("op_1587_cast_fp16")]; + fp32 var_1588_epsilon_0 = const()[name = string("op_1588_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1588_cast_fp16 = rsqrt(epsilon = var_1588_epsilon_0, x = var_1587_cast_fp16)[name = string("op_1588_cast_fp16")]; + tensor var_1589_cast_fp16 = mul(x = x_47_cast_fp16, y = var_1588_cast_fp16)[name = string("op_1589_cast_fp16")]; + tensor q_39_cast_fp16 = mul(x = var_1589_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_39_cast_fp16")]; + tensor var_1591 = const()[name = string("op_1591"), val = tensor([8, 128, 1, 1])]; + tensor x_49_cast_fp16 = reshape(shape = var_1591, x = k_37_cast_fp16)[name = string("x_49_cast_fp16")]; + tensor var_1594_cast_fp16 = mul(x = x_49_cast_fp16, y = x_49_cast_fp16)[name = string("op_1594_cast_fp16")]; + tensor variance_53_axes_0 = const()[name = string("variance_53_axes_0"), val = tensor([1])]; + bool variance_53_keep_dims_0 = const()[name = string("variance_53_keep_dims_0"), val = bool(true)]; + tensor variance_53_cast_fp16 = reduce_mean(axes = variance_53_axes_0, keep_dims = variance_53_keep_dims_0, x = var_1594_cast_fp16)[name = string("variance_53_cast_fp16")]; + fp16 var_1597_to_fp16 = const()[name = string("op_1597_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1598_cast_fp16 = add(x = variance_53_cast_fp16, y = var_1597_to_fp16)[name = string("op_1598_cast_fp16")]; + fp32 var_1599_epsilon_0 = const()[name = string("op_1599_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1599_cast_fp16 = rsqrt(epsilon = var_1599_epsilon_0, x = var_1598_cast_fp16)[name = string("op_1599_cast_fp16")]; + tensor var_1600_cast_fp16 = mul(x = x_49_cast_fp16, y = var_1599_cast_fp16)[name = string("op_1600_cast_fp16")]; + tensor k_39_cast_fp16 = mul(x = var_1600_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_39_cast_fp16")]; + tensor var_1602 = const()[name = string("op_1602"), val = tensor([1, 16, 128, 1])]; + tensor z_25_cast_fp16 = reshape(shape = var_1602, x = q_39_cast_fp16)[name = string("z_25_cast_fp16")]; + tensor var_1604 = const()[name = string("op_1604"), val = tensor([1, 8, 128, 1])]; + tensor z_27_cast_fp16 = reshape(shape = var_1604, x = k_39_cast_fp16)[name = string("z_27_cast_fp16")]; + tensor z1_25_begin_0 = const()[name = string("z1_25_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_25_end_0 = const()[name = string("z1_25_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_25_end_mask_0 = const()[name = string("z1_25_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_25_cast_fp16 = slice_by_index(begin = z1_25_begin_0, end = z1_25_end_0, end_mask = z1_25_end_mask_0, x = z_25_cast_fp16)[name = string("z1_25_cast_fp16")]; + tensor z2_25_begin_0 = const()[name = string("z2_25_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_25_end_0 = const()[name = string("z2_25_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_25_end_mask_0 = const()[name = string("z2_25_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_25_cast_fp16 = slice_by_index(begin = z2_25_begin_0, end = z2_25_end_0, end_mask = z2_25_end_mask_0, x = z_25_cast_fp16)[name = string("z2_25_cast_fp16")]; + tensor var_1612_cast_fp16 = mul(x = z_25_cast_fp16, y = cos_11_to_fp16)[name = string("op_1612_cast_fp16")]; + fp16 const_14_promoted_to_fp16 = const()[name = string("const_14_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1613_cast_fp16 = mul(x = z2_25_cast_fp16, y = const_14_promoted_to_fp16)[name = string("op_1613_cast_fp16")]; + bool var_1615_interleave_0 = const()[name = string("op_1615_interleave_0"), val = bool(false)]; + tensor var_1615_cast_fp16 = concat(axis = var_1521, interleave = var_1615_interleave_0, values = (var_1613_cast_fp16, z1_25_cast_fp16))[name = string("op_1615_cast_fp16")]; + tensor var_1616_cast_fp16 = mul(x = var_1615_cast_fp16, y = sin_11_to_fp16)[name = string("op_1616_cast_fp16")]; + tensor q_41_cast_fp16 = add(x = var_1612_cast_fp16, y = var_1616_cast_fp16)[name = string("q_41_cast_fp16")]; + tensor z1_27_begin_0 = const()[name = string("z1_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_27_end_0 = const()[name = string("z1_27_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_27_end_mask_0 = const()[name = string("z1_27_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_27_cast_fp16 = slice_by_index(begin = z1_27_begin_0, end = z1_27_end_0, end_mask = z1_27_end_mask_0, x = z_27_cast_fp16)[name = string("z1_27_cast_fp16")]; + tensor z2_27_begin_0 = const()[name = string("z2_27_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_27_end_0 = const()[name = string("z2_27_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_27_end_mask_0 = const()[name = string("z2_27_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_27_cast_fp16 = slice_by_index(begin = z2_27_begin_0, end = z2_27_end_0, end_mask = z2_27_end_mask_0, x = z_27_cast_fp16)[name = string("z2_27_cast_fp16")]; + tensor var_1624_cast_fp16 = mul(x = z_27_cast_fp16, y = cos_11_to_fp16)[name = string("op_1624_cast_fp16")]; + fp16 const_15_promoted_to_fp16 = const()[name = string("const_15_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1625_cast_fp16 = mul(x = z2_27_cast_fp16, y = const_15_promoted_to_fp16)[name = string("op_1625_cast_fp16")]; + bool var_1627_interleave_0 = const()[name = string("op_1627_interleave_0"), val = bool(false)]; + tensor var_1627_cast_fp16 = concat(axis = var_1521, interleave = var_1627_interleave_0, values = (var_1625_cast_fp16, z1_27_cast_fp16))[name = string("op_1627_cast_fp16")]; + tensor var_1628_cast_fp16 = mul(x = var_1627_cast_fp16, y = sin_11_to_fp16)[name = string("op_1628_cast_fp16")]; + tensor k_41_cast_fp16 = add(x = var_1624_cast_fp16, y = var_1628_cast_fp16)[name = string("k_41_cast_fp16")]; + tensor var_1630 = const()[name = string("op_1630"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_13_cast_fp16 = reshape(shape = var_1630, x = k_41_cast_fp16)[name = string("cur_key_13_cast_fp16")]; + tensor var_1632_to_fp16 = const()[name = string("op_1632_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174784)))]; + tensor var_1633_cast_fp16 = mul(x = key_cache_13_cast_fp16, y = var_1632_to_fp16)[name = string("op_1633_cast_fp16")]; + tensor upd_13_to_fp16 = const()[name = string("upd_13_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174912)))]; + tensor var_1634_cast_fp16 = mul(x = cur_key_13_cast_fp16, y = upd_13_to_fp16)[name = string("op_1634_cast_fp16")]; + tensor key_13_cast_fp16 = add(x = var_1633_cast_fp16, y = var_1634_cast_fp16)[name = string("key_13_cast_fp16")]; + tensor var_1636_to_fp16 = const()[name = string("op_1636_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174784)))]; + tensor var_1637_cast_fp16 = mul(x = value_cache_13_cast_fp16, y = var_1636_to_fp16)[name = string("op_1637_cast_fp16")]; + tensor var_1638_cast_fp16 = mul(x = v_13_cast_fp16, y = upd_13_to_fp16)[name = string("op_1638_cast_fp16")]; + tensor value_13_cast_fp16 = add(x = var_1637_cast_fp16, y = var_1638_cast_fp16)[name = string("value_13_cast_fp16")]; + tensor var_1640 = const()[name = string("op_1640"), val = tensor([1, 8, 128, 16])]; + tensor kh_25_cast_fp16 = reshape(shape = var_1640, x = key_13_cast_fp16)[name = string("kh_25_cast_fp16")]; + tensor var_1642 = const()[name = string("op_1642"), val = tensor([1, 8, 128, 16])]; + tensor vh_25_cast_fp16 = reshape(shape = var_1642, x = value_13_cast_fp16)[name = string("vh_25_cast_fp16")]; + tensor transpose_24_perm_0 = const()[name = string("transpose_24_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_12_reps_0 = const()[name = string("tile_12_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_24_cast_fp16 = transpose(perm = transpose_24_perm_0, x = kh_25_cast_fp16)[name = string("transpose_443")]; + tensor tile_12_cast_fp16 = tile(reps = tile_12_reps_0, x = transpose_24_cast_fp16)[name = string("tile_12_cast_fp16")]; + tensor concat_32 = const()[name = string("concat_32"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_24_cast_fp16 = reshape(shape = concat_32, x = tile_12_cast_fp16)[name = string("reshape_24_cast_fp16")]; + tensor transpose_25_perm_0 = const()[name = string("transpose_25_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_33 = const()[name = string("concat_33"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_25_cast_fp16 = transpose(perm = transpose_25_perm_0, x = reshape_24_cast_fp16)[name = string("transpose_442")]; + tensor reshape_25_cast_fp16 = reshape(shape = concat_33, x = transpose_25_cast_fp16)[name = string("reshape_25_cast_fp16")]; + tensor transpose_26_perm_0 = const()[name = string("transpose_26_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_13_reps_0 = const()[name = string("tile_13_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_26_cast_fp16 = transpose(perm = transpose_26_perm_0, x = vh_25_cast_fp16)[name = string("transpose_441")]; + tensor tile_13_cast_fp16 = tile(reps = tile_13_reps_0, x = transpose_26_cast_fp16)[name = string("tile_13_cast_fp16")]; + tensor concat_34 = const()[name = string("concat_34"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_26_cast_fp16 = reshape(shape = concat_34, x = tile_13_cast_fp16)[name = string("reshape_26_cast_fp16")]; + tensor transpose_27_perm_0 = const()[name = string("transpose_27_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_35 = const()[name = string("concat_35"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_27_cast_fp16 = transpose(perm = transpose_27_perm_0, x = reshape_26_cast_fp16)[name = string("transpose_440")]; + tensor reshape_27_cast_fp16 = reshape(shape = concat_35, x = transpose_27_cast_fp16)[name = string("reshape_27_cast_fp16")]; + fp16 var_1646_to_fp16 = const()[name = string("op_1646_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_1647_cast_fp16 = mul(x = q_41_cast_fp16, y = var_1646_to_fp16)[name = string("op_1647_cast_fp16")]; + tensor transpose_341_perm_0 = const()[name = string("transpose_341_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_25_transpose_x_1 = const()[name = string("w_25_transpose_x_1"), val = bool(true)]; + bool w_25_transpose_y_1 = const()[name = string("w_25_transpose_y_1"), val = bool(false)]; + tensor transpose_341_cast_fp16 = transpose(perm = transpose_341_perm_0, x = reshape_25_cast_fp16)[name = string("transpose_439")]; + tensor w_25_cast_fp16 = matmul(transpose_x = w_25_transpose_x_1, transpose_y = w_25_transpose_y_1, x = var_1647_cast_fp16, y = transpose_341_cast_fp16)[name = string("w_25_cast_fp16")]; + tensor pad_13_to_fp16 = const()[name = string("pad_13_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110175040)))]; + tensor var_1650_cast_fp16 = add(x = w_25_cast_fp16, y = pad_13_to_fp16)[name = string("op_1650_cast_fp16")]; + tensor w_27_cast_fp16 = softmax(axis = var_1525, x = var_1650_cast_fp16)[name = string("w_27_cast_fp16")]; + tensor transpose_342_perm_0 = const()[name = string("transpose_342_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_13_transpose_x_1 = const()[name = string("attn_13_transpose_x_1"), val = bool(false)]; + bool attn_13_transpose_y_1 = const()[name = string("attn_13_transpose_y_1"), val = bool(true)]; + tensor transpose_342_cast_fp16 = transpose(perm = transpose_342_perm_0, x = reshape_27_cast_fp16)[name = string("transpose_438")]; + tensor attn_13_cast_fp16 = matmul(transpose_x = attn_13_transpose_x_1, transpose_y = attn_13_transpose_y_1, x = transpose_342_cast_fp16, y = w_27_cast_fp16)[name = string("attn_13_cast_fp16")]; + tensor var_1654 = const()[name = string("op_1654"), val = tensor([1, 2048, 1, 1])]; + tensor input_63_cast_fp16 = reshape(shape = var_1654, x = attn_13_cast_fp16)[name = string("input_63_cast_fp16")]; + string attn_output_13_pad_type_0 = const()[name = string("attn_output_13_pad_type_0"), val = string("valid")]; + tensor attn_output_13_strides_0 = const()[name = string("attn_output_13_strides_0"), val = tensor([1, 1])]; + tensor attn_output_13_pad_0 = const()[name = string("attn_output_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_13_dilations_0 = const()[name = string("attn_output_13_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_13_groups_0 = const()[name = string("attn_output_13_groups_0"), val = int32(1)]; + tensor attn_output_13_cast_fp16 = conv(dilations = attn_output_13_dilations_0, groups = attn_output_13_groups_0, pad = attn_output_13_pad_0, pad_type = attn_output_13_pad_type_0, strides = attn_output_13_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_63_cast_fp16)[name = string("attn_output_13_cast_fp16")]; + tensor x_51_cast_fp16 = add(x = x_45_cast_fp16, y = attn_output_13_cast_fp16)[name = string("x_51_cast_fp16")]; + tensor var_1668_cast_fp16 = mul(x = x_51_cast_fp16, y = x_51_cast_fp16)[name = string("op_1668_cast_fp16")]; + tensor variance_55_axes_0 = const()[name = string("variance_55_axes_0"), val = tensor([1])]; + bool variance_55_keep_dims_0 = const()[name = string("variance_55_keep_dims_0"), val = bool(true)]; + tensor variance_55_cast_fp16 = reduce_mean(axes = variance_55_axes_0, keep_dims = variance_55_keep_dims_0, x = var_1668_cast_fp16)[name = string("variance_55_cast_fp16")]; + fp16 var_1671_to_fp16 = const()[name = string("op_1671_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1672_cast_fp16 = add(x = variance_55_cast_fp16, y = var_1671_to_fp16)[name = string("op_1672_cast_fp16")]; + fp32 var_1673_epsilon_0 = const()[name = string("op_1673_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1673_cast_fp16 = rsqrt(epsilon = var_1673_epsilon_0, x = var_1672_cast_fp16)[name = string("op_1673_cast_fp16")]; + tensor var_1674_cast_fp16 = mul(x = x_51_cast_fp16, y = var_1673_cast_fp16)[name = string("op_1674_cast_fp16")]; + tensor input_65_cast_fp16 = mul(x = var_1674_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_65_cast_fp16")]; + string input_67_pad_type_0 = const()[name = string("input_67_pad_type_0"), val = string("valid")]; + tensor input_67_strides_0 = const()[name = string("input_67_strides_0"), val = tensor([1, 1])]; + tensor input_67_pad_0 = const()[name = string("input_67_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_67_dilations_0 = const()[name = string("input_67_dilations_0"), val = tensor([1, 1])]; + int32 input_67_groups_0 = const()[name = string("input_67_groups_0"), val = int32(1)]; + tensor input_67_cast_fp16 = conv(dilations = input_67_dilations_0, groups = input_67_groups_0, pad = input_67_pad_0, pad_type = input_67_pad_type_0, strides = input_67_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_65_cast_fp16)[name = string("input_67_cast_fp16")]; + tensor var_1682_cast_fp16 = silu(x = input_67_cast_fp16)[name = string("op_1682_cast_fp16")]; + string var_1688_pad_type_0 = const()[name = string("op_1688_pad_type_0"), val = string("valid")]; + tensor var_1688_strides_0 = const()[name = string("op_1688_strides_0"), val = tensor([1, 1])]; + tensor var_1688_pad_0 = const()[name = string("op_1688_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1688_dilations_0 = const()[name = string("op_1688_dilations_0"), val = tensor([1, 1])]; + int32 var_1688_groups_0 = const()[name = string("op_1688_groups_0"), val = int32(1)]; + tensor var_1688_cast_fp16 = conv(dilations = var_1688_dilations_0, groups = var_1688_groups_0, pad = var_1688_pad_0, pad_type = var_1688_pad_type_0, strides = var_1688_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_65_cast_fp16)[name = string("op_1688_cast_fp16")]; + tensor input_69_cast_fp16 = mul(x = var_1682_cast_fp16, y = var_1688_cast_fp16)[name = string("input_69_cast_fp16")]; + string h_13_pad_type_0 = const()[name = string("h_13_pad_type_0"), val = string("valid")]; + tensor h_13_strides_0 = const()[name = string("h_13_strides_0"), val = tensor([1, 1])]; + tensor h_13_pad_0 = const()[name = string("h_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_13_dilations_0 = const()[name = string("h_13_dilations_0"), val = tensor([1, 1])]; + int32 h_13_groups_0 = const()[name = string("h_13_groups_0"), val = int32(1)]; + tensor h_13_cast_fp16 = conv(dilations = h_13_dilations_0, groups = h_13_groups_0, pad = h_13_pad_0, pad_type = h_13_pad_type_0, strides = h_13_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_69_cast_fp16)[name = string("h_13_cast_fp16")]; + tensor x_53_cast_fp16 = add(x = x_51_cast_fp16, y = h_13_cast_fp16)[name = string("x_53_cast_fp16")]; + tensor key_cache_15_begin_0 = const()[name = string("key_cache_15_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_15_end_0 = const()[name = string("key_cache_15_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_15_end_mask_0 = const()[name = string("key_cache_15_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_15_cast_fp16 = slice_by_index(begin = key_cache_15_begin_0, end = key_cache_15_end_0, end_mask = key_cache_15_end_mask_0, x = layer_key_caches_3_cast_fp16)[name = string("key_cache_15_cast_fp16")]; + tensor value_cache_15_begin_0 = const()[name = string("value_cache_15_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_15_end_0 = const()[name = string("value_cache_15_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_15_end_mask_0 = const()[name = string("value_cache_15_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_15_cast_fp16 = slice_by_index(begin = value_cache_15_begin_0, end = value_cache_15_end_0, end_mask = value_cache_15_end_mask_0, x = layer_value_caches_3_cast_fp16)[name = string("value_cache_15_cast_fp16")]; + int32 var_1741 = const()[name = string("op_1741"), val = int32(2)]; + int32 var_1745 = const()[name = string("op_1745"), val = int32(3)]; + tensor var_1760_cast_fp16 = mul(x = x_53_cast_fp16, y = x_53_cast_fp16)[name = string("op_1760_cast_fp16")]; + tensor variance_57_axes_0 = const()[name = string("variance_57_axes_0"), val = tensor([1])]; + bool variance_57_keep_dims_0 = const()[name = string("variance_57_keep_dims_0"), val = bool(true)]; + tensor variance_57_cast_fp16 = reduce_mean(axes = variance_57_axes_0, keep_dims = variance_57_keep_dims_0, x = var_1760_cast_fp16)[name = string("variance_57_cast_fp16")]; + fp16 var_1763_to_fp16 = const()[name = string("op_1763_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1764_cast_fp16 = add(x = variance_57_cast_fp16, y = var_1763_to_fp16)[name = string("op_1764_cast_fp16")]; + fp32 var_1765_epsilon_0 = const()[name = string("op_1765_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1765_cast_fp16 = rsqrt(epsilon = var_1765_epsilon_0, x = var_1764_cast_fp16)[name = string("op_1765_cast_fp16")]; + tensor var_1766_cast_fp16 = mul(x = x_53_cast_fp16, y = var_1765_cast_fp16)[name = string("op_1766_cast_fp16")]; + tensor input_71_cast_fp16 = mul(x = var_1766_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_71_cast_fp16")]; + string q_43_pad_type_0 = const()[name = string("q_43_pad_type_0"), val = string("valid")]; + tensor q_43_strides_0 = const()[name = string("q_43_strides_0"), val = tensor([1, 1])]; + tensor q_43_pad_0 = const()[name = string("q_43_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_43_dilations_0 = const()[name = string("q_43_dilations_0"), val = tensor([1, 1])]; + int32 q_43_groups_0 = const()[name = string("q_43_groups_0"), val = int32(1)]; + tensor q_43_cast_fp16 = conv(dilations = q_43_dilations_0, groups = q_43_groups_0, pad = q_43_pad_0, pad_type = q_43_pad_type_0, strides = q_43_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_71_cast_fp16)[name = string("q_43_cast_fp16")]; + string k_43_pad_type_0 = const()[name = string("k_43_pad_type_0"), val = string("valid")]; + tensor k_43_strides_0 = const()[name = string("k_43_strides_0"), val = tensor([1, 1])]; + tensor k_43_pad_0 = const()[name = string("k_43_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_43_dilations_0 = const()[name = string("k_43_dilations_0"), val = tensor([1, 1])]; + int32 k_43_groups_0 = const()[name = string("k_43_groups_0"), val = int32(1)]; + tensor k_43_cast_fp16 = conv(dilations = k_43_dilations_0, groups = k_43_groups_0, pad = k_43_pad_0, pad_type = k_43_pad_type_0, strides = k_43_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_71_cast_fp16)[name = string("k_43_cast_fp16")]; + string v_15_pad_type_0 = const()[name = string("v_15_pad_type_0"), val = string("valid")]; + tensor v_15_strides_0 = const()[name = string("v_15_strides_0"), val = tensor([1, 1])]; + tensor v_15_pad_0 = const()[name = string("v_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_15_dilations_0 = const()[name = string("v_15_dilations_0"), val = tensor([1, 1])]; + int32 v_15_groups_0 = const()[name = string("v_15_groups_0"), val = int32(1)]; + tensor v_15_cast_fp16 = conv(dilations = v_15_dilations_0, groups = v_15_groups_0, pad = v_15_pad_0, pad_type = v_15_pad_type_0, strides = v_15_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_71_cast_fp16)[name = string("v_15_cast_fp16")]; + tensor var_1800 = const()[name = string("op_1800"), val = tensor([16, 128, 1, 1])]; + tensor x_55_cast_fp16 = reshape(shape = var_1800, x = q_43_cast_fp16)[name = string("x_55_cast_fp16")]; + tensor var_1803_cast_fp16 = mul(x = x_55_cast_fp16, y = x_55_cast_fp16)[name = string("op_1803_cast_fp16")]; + tensor variance_59_axes_0 = const()[name = string("variance_59_axes_0"), val = tensor([1])]; + bool variance_59_keep_dims_0 = const()[name = string("variance_59_keep_dims_0"), val = bool(true)]; + tensor variance_59_cast_fp16 = reduce_mean(axes = variance_59_axes_0, keep_dims = variance_59_keep_dims_0, x = var_1803_cast_fp16)[name = string("variance_59_cast_fp16")]; + fp16 var_1806_to_fp16 = const()[name = string("op_1806_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1807_cast_fp16 = add(x = variance_59_cast_fp16, y = var_1806_to_fp16)[name = string("op_1807_cast_fp16")]; + fp32 var_1808_epsilon_0 = const()[name = string("op_1808_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1808_cast_fp16 = rsqrt(epsilon = var_1808_epsilon_0, x = var_1807_cast_fp16)[name = string("op_1808_cast_fp16")]; + tensor var_1809_cast_fp16 = mul(x = x_55_cast_fp16, y = var_1808_cast_fp16)[name = string("op_1809_cast_fp16")]; + tensor q_45_cast_fp16 = mul(x = var_1809_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_45_cast_fp16")]; + tensor var_1811 = const()[name = string("op_1811"), val = tensor([8, 128, 1, 1])]; + tensor x_57_cast_fp16 = reshape(shape = var_1811, x = k_43_cast_fp16)[name = string("x_57_cast_fp16")]; + tensor var_1814_cast_fp16 = mul(x = x_57_cast_fp16, y = x_57_cast_fp16)[name = string("op_1814_cast_fp16")]; + tensor variance_61_axes_0 = const()[name = string("variance_61_axes_0"), val = tensor([1])]; + bool variance_61_keep_dims_0 = const()[name = string("variance_61_keep_dims_0"), val = bool(true)]; + tensor variance_61_cast_fp16 = reduce_mean(axes = variance_61_axes_0, keep_dims = variance_61_keep_dims_0, x = var_1814_cast_fp16)[name = string("variance_61_cast_fp16")]; + fp16 var_1817_to_fp16 = const()[name = string("op_1817_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1818_cast_fp16 = add(x = variance_61_cast_fp16, y = var_1817_to_fp16)[name = string("op_1818_cast_fp16")]; + fp32 var_1819_epsilon_0 = const()[name = string("op_1819_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1819_cast_fp16 = rsqrt(epsilon = var_1819_epsilon_0, x = var_1818_cast_fp16)[name = string("op_1819_cast_fp16")]; + tensor var_1820_cast_fp16 = mul(x = x_57_cast_fp16, y = var_1819_cast_fp16)[name = string("op_1820_cast_fp16")]; + tensor k_45_cast_fp16 = mul(x = var_1820_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_45_cast_fp16")]; + tensor var_1822 = const()[name = string("op_1822"), val = tensor([1, 16, 128, 1])]; + tensor z_29_cast_fp16 = reshape(shape = var_1822, x = q_45_cast_fp16)[name = string("z_29_cast_fp16")]; + tensor var_1824 = const()[name = string("op_1824"), val = tensor([1, 8, 128, 1])]; + tensor z_31_cast_fp16 = reshape(shape = var_1824, x = k_45_cast_fp16)[name = string("z_31_cast_fp16")]; + tensor z1_29_begin_0 = const()[name = string("z1_29_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_29_end_0 = const()[name = string("z1_29_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_29_end_mask_0 = const()[name = string("z1_29_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_29_cast_fp16 = slice_by_index(begin = z1_29_begin_0, end = z1_29_end_0, end_mask = z1_29_end_mask_0, x = z_29_cast_fp16)[name = string("z1_29_cast_fp16")]; + tensor z2_29_begin_0 = const()[name = string("z2_29_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_29_end_0 = const()[name = string("z2_29_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_29_end_mask_0 = const()[name = string("z2_29_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_29_cast_fp16 = slice_by_index(begin = z2_29_begin_0, end = z2_29_end_0, end_mask = z2_29_end_mask_0, x = z_29_cast_fp16)[name = string("z2_29_cast_fp16")]; + tensor var_1832_cast_fp16 = mul(x = z_29_cast_fp16, y = cos_11_to_fp16)[name = string("op_1832_cast_fp16")]; + fp16 const_16_promoted_to_fp16 = const()[name = string("const_16_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1833_cast_fp16 = mul(x = z2_29_cast_fp16, y = const_16_promoted_to_fp16)[name = string("op_1833_cast_fp16")]; + bool var_1835_interleave_0 = const()[name = string("op_1835_interleave_0"), val = bool(false)]; + tensor var_1835_cast_fp16 = concat(axis = var_1741, interleave = var_1835_interleave_0, values = (var_1833_cast_fp16, z1_29_cast_fp16))[name = string("op_1835_cast_fp16")]; + tensor var_1836_cast_fp16 = mul(x = var_1835_cast_fp16, y = sin_11_to_fp16)[name = string("op_1836_cast_fp16")]; + tensor q_47_cast_fp16 = add(x = var_1832_cast_fp16, y = var_1836_cast_fp16)[name = string("q_47_cast_fp16")]; + tensor z1_31_begin_0 = const()[name = string("z1_31_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_31_end_0 = const()[name = string("z1_31_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_31_end_mask_0 = const()[name = string("z1_31_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_31_cast_fp16 = slice_by_index(begin = z1_31_begin_0, end = z1_31_end_0, end_mask = z1_31_end_mask_0, x = z_31_cast_fp16)[name = string("z1_31_cast_fp16")]; + tensor z2_31_begin_0 = const()[name = string("z2_31_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_31_end_0 = const()[name = string("z2_31_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_31_end_mask_0 = const()[name = string("z2_31_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_31_cast_fp16 = slice_by_index(begin = z2_31_begin_0, end = z2_31_end_0, end_mask = z2_31_end_mask_0, x = z_31_cast_fp16)[name = string("z2_31_cast_fp16")]; + tensor var_1844_cast_fp16 = mul(x = z_31_cast_fp16, y = cos_11_to_fp16)[name = string("op_1844_cast_fp16")]; + fp16 const_17_promoted_to_fp16 = const()[name = string("const_17_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1845_cast_fp16 = mul(x = z2_31_cast_fp16, y = const_17_promoted_to_fp16)[name = string("op_1845_cast_fp16")]; + bool var_1847_interleave_0 = const()[name = string("op_1847_interleave_0"), val = bool(false)]; + tensor var_1847_cast_fp16 = concat(axis = var_1741, interleave = var_1847_interleave_0, values = (var_1845_cast_fp16, z1_31_cast_fp16))[name = string("op_1847_cast_fp16")]; + tensor var_1848_cast_fp16 = mul(x = var_1847_cast_fp16, y = sin_11_to_fp16)[name = string("op_1848_cast_fp16")]; + tensor k_47_cast_fp16 = add(x = var_1844_cast_fp16, y = var_1848_cast_fp16)[name = string("k_47_cast_fp16")]; + tensor var_1850 = const()[name = string("op_1850"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_15_cast_fp16 = reshape(shape = var_1850, x = k_47_cast_fp16)[name = string("cur_key_15_cast_fp16")]; + tensor var_1852_to_fp16 = const()[name = string("op_1852_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174784)))]; + tensor var_1853_cast_fp16 = mul(x = key_cache_15_cast_fp16, y = var_1852_to_fp16)[name = string("op_1853_cast_fp16")]; + tensor upd_15_to_fp16 = const()[name = string("upd_15_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174912)))]; + tensor var_1854_cast_fp16 = mul(x = cur_key_15_cast_fp16, y = upd_15_to_fp16)[name = string("op_1854_cast_fp16")]; + tensor key_15_cast_fp16 = add(x = var_1853_cast_fp16, y = var_1854_cast_fp16)[name = string("key_15_cast_fp16")]; + tensor var_1856_to_fp16 = const()[name = string("op_1856_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174784)))]; + tensor var_1857_cast_fp16 = mul(x = value_cache_15_cast_fp16, y = var_1856_to_fp16)[name = string("op_1857_cast_fp16")]; + tensor var_1858_cast_fp16 = mul(x = v_15_cast_fp16, y = upd_15_to_fp16)[name = string("op_1858_cast_fp16")]; + tensor value_15_cast_fp16 = add(x = var_1857_cast_fp16, y = var_1858_cast_fp16)[name = string("value_15_cast_fp16")]; + tensor var_1860 = const()[name = string("op_1860"), val = tensor([1, 8, 128, 16])]; + tensor kh_29_cast_fp16 = reshape(shape = var_1860, x = key_15_cast_fp16)[name = string("kh_29_cast_fp16")]; + tensor var_1862 = const()[name = string("op_1862"), val = tensor([1, 8, 128, 16])]; + tensor vh_29_cast_fp16 = reshape(shape = var_1862, x = value_15_cast_fp16)[name = string("vh_29_cast_fp16")]; + tensor transpose_28_perm_0 = const()[name = string("transpose_28_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_14_reps_0 = const()[name = string("tile_14_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_28_cast_fp16 = transpose(perm = transpose_28_perm_0, x = kh_29_cast_fp16)[name = string("transpose_437")]; + tensor tile_14_cast_fp16 = tile(reps = tile_14_reps_0, x = transpose_28_cast_fp16)[name = string("tile_14_cast_fp16")]; + tensor concat_36 = const()[name = string("concat_36"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_28_cast_fp16 = reshape(shape = concat_36, x = tile_14_cast_fp16)[name = string("reshape_28_cast_fp16")]; + tensor transpose_29_perm_0 = const()[name = string("transpose_29_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_37 = const()[name = string("concat_37"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_29_cast_fp16 = transpose(perm = transpose_29_perm_0, x = reshape_28_cast_fp16)[name = string("transpose_436")]; + tensor reshape_29_cast_fp16 = reshape(shape = concat_37, x = transpose_29_cast_fp16)[name = string("reshape_29_cast_fp16")]; + tensor transpose_30_perm_0 = const()[name = string("transpose_30_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_15_reps_0 = const()[name = string("tile_15_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_30_cast_fp16 = transpose(perm = transpose_30_perm_0, x = vh_29_cast_fp16)[name = string("transpose_435")]; + tensor tile_15_cast_fp16 = tile(reps = tile_15_reps_0, x = transpose_30_cast_fp16)[name = string("tile_15_cast_fp16")]; + tensor concat_38 = const()[name = string("concat_38"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_30_cast_fp16 = reshape(shape = concat_38, x = tile_15_cast_fp16)[name = string("reshape_30_cast_fp16")]; + tensor transpose_31_perm_0 = const()[name = string("transpose_31_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_39 = const()[name = string("concat_39"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_31_cast_fp16 = transpose(perm = transpose_31_perm_0, x = reshape_30_cast_fp16)[name = string("transpose_434")]; + tensor reshape_31_cast_fp16 = reshape(shape = concat_39, x = transpose_31_cast_fp16)[name = string("reshape_31_cast_fp16")]; + fp16 var_1866_to_fp16 = const()[name = string("op_1866_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_1867_cast_fp16 = mul(x = q_47_cast_fp16, y = var_1866_to_fp16)[name = string("op_1867_cast_fp16")]; + tensor transpose_345_perm_0 = const()[name = string("transpose_345_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_29_transpose_x_1 = const()[name = string("w_29_transpose_x_1"), val = bool(true)]; + bool w_29_transpose_y_1 = const()[name = string("w_29_transpose_y_1"), val = bool(false)]; + tensor transpose_345_cast_fp16 = transpose(perm = transpose_345_perm_0, x = reshape_29_cast_fp16)[name = string("transpose_433")]; + tensor w_29_cast_fp16 = matmul(transpose_x = w_29_transpose_x_1, transpose_y = w_29_transpose_y_1, x = var_1867_cast_fp16, y = transpose_345_cast_fp16)[name = string("w_29_cast_fp16")]; + tensor pad_15_to_fp16 = const()[name = string("pad_15_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110175040)))]; + tensor var_1870_cast_fp16 = add(x = w_29_cast_fp16, y = pad_15_to_fp16)[name = string("op_1870_cast_fp16")]; + tensor w_31_cast_fp16 = softmax(axis = var_1745, x = var_1870_cast_fp16)[name = string("w_31_cast_fp16")]; + tensor transpose_346_perm_0 = const()[name = string("transpose_346_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_15_transpose_x_1 = const()[name = string("attn_15_transpose_x_1"), val = bool(false)]; + bool attn_15_transpose_y_1 = const()[name = string("attn_15_transpose_y_1"), val = bool(true)]; + tensor transpose_346_cast_fp16 = transpose(perm = transpose_346_perm_0, x = reshape_31_cast_fp16)[name = string("transpose_432")]; + tensor attn_15_cast_fp16 = matmul(transpose_x = attn_15_transpose_x_1, transpose_y = attn_15_transpose_y_1, x = transpose_346_cast_fp16, y = w_31_cast_fp16)[name = string("attn_15_cast_fp16")]; + tensor var_1874 = const()[name = string("op_1874"), val = tensor([1, 2048, 1, 1])]; + tensor input_73_cast_fp16 = reshape(shape = var_1874, x = attn_15_cast_fp16)[name = string("input_73_cast_fp16")]; + string attn_output_15_pad_type_0 = const()[name = string("attn_output_15_pad_type_0"), val = string("valid")]; + tensor attn_output_15_strides_0 = const()[name = string("attn_output_15_strides_0"), val = tensor([1, 1])]; + tensor attn_output_15_pad_0 = const()[name = string("attn_output_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_15_dilations_0 = const()[name = string("attn_output_15_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_15_groups_0 = const()[name = string("attn_output_15_groups_0"), val = int32(1)]; + tensor attn_output_15_cast_fp16 = conv(dilations = attn_output_15_dilations_0, groups = attn_output_15_groups_0, pad = attn_output_15_pad_0, pad_type = attn_output_15_pad_type_0, strides = attn_output_15_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_73_cast_fp16)[name = string("attn_output_15_cast_fp16")]; + tensor x_59_cast_fp16 = add(x = x_53_cast_fp16, y = attn_output_15_cast_fp16)[name = string("x_59_cast_fp16")]; + tensor var_1888_cast_fp16 = mul(x = x_59_cast_fp16, y = x_59_cast_fp16)[name = string("op_1888_cast_fp16")]; + tensor variance_63_axes_0 = const()[name = string("variance_63_axes_0"), val = tensor([1])]; + bool variance_63_keep_dims_0 = const()[name = string("variance_63_keep_dims_0"), val = bool(true)]; + tensor variance_63_cast_fp16 = reduce_mean(axes = variance_63_axes_0, keep_dims = variance_63_keep_dims_0, x = var_1888_cast_fp16)[name = string("variance_63_cast_fp16")]; + fp16 var_1891_to_fp16 = const()[name = string("op_1891_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1892_cast_fp16 = add(x = variance_63_cast_fp16, y = var_1891_to_fp16)[name = string("op_1892_cast_fp16")]; + fp32 var_1893_epsilon_0 = const()[name = string("op_1893_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1893_cast_fp16 = rsqrt(epsilon = var_1893_epsilon_0, x = var_1892_cast_fp16)[name = string("op_1893_cast_fp16")]; + tensor var_1894_cast_fp16 = mul(x = x_59_cast_fp16, y = var_1893_cast_fp16)[name = string("op_1894_cast_fp16")]; + tensor input_75_cast_fp16 = mul(x = var_1894_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_75_cast_fp16")]; + string input_77_pad_type_0 = const()[name = string("input_77_pad_type_0"), val = string("valid")]; + tensor input_77_strides_0 = const()[name = string("input_77_strides_0"), val = tensor([1, 1])]; + tensor input_77_pad_0 = const()[name = string("input_77_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_77_dilations_0 = const()[name = string("input_77_dilations_0"), val = tensor([1, 1])]; + int32 input_77_groups_0 = const()[name = string("input_77_groups_0"), val = int32(1)]; + tensor input_77_cast_fp16 = conv(dilations = input_77_dilations_0, groups = input_77_groups_0, pad = input_77_pad_0, pad_type = input_77_pad_type_0, strides = input_77_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_75_cast_fp16)[name = string("input_77_cast_fp16")]; + tensor var_1902_cast_fp16 = silu(x = input_77_cast_fp16)[name = string("op_1902_cast_fp16")]; + string var_1908_pad_type_0 = const()[name = string("op_1908_pad_type_0"), val = string("valid")]; + tensor var_1908_strides_0 = const()[name = string("op_1908_strides_0"), val = tensor([1, 1])]; + tensor var_1908_pad_0 = const()[name = string("op_1908_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1908_dilations_0 = const()[name = string("op_1908_dilations_0"), val = tensor([1, 1])]; + int32 var_1908_groups_0 = const()[name = string("op_1908_groups_0"), val = int32(1)]; + tensor var_1908_cast_fp16 = conv(dilations = var_1908_dilations_0, groups = var_1908_groups_0, pad = var_1908_pad_0, pad_type = var_1908_pad_type_0, strides = var_1908_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_75_cast_fp16)[name = string("op_1908_cast_fp16")]; + tensor input_79_cast_fp16 = mul(x = var_1902_cast_fp16, y = var_1908_cast_fp16)[name = string("input_79_cast_fp16")]; + string h_15_pad_type_0 = const()[name = string("h_15_pad_type_0"), val = string("valid")]; + tensor h_15_strides_0 = const()[name = string("h_15_strides_0"), val = tensor([1, 1])]; + tensor h_15_pad_0 = const()[name = string("h_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_15_dilations_0 = const()[name = string("h_15_dilations_0"), val = tensor([1, 1])]; + int32 h_15_groups_0 = const()[name = string("h_15_groups_0"), val = int32(1)]; + tensor h_15_cast_fp16 = conv(dilations = h_15_dilations_0, groups = h_15_groups_0, pad = h_15_pad_0, pad_type = h_15_pad_type_0, strides = h_15_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_79_cast_fp16)[name = string("h_15_cast_fp16")]; + tensor x_61_cast_fp16 = add(x = x_59_cast_fp16, y = h_15_cast_fp16)[name = string("x_61_cast_fp16")]; + tensor key_cache_17_begin_0 = const()[name = string("key_cache_17_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_17_end_0 = const()[name = string("key_cache_17_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_17_end_mask_0 = const()[name = string("key_cache_17_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_17_cast_fp16 = slice_by_index(begin = key_cache_17_begin_0, end = key_cache_17_end_0, end_mask = key_cache_17_end_mask_0, x = layer_key_caches_3_cast_fp16)[name = string("key_cache_17_cast_fp16")]; + tensor value_cache_17_begin_0 = const()[name = string("value_cache_17_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_17_end_0 = const()[name = string("value_cache_17_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_17_end_mask_0 = const()[name = string("value_cache_17_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_17_cast_fp16 = slice_by_index(begin = value_cache_17_begin_0, end = value_cache_17_end_0, end_mask = value_cache_17_end_mask_0, x = layer_value_caches_3_cast_fp16)[name = string("value_cache_17_cast_fp16")]; + int32 var_1961 = const()[name = string("op_1961"), val = int32(2)]; + int32 var_1965 = const()[name = string("op_1965"), val = int32(3)]; + tensor var_1980_cast_fp16 = mul(x = x_61_cast_fp16, y = x_61_cast_fp16)[name = string("op_1980_cast_fp16")]; + tensor variance_65_axes_0 = const()[name = string("variance_65_axes_0"), val = tensor([1])]; + bool variance_65_keep_dims_0 = const()[name = string("variance_65_keep_dims_0"), val = bool(true)]; + tensor variance_65_cast_fp16 = reduce_mean(axes = variance_65_axes_0, keep_dims = variance_65_keep_dims_0, x = var_1980_cast_fp16)[name = string("variance_65_cast_fp16")]; + fp16 var_1983_to_fp16 = const()[name = string("op_1983_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1984_cast_fp16 = add(x = variance_65_cast_fp16, y = var_1983_to_fp16)[name = string("op_1984_cast_fp16")]; + fp32 var_1985_epsilon_0 = const()[name = string("op_1985_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1985_cast_fp16 = rsqrt(epsilon = var_1985_epsilon_0, x = var_1984_cast_fp16)[name = string("op_1985_cast_fp16")]; + tensor var_1986_cast_fp16 = mul(x = x_61_cast_fp16, y = var_1985_cast_fp16)[name = string("op_1986_cast_fp16")]; + tensor input_81_cast_fp16 = mul(x = var_1986_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_81_cast_fp16")]; + string q_49_pad_type_0 = const()[name = string("q_49_pad_type_0"), val = string("valid")]; + tensor q_49_strides_0 = const()[name = string("q_49_strides_0"), val = tensor([1, 1])]; + tensor q_49_pad_0 = const()[name = string("q_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_49_dilations_0 = const()[name = string("q_49_dilations_0"), val = tensor([1, 1])]; + int32 q_49_groups_0 = const()[name = string("q_49_groups_0"), val = int32(1)]; + tensor q_49_cast_fp16 = conv(dilations = q_49_dilations_0, groups = q_49_groups_0, pad = q_49_pad_0, pad_type = q_49_pad_type_0, strides = q_49_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_81_cast_fp16)[name = string("q_49_cast_fp16")]; + string k_49_pad_type_0 = const()[name = string("k_49_pad_type_0"), val = string("valid")]; + tensor k_49_strides_0 = const()[name = string("k_49_strides_0"), val = tensor([1, 1])]; + tensor k_49_pad_0 = const()[name = string("k_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_49_dilations_0 = const()[name = string("k_49_dilations_0"), val = tensor([1, 1])]; + int32 k_49_groups_0 = const()[name = string("k_49_groups_0"), val = int32(1)]; + tensor k_49_cast_fp16 = conv(dilations = k_49_dilations_0, groups = k_49_groups_0, pad = k_49_pad_0, pad_type = k_49_pad_type_0, strides = k_49_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_81_cast_fp16)[name = string("k_49_cast_fp16")]; + string v_17_pad_type_0 = const()[name = string("v_17_pad_type_0"), val = string("valid")]; + tensor v_17_strides_0 = const()[name = string("v_17_strides_0"), val = tensor([1, 1])]; + tensor v_17_pad_0 = const()[name = string("v_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_17_dilations_0 = const()[name = string("v_17_dilations_0"), val = tensor([1, 1])]; + int32 v_17_groups_0 = const()[name = string("v_17_groups_0"), val = int32(1)]; + tensor v_17_cast_fp16 = conv(dilations = v_17_dilations_0, groups = v_17_groups_0, pad = v_17_pad_0, pad_type = v_17_pad_type_0, strides = v_17_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_81_cast_fp16)[name = string("v_17_cast_fp16")]; + tensor var_2020 = const()[name = string("op_2020"), val = tensor([16, 128, 1, 1])]; + tensor x_63_cast_fp16 = reshape(shape = var_2020, x = q_49_cast_fp16)[name = string("x_63_cast_fp16")]; + tensor var_2023_cast_fp16 = mul(x = x_63_cast_fp16, y = x_63_cast_fp16)[name = string("op_2023_cast_fp16")]; + tensor variance_67_axes_0 = const()[name = string("variance_67_axes_0"), val = tensor([1])]; + bool variance_67_keep_dims_0 = const()[name = string("variance_67_keep_dims_0"), val = bool(true)]; + tensor variance_67_cast_fp16 = reduce_mean(axes = variance_67_axes_0, keep_dims = variance_67_keep_dims_0, x = var_2023_cast_fp16)[name = string("variance_67_cast_fp16")]; + fp16 var_2026_to_fp16 = const()[name = string("op_2026_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2027_cast_fp16 = add(x = variance_67_cast_fp16, y = var_2026_to_fp16)[name = string("op_2027_cast_fp16")]; + fp32 var_2028_epsilon_0 = const()[name = string("op_2028_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2028_cast_fp16 = rsqrt(epsilon = var_2028_epsilon_0, x = var_2027_cast_fp16)[name = string("op_2028_cast_fp16")]; + tensor var_2029_cast_fp16 = mul(x = x_63_cast_fp16, y = var_2028_cast_fp16)[name = string("op_2029_cast_fp16")]; + tensor q_51_cast_fp16 = mul(x = var_2029_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_51_cast_fp16")]; + tensor var_2031 = const()[name = string("op_2031"), val = tensor([8, 128, 1, 1])]; + tensor x_65_cast_fp16 = reshape(shape = var_2031, x = k_49_cast_fp16)[name = string("x_65_cast_fp16")]; + tensor var_2034_cast_fp16 = mul(x = x_65_cast_fp16, y = x_65_cast_fp16)[name = string("op_2034_cast_fp16")]; + tensor variance_69_axes_0 = const()[name = string("variance_69_axes_0"), val = tensor([1])]; + bool variance_69_keep_dims_0 = const()[name = string("variance_69_keep_dims_0"), val = bool(true)]; + tensor variance_69_cast_fp16 = reduce_mean(axes = variance_69_axes_0, keep_dims = variance_69_keep_dims_0, x = var_2034_cast_fp16)[name = string("variance_69_cast_fp16")]; + fp16 var_2037_to_fp16 = const()[name = string("op_2037_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2038_cast_fp16 = add(x = variance_69_cast_fp16, y = var_2037_to_fp16)[name = string("op_2038_cast_fp16")]; + fp32 var_2039_epsilon_0 = const()[name = string("op_2039_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2039_cast_fp16 = rsqrt(epsilon = var_2039_epsilon_0, x = var_2038_cast_fp16)[name = string("op_2039_cast_fp16")]; + tensor var_2040_cast_fp16 = mul(x = x_65_cast_fp16, y = var_2039_cast_fp16)[name = string("op_2040_cast_fp16")]; + tensor k_51_cast_fp16 = mul(x = var_2040_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_51_cast_fp16")]; + tensor var_2042 = const()[name = string("op_2042"), val = tensor([1, 16, 128, 1])]; + tensor z_33_cast_fp16 = reshape(shape = var_2042, x = q_51_cast_fp16)[name = string("z_33_cast_fp16")]; + tensor var_2044 = const()[name = string("op_2044"), val = tensor([1, 8, 128, 1])]; + tensor z_35_cast_fp16 = reshape(shape = var_2044, x = k_51_cast_fp16)[name = string("z_35_cast_fp16")]; + tensor z1_33_begin_0 = const()[name = string("z1_33_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_33_end_0 = const()[name = string("z1_33_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_33_end_mask_0 = const()[name = string("z1_33_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_33_cast_fp16 = slice_by_index(begin = z1_33_begin_0, end = z1_33_end_0, end_mask = z1_33_end_mask_0, x = z_33_cast_fp16)[name = string("z1_33_cast_fp16")]; + tensor z2_33_begin_0 = const()[name = string("z2_33_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_33_end_0 = const()[name = string("z2_33_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_33_end_mask_0 = const()[name = string("z2_33_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_33_cast_fp16 = slice_by_index(begin = z2_33_begin_0, end = z2_33_end_0, end_mask = z2_33_end_mask_0, x = z_33_cast_fp16)[name = string("z2_33_cast_fp16")]; + tensor var_2052_cast_fp16 = mul(x = z_33_cast_fp16, y = cos_11_to_fp16)[name = string("op_2052_cast_fp16")]; + fp16 const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2053_cast_fp16 = mul(x = z2_33_cast_fp16, y = const_18_promoted_to_fp16)[name = string("op_2053_cast_fp16")]; + bool var_2055_interleave_0 = const()[name = string("op_2055_interleave_0"), val = bool(false)]; + tensor var_2055_cast_fp16 = concat(axis = var_1961, interleave = var_2055_interleave_0, values = (var_2053_cast_fp16, z1_33_cast_fp16))[name = string("op_2055_cast_fp16")]; + tensor var_2056_cast_fp16 = mul(x = var_2055_cast_fp16, y = sin_11_to_fp16)[name = string("op_2056_cast_fp16")]; + tensor q_53_cast_fp16 = add(x = var_2052_cast_fp16, y = var_2056_cast_fp16)[name = string("q_53_cast_fp16")]; + tensor z1_35_begin_0 = const()[name = string("z1_35_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_35_end_0 = const()[name = string("z1_35_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_35_end_mask_0 = const()[name = string("z1_35_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_35_cast_fp16 = slice_by_index(begin = z1_35_begin_0, end = z1_35_end_0, end_mask = z1_35_end_mask_0, x = z_35_cast_fp16)[name = string("z1_35_cast_fp16")]; + tensor z2_35_begin_0 = const()[name = string("z2_35_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_35_end_0 = const()[name = string("z2_35_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_35_end_mask_0 = const()[name = string("z2_35_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_35_cast_fp16 = slice_by_index(begin = z2_35_begin_0, end = z2_35_end_0, end_mask = z2_35_end_mask_0, x = z_35_cast_fp16)[name = string("z2_35_cast_fp16")]; + tensor var_2064_cast_fp16 = mul(x = z_35_cast_fp16, y = cos_11_to_fp16)[name = string("op_2064_cast_fp16")]; + fp16 const_19_promoted_to_fp16 = const()[name = string("const_19_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2065_cast_fp16 = mul(x = z2_35_cast_fp16, y = const_19_promoted_to_fp16)[name = string("op_2065_cast_fp16")]; + bool var_2067_interleave_0 = const()[name = string("op_2067_interleave_0"), val = bool(false)]; + tensor var_2067_cast_fp16 = concat(axis = var_1961, interleave = var_2067_interleave_0, values = (var_2065_cast_fp16, z1_35_cast_fp16))[name = string("op_2067_cast_fp16")]; + tensor var_2068_cast_fp16 = mul(x = var_2067_cast_fp16, y = sin_11_to_fp16)[name = string("op_2068_cast_fp16")]; + tensor k_53_cast_fp16 = add(x = var_2064_cast_fp16, y = var_2068_cast_fp16)[name = string("k_53_cast_fp16")]; + tensor var_2070 = const()[name = string("op_2070"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_17_cast_fp16 = reshape(shape = var_2070, x = k_53_cast_fp16)[name = string("cur_key_17_cast_fp16")]; + tensor var_2072_to_fp16 = const()[name = string("op_2072_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174784)))]; + tensor var_2073_cast_fp16 = mul(x = key_cache_17_cast_fp16, y = var_2072_to_fp16)[name = string("op_2073_cast_fp16")]; + tensor upd_17_to_fp16 = const()[name = string("upd_17_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174912)))]; + tensor var_2074_cast_fp16 = mul(x = cur_key_17_cast_fp16, y = upd_17_to_fp16)[name = string("op_2074_cast_fp16")]; + tensor key_17_cast_fp16 = add(x = var_2073_cast_fp16, y = var_2074_cast_fp16)[name = string("key_17_cast_fp16")]; + tensor var_2076_to_fp16 = const()[name = string("op_2076_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174784)))]; + tensor var_2077_cast_fp16 = mul(x = value_cache_17_cast_fp16, y = var_2076_to_fp16)[name = string("op_2077_cast_fp16")]; + tensor var_2078_cast_fp16 = mul(x = v_17_cast_fp16, y = upd_17_to_fp16)[name = string("op_2078_cast_fp16")]; + tensor value_17_cast_fp16 = add(x = var_2077_cast_fp16, y = var_2078_cast_fp16)[name = string("value_17_cast_fp16")]; + tensor var_2080 = const()[name = string("op_2080"), val = tensor([1, 8, 128, 16])]; + tensor kh_33_cast_fp16 = reshape(shape = var_2080, x = key_17_cast_fp16)[name = string("kh_33_cast_fp16")]; + tensor var_2082 = const()[name = string("op_2082"), val = tensor([1, 8, 128, 16])]; + tensor vh_33_cast_fp16 = reshape(shape = var_2082, x = value_17_cast_fp16)[name = string("vh_33_cast_fp16")]; + tensor transpose_32_perm_0 = const()[name = string("transpose_32_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_16_reps_0 = const()[name = string("tile_16_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_32_cast_fp16 = transpose(perm = transpose_32_perm_0, x = kh_33_cast_fp16)[name = string("transpose_431")]; + tensor tile_16_cast_fp16 = tile(reps = tile_16_reps_0, x = transpose_32_cast_fp16)[name = string("tile_16_cast_fp16")]; + tensor concat_40 = const()[name = string("concat_40"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_32_cast_fp16 = reshape(shape = concat_40, x = tile_16_cast_fp16)[name = string("reshape_32_cast_fp16")]; + tensor transpose_33_perm_0 = const()[name = string("transpose_33_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_41 = const()[name = string("concat_41"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_33_cast_fp16 = transpose(perm = transpose_33_perm_0, x = reshape_32_cast_fp16)[name = string("transpose_430")]; + tensor reshape_33_cast_fp16 = reshape(shape = concat_41, x = transpose_33_cast_fp16)[name = string("reshape_33_cast_fp16")]; + tensor transpose_34_perm_0 = const()[name = string("transpose_34_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_17_reps_0 = const()[name = string("tile_17_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_34_cast_fp16 = transpose(perm = transpose_34_perm_0, x = vh_33_cast_fp16)[name = string("transpose_429")]; + tensor tile_17_cast_fp16 = tile(reps = tile_17_reps_0, x = transpose_34_cast_fp16)[name = string("tile_17_cast_fp16")]; + tensor concat_42 = const()[name = string("concat_42"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_34_cast_fp16 = reshape(shape = concat_42, x = tile_17_cast_fp16)[name = string("reshape_34_cast_fp16")]; + tensor transpose_35_perm_0 = const()[name = string("transpose_35_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_43 = const()[name = string("concat_43"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_35_cast_fp16 = transpose(perm = transpose_35_perm_0, x = reshape_34_cast_fp16)[name = string("transpose_428")]; + tensor reshape_35_cast_fp16 = reshape(shape = concat_43, x = transpose_35_cast_fp16)[name = string("reshape_35_cast_fp16")]; + fp16 var_2086_to_fp16 = const()[name = string("op_2086_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_2087_cast_fp16 = mul(x = q_53_cast_fp16, y = var_2086_to_fp16)[name = string("op_2087_cast_fp16")]; + tensor transpose_349_perm_0 = const()[name = string("transpose_349_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_33_transpose_x_1 = const()[name = string("w_33_transpose_x_1"), val = bool(true)]; + bool w_33_transpose_y_1 = const()[name = string("w_33_transpose_y_1"), val = bool(false)]; + tensor transpose_349_cast_fp16 = transpose(perm = transpose_349_perm_0, x = reshape_33_cast_fp16)[name = string("transpose_427")]; + tensor w_33_cast_fp16 = matmul(transpose_x = w_33_transpose_x_1, transpose_y = w_33_transpose_y_1, x = var_2087_cast_fp16, y = transpose_349_cast_fp16)[name = string("w_33_cast_fp16")]; + tensor pad_17_to_fp16 = const()[name = string("pad_17_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110175040)))]; + tensor var_2090_cast_fp16 = add(x = w_33_cast_fp16, y = pad_17_to_fp16)[name = string("op_2090_cast_fp16")]; + tensor w_35_cast_fp16 = softmax(axis = var_1965, x = var_2090_cast_fp16)[name = string("w_35_cast_fp16")]; + tensor transpose_350_perm_0 = const()[name = string("transpose_350_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_17_transpose_x_1 = const()[name = string("attn_17_transpose_x_1"), val = bool(false)]; + bool attn_17_transpose_y_1 = const()[name = string("attn_17_transpose_y_1"), val = bool(true)]; + tensor transpose_350_cast_fp16 = transpose(perm = transpose_350_perm_0, x = reshape_35_cast_fp16)[name = string("transpose_426")]; + tensor attn_17_cast_fp16 = matmul(transpose_x = attn_17_transpose_x_1, transpose_y = attn_17_transpose_y_1, x = transpose_350_cast_fp16, y = w_35_cast_fp16)[name = string("attn_17_cast_fp16")]; + tensor var_2094 = const()[name = string("op_2094"), val = tensor([1, 2048, 1, 1])]; + tensor input_83_cast_fp16 = reshape(shape = var_2094, x = attn_17_cast_fp16)[name = string("input_83_cast_fp16")]; + string attn_output_17_pad_type_0 = const()[name = string("attn_output_17_pad_type_0"), val = string("valid")]; + tensor attn_output_17_strides_0 = const()[name = string("attn_output_17_strides_0"), val = tensor([1, 1])]; + tensor attn_output_17_pad_0 = const()[name = string("attn_output_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_17_dilations_0 = const()[name = string("attn_output_17_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_17_groups_0 = const()[name = string("attn_output_17_groups_0"), val = int32(1)]; + tensor attn_output_17_cast_fp16 = conv(dilations = attn_output_17_dilations_0, groups = attn_output_17_groups_0, pad = attn_output_17_pad_0, pad_type = attn_output_17_pad_type_0, strides = attn_output_17_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_83_cast_fp16)[name = string("attn_output_17_cast_fp16")]; + tensor x_67_cast_fp16 = add(x = x_61_cast_fp16, y = attn_output_17_cast_fp16)[name = string("x_67_cast_fp16")]; + tensor var_2108_cast_fp16 = mul(x = x_67_cast_fp16, y = x_67_cast_fp16)[name = string("op_2108_cast_fp16")]; + tensor variance_71_axes_0 = const()[name = string("variance_71_axes_0"), val = tensor([1])]; + bool variance_71_keep_dims_0 = const()[name = string("variance_71_keep_dims_0"), val = bool(true)]; + tensor variance_71_cast_fp16 = reduce_mean(axes = variance_71_axes_0, keep_dims = variance_71_keep_dims_0, x = var_2108_cast_fp16)[name = string("variance_71_cast_fp16")]; + fp16 var_2111_to_fp16 = const()[name = string("op_2111_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2112_cast_fp16 = add(x = variance_71_cast_fp16, y = var_2111_to_fp16)[name = string("op_2112_cast_fp16")]; + fp32 var_2113_epsilon_0 = const()[name = string("op_2113_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2113_cast_fp16 = rsqrt(epsilon = var_2113_epsilon_0, x = var_2112_cast_fp16)[name = string("op_2113_cast_fp16")]; + tensor var_2114_cast_fp16 = mul(x = x_67_cast_fp16, y = var_2113_cast_fp16)[name = string("op_2114_cast_fp16")]; + tensor input_85_cast_fp16 = mul(x = var_2114_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_85_cast_fp16")]; + string input_87_pad_type_0 = const()[name = string("input_87_pad_type_0"), val = string("valid")]; + tensor input_87_strides_0 = const()[name = string("input_87_strides_0"), val = tensor([1, 1])]; + tensor input_87_pad_0 = const()[name = string("input_87_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_87_dilations_0 = const()[name = string("input_87_dilations_0"), val = tensor([1, 1])]; + int32 input_87_groups_0 = const()[name = string("input_87_groups_0"), val = int32(1)]; + tensor input_87_cast_fp16 = conv(dilations = input_87_dilations_0, groups = input_87_groups_0, pad = input_87_pad_0, pad_type = input_87_pad_type_0, strides = input_87_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_85_cast_fp16)[name = string("input_87_cast_fp16")]; + tensor var_2122_cast_fp16 = silu(x = input_87_cast_fp16)[name = string("op_2122_cast_fp16")]; + string var_2128_pad_type_0 = const()[name = string("op_2128_pad_type_0"), val = string("valid")]; + tensor var_2128_strides_0 = const()[name = string("op_2128_strides_0"), val = tensor([1, 1])]; + tensor var_2128_pad_0 = const()[name = string("op_2128_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2128_dilations_0 = const()[name = string("op_2128_dilations_0"), val = tensor([1, 1])]; + int32 var_2128_groups_0 = const()[name = string("op_2128_groups_0"), val = int32(1)]; + tensor var_2128_cast_fp16 = conv(dilations = var_2128_dilations_0, groups = var_2128_groups_0, pad = var_2128_pad_0, pad_type = var_2128_pad_type_0, strides = var_2128_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_85_cast_fp16)[name = string("op_2128_cast_fp16")]; + tensor input_89_cast_fp16 = mul(x = var_2122_cast_fp16, y = var_2128_cast_fp16)[name = string("input_89_cast_fp16")]; + string h_17_pad_type_0 = const()[name = string("h_17_pad_type_0"), val = string("valid")]; + tensor h_17_strides_0 = const()[name = string("h_17_strides_0"), val = tensor([1, 1])]; + tensor h_17_pad_0 = const()[name = string("h_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_17_dilations_0 = const()[name = string("h_17_dilations_0"), val = tensor([1, 1])]; + int32 h_17_groups_0 = const()[name = string("h_17_groups_0"), val = int32(1)]; + tensor h_17_cast_fp16 = conv(dilations = h_17_dilations_0, groups = h_17_groups_0, pad = h_17_pad_0, pad_type = h_17_pad_type_0, strides = h_17_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_89_cast_fp16)[name = string("h_17_cast_fp16")]; + tensor x_69_cast_fp16 = add(x = x_67_cast_fp16, y = h_17_cast_fp16)[name = string("x_69_cast_fp16")]; + tensor key_cache_19_begin_0 = const()[name = string("key_cache_19_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_19_end_0 = const()[name = string("key_cache_19_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_19_end_mask_0 = const()[name = string("key_cache_19_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_19_cast_fp16 = slice_by_index(begin = key_cache_19_begin_0, end = key_cache_19_end_0, end_mask = key_cache_19_end_mask_0, x = layer_key_caches_3_cast_fp16)[name = string("key_cache_19_cast_fp16")]; + tensor value_cache_19_begin_0 = const()[name = string("value_cache_19_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_19_end_0 = const()[name = string("value_cache_19_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_19_end_mask_0 = const()[name = string("value_cache_19_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_19_cast_fp16 = slice_by_index(begin = value_cache_19_begin_0, end = value_cache_19_end_0, end_mask = value_cache_19_end_mask_0, x = layer_value_caches_3_cast_fp16)[name = string("value_cache_19_cast_fp16")]; + int32 var_2181 = const()[name = string("op_2181"), val = int32(2)]; + int32 var_2185 = const()[name = string("op_2185"), val = int32(3)]; + tensor var_2200_cast_fp16 = mul(x = x_69_cast_fp16, y = x_69_cast_fp16)[name = string("op_2200_cast_fp16")]; + tensor variance_73_axes_0 = const()[name = string("variance_73_axes_0"), val = tensor([1])]; + bool variance_73_keep_dims_0 = const()[name = string("variance_73_keep_dims_0"), val = bool(true)]; + tensor variance_73_cast_fp16 = reduce_mean(axes = variance_73_axes_0, keep_dims = variance_73_keep_dims_0, x = var_2200_cast_fp16)[name = string("variance_73_cast_fp16")]; + fp16 var_2203_to_fp16 = const()[name = string("op_2203_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2204_cast_fp16 = add(x = variance_73_cast_fp16, y = var_2203_to_fp16)[name = string("op_2204_cast_fp16")]; + fp32 var_2205_epsilon_0 = const()[name = string("op_2205_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2205_cast_fp16 = rsqrt(epsilon = var_2205_epsilon_0, x = var_2204_cast_fp16)[name = string("op_2205_cast_fp16")]; + tensor var_2206_cast_fp16 = mul(x = x_69_cast_fp16, y = var_2205_cast_fp16)[name = string("op_2206_cast_fp16")]; + tensor input_91_cast_fp16 = mul(x = var_2206_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_91_cast_fp16")]; + string q_55_pad_type_0 = const()[name = string("q_55_pad_type_0"), val = string("valid")]; + tensor q_55_strides_0 = const()[name = string("q_55_strides_0"), val = tensor([1, 1])]; + tensor q_55_pad_0 = const()[name = string("q_55_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_55_dilations_0 = const()[name = string("q_55_dilations_0"), val = tensor([1, 1])]; + int32 q_55_groups_0 = const()[name = string("q_55_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62968704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65065920))))[name = string("layers_4_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor q_55_cast_fp16 = conv(dilations = q_55_dilations_0, groups = q_55_groups_0, pad = q_55_pad_0, pad_type = q_55_pad_type_0, strides = q_55_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_91_cast_fp16)[name = string("q_55_cast_fp16")]; + string k_55_pad_type_0 = const()[name = string("k_55_pad_type_0"), val = string("valid")]; + tensor k_55_strides_0 = const()[name = string("k_55_strides_0"), val = tensor([1, 1])]; + tensor k_55_pad_0 = const()[name = string("k_55_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_55_dilations_0 = const()[name = string("k_55_dilations_0"), val = tensor([1, 1])]; + int32 k_55_groups_0 = const()[name = string("k_55_groups_0"), val = int32(1)]; + tensor k_55_cast_fp16 = conv(dilations = k_55_dilations_0, groups = k_55_groups_0, pad = k_55_pad_0, pad_type = k_55_pad_type_0, strides = k_55_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_91_cast_fp16)[name = string("k_55_cast_fp16")]; + string v_19_pad_type_0 = const()[name = string("v_19_pad_type_0"), val = string("valid")]; + tensor v_19_strides_0 = const()[name = string("v_19_strides_0"), val = tensor([1, 1])]; + tensor v_19_pad_0 = const()[name = string("v_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_19_dilations_0 = const()[name = string("v_19_dilations_0"), val = tensor([1, 1])]; + int32 v_19_groups_0 = const()[name = string("v_19_groups_0"), val = int32(1)]; + tensor v_19_cast_fp16 = conv(dilations = v_19_dilations_0, groups = v_19_groups_0, pad = v_19_pad_0, pad_type = v_19_pad_type_0, strides = v_19_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_91_cast_fp16)[name = string("v_19_cast_fp16")]; + tensor var_2240 = const()[name = string("op_2240"), val = tensor([16, 128, 1, 1])]; + tensor x_71_cast_fp16 = reshape(shape = var_2240, x = q_55_cast_fp16)[name = string("x_71_cast_fp16")]; + tensor var_2243_cast_fp16 = mul(x = x_71_cast_fp16, y = x_71_cast_fp16)[name = string("op_2243_cast_fp16")]; + tensor variance_75_axes_0 = const()[name = string("variance_75_axes_0"), val = tensor([1])]; + bool variance_75_keep_dims_0 = const()[name = string("variance_75_keep_dims_0"), val = bool(true)]; + tensor variance_75_cast_fp16 = reduce_mean(axes = variance_75_axes_0, keep_dims = variance_75_keep_dims_0, x = var_2243_cast_fp16)[name = string("variance_75_cast_fp16")]; + fp16 var_2246_to_fp16 = const()[name = string("op_2246_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2247_cast_fp16 = add(x = variance_75_cast_fp16, y = var_2246_to_fp16)[name = string("op_2247_cast_fp16")]; + fp32 var_2248_epsilon_0 = const()[name = string("op_2248_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2248_cast_fp16 = rsqrt(epsilon = var_2248_epsilon_0, x = var_2247_cast_fp16)[name = string("op_2248_cast_fp16")]; + tensor var_2249_cast_fp16 = mul(x = x_71_cast_fp16, y = var_2248_cast_fp16)[name = string("op_2249_cast_fp16")]; + tensor layers_4_self_attn_q_norm_weight_to_fp16 = const()[name = string("layers_4_self_attn_q_norm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67164928)))]; + tensor q_57_cast_fp16 = mul(x = var_2249_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_57_cast_fp16")]; + tensor var_2251 = const()[name = string("op_2251"), val = tensor([8, 128, 1, 1])]; + tensor x_73_cast_fp16 = reshape(shape = var_2251, x = k_55_cast_fp16)[name = string("x_73_cast_fp16")]; + tensor var_2254_cast_fp16 = mul(x = x_73_cast_fp16, y = x_73_cast_fp16)[name = string("op_2254_cast_fp16")]; + tensor variance_77_axes_0 = const()[name = string("variance_77_axes_0"), val = tensor([1])]; + bool variance_77_keep_dims_0 = const()[name = string("variance_77_keep_dims_0"), val = bool(true)]; + tensor variance_77_cast_fp16 = reduce_mean(axes = variance_77_axes_0, keep_dims = variance_77_keep_dims_0, x = var_2254_cast_fp16)[name = string("variance_77_cast_fp16")]; + fp16 var_2257_to_fp16 = const()[name = string("op_2257_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2258_cast_fp16 = add(x = variance_77_cast_fp16, y = var_2257_to_fp16)[name = string("op_2258_cast_fp16")]; + fp32 var_2259_epsilon_0 = const()[name = string("op_2259_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2259_cast_fp16 = rsqrt(epsilon = var_2259_epsilon_0, x = var_2258_cast_fp16)[name = string("op_2259_cast_fp16")]; + tensor var_2260_cast_fp16 = mul(x = x_73_cast_fp16, y = var_2259_cast_fp16)[name = string("op_2260_cast_fp16")]; + tensor k_57_cast_fp16 = mul(x = var_2260_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_57_cast_fp16")]; + tensor var_2262 = const()[name = string("op_2262"), val = tensor([1, 16, 128, 1])]; + tensor z_37_cast_fp16 = reshape(shape = var_2262, x = q_57_cast_fp16)[name = string("z_37_cast_fp16")]; + tensor var_2264 = const()[name = string("op_2264"), val = tensor([1, 8, 128, 1])]; + tensor z_39_cast_fp16 = reshape(shape = var_2264, x = k_57_cast_fp16)[name = string("z_39_cast_fp16")]; + tensor z1_37_begin_0 = const()[name = string("z1_37_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_37_end_0 = const()[name = string("z1_37_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_37_end_mask_0 = const()[name = string("z1_37_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_37_cast_fp16 = slice_by_index(begin = z1_37_begin_0, end = z1_37_end_0, end_mask = z1_37_end_mask_0, x = z_37_cast_fp16)[name = string("z1_37_cast_fp16")]; + tensor z2_37_begin_0 = const()[name = string("z2_37_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_37_end_0 = const()[name = string("z2_37_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_37_end_mask_0 = const()[name = string("z2_37_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_37_cast_fp16 = slice_by_index(begin = z2_37_begin_0, end = z2_37_end_0, end_mask = z2_37_end_mask_0, x = z_37_cast_fp16)[name = string("z2_37_cast_fp16")]; + tensor var_2272_cast_fp16 = mul(x = z_37_cast_fp16, y = cos_11_to_fp16)[name = string("op_2272_cast_fp16")]; + fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2273_cast_fp16 = mul(x = z2_37_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_2273_cast_fp16")]; + bool var_2275_interleave_0 = const()[name = string("op_2275_interleave_0"), val = bool(false)]; + tensor var_2275_cast_fp16 = concat(axis = var_2181, interleave = var_2275_interleave_0, values = (var_2273_cast_fp16, z1_37_cast_fp16))[name = string("op_2275_cast_fp16")]; + tensor var_2276_cast_fp16 = mul(x = var_2275_cast_fp16, y = sin_11_to_fp16)[name = string("op_2276_cast_fp16")]; + tensor q_59_cast_fp16 = add(x = var_2272_cast_fp16, y = var_2276_cast_fp16)[name = string("q_59_cast_fp16")]; + tensor z1_39_begin_0 = const()[name = string("z1_39_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_39_end_0 = const()[name = string("z1_39_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_39_end_mask_0 = const()[name = string("z1_39_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_39_cast_fp16 = slice_by_index(begin = z1_39_begin_0, end = z1_39_end_0, end_mask = z1_39_end_mask_0, x = z_39_cast_fp16)[name = string("z1_39_cast_fp16")]; + tensor z2_39_begin_0 = const()[name = string("z2_39_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_39_end_0 = const()[name = string("z2_39_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_39_end_mask_0 = const()[name = string("z2_39_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_39_cast_fp16 = slice_by_index(begin = z2_39_begin_0, end = z2_39_end_0, end_mask = z2_39_end_mask_0, x = z_39_cast_fp16)[name = string("z2_39_cast_fp16")]; + tensor var_2284_cast_fp16 = mul(x = z_39_cast_fp16, y = cos_11_to_fp16)[name = string("op_2284_cast_fp16")]; + fp16 const_21_promoted_to_fp16 = const()[name = string("const_21_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2285_cast_fp16 = mul(x = z2_39_cast_fp16, y = const_21_promoted_to_fp16)[name = string("op_2285_cast_fp16")]; + bool var_2287_interleave_0 = const()[name = string("op_2287_interleave_0"), val = bool(false)]; + tensor var_2287_cast_fp16 = concat(axis = var_2181, interleave = var_2287_interleave_0, values = (var_2285_cast_fp16, z1_39_cast_fp16))[name = string("op_2287_cast_fp16")]; + tensor var_2288_cast_fp16 = mul(x = var_2287_cast_fp16, y = sin_11_to_fp16)[name = string("op_2288_cast_fp16")]; + tensor k_59_cast_fp16 = add(x = var_2284_cast_fp16, y = var_2288_cast_fp16)[name = string("k_59_cast_fp16")]; + tensor var_2290 = const()[name = string("op_2290"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_19_cast_fp16 = reshape(shape = var_2290, x = k_59_cast_fp16)[name = string("cur_key_19_cast_fp16")]; + tensor var_2292_to_fp16 = const()[name = string("op_2292_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174784)))]; + tensor var_2293_cast_fp16 = mul(x = key_cache_19_cast_fp16, y = var_2292_to_fp16)[name = string("op_2293_cast_fp16")]; + tensor upd_19_to_fp16 = const()[name = string("upd_19_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174912)))]; + tensor var_2294_cast_fp16 = mul(x = cur_key_19_cast_fp16, y = upd_19_to_fp16)[name = string("op_2294_cast_fp16")]; + tensor key_19_cast_fp16 = add(x = var_2293_cast_fp16, y = var_2294_cast_fp16)[name = string("key_19_cast_fp16")]; + tensor var_2296_to_fp16 = const()[name = string("op_2296_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110174784)))]; + tensor var_2297_cast_fp16 = mul(x = value_cache_19_cast_fp16, y = var_2296_to_fp16)[name = string("op_2297_cast_fp16")]; + tensor var_2298_cast_fp16 = mul(x = v_19_cast_fp16, y = upd_19_to_fp16)[name = string("op_2298_cast_fp16")]; + tensor value_19_cast_fp16 = add(x = var_2297_cast_fp16, y = var_2298_cast_fp16)[name = string("value_19_cast_fp16")]; + tensor var_2300 = const()[name = string("op_2300"), val = tensor([1, 8, 128, 16])]; + tensor kh_37_cast_fp16 = reshape(shape = var_2300, x = key_19_cast_fp16)[name = string("kh_37_cast_fp16")]; + tensor var_2302 = const()[name = string("op_2302"), val = tensor([1, 8, 128, 16])]; + tensor vh_37_cast_fp16 = reshape(shape = var_2302, x = value_19_cast_fp16)[name = string("vh_37_cast_fp16")]; + tensor transpose_36_perm_0 = const()[name = string("transpose_36_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_18_reps_0 = const()[name = string("tile_18_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_36_cast_fp16 = transpose(perm = transpose_36_perm_0, x = kh_37_cast_fp16)[name = string("transpose_425")]; + tensor tile_18_cast_fp16 = tile(reps = tile_18_reps_0, x = transpose_36_cast_fp16)[name = string("tile_18_cast_fp16")]; + tensor concat_44 = const()[name = string("concat_44"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_36_cast_fp16 = reshape(shape = concat_44, x = tile_18_cast_fp16)[name = string("reshape_36_cast_fp16")]; + tensor transpose_37_perm_0 = const()[name = string("transpose_37_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_45 = const()[name = string("concat_45"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_37_cast_fp16 = transpose(perm = transpose_37_perm_0, x = reshape_36_cast_fp16)[name = string("transpose_424")]; + tensor reshape_37_cast_fp16 = reshape(shape = concat_45, x = transpose_37_cast_fp16)[name = string("reshape_37_cast_fp16")]; + tensor transpose_38_perm_0 = const()[name = string("transpose_38_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_19_reps_0 = const()[name = string("tile_19_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_38_cast_fp16 = transpose(perm = transpose_38_perm_0, x = vh_37_cast_fp16)[name = string("transpose_423")]; + tensor tile_19_cast_fp16 = tile(reps = tile_19_reps_0, x = transpose_38_cast_fp16)[name = string("tile_19_cast_fp16")]; + tensor concat_46 = const()[name = string("concat_46"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_38_cast_fp16 = reshape(shape = concat_46, x = tile_19_cast_fp16)[name = string("reshape_38_cast_fp16")]; + tensor transpose_39_perm_0 = const()[name = string("transpose_39_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_47 = const()[name = string("concat_47"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_39_cast_fp16 = transpose(perm = transpose_39_perm_0, x = reshape_38_cast_fp16)[name = string("transpose_422")]; + tensor reshape_39_cast_fp16 = reshape(shape = concat_47, x = transpose_39_cast_fp16)[name = string("reshape_39_cast_fp16")]; + fp16 var_2306_to_fp16 = const()[name = string("op_2306_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_2307_cast_fp16 = mul(x = q_59_cast_fp16, y = var_2306_to_fp16)[name = string("op_2307_cast_fp16")]; + tensor transpose_353_perm_0 = const()[name = string("transpose_353_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_37_transpose_x_1 = const()[name = string("w_37_transpose_x_1"), val = bool(true)]; + bool w_37_transpose_y_1 = const()[name = string("w_37_transpose_y_1"), val = bool(false)]; + tensor transpose_353_cast_fp16 = transpose(perm = transpose_353_perm_0, x = reshape_37_cast_fp16)[name = string("transpose_421")]; + tensor w_37_cast_fp16 = matmul(transpose_x = w_37_transpose_x_1, transpose_y = w_37_transpose_y_1, x = var_2307_cast_fp16, y = transpose_353_cast_fp16)[name = string("w_37_cast_fp16")]; + tensor pad_19_to_fp16 = const()[name = string("pad_19_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110175040)))]; + tensor var_2310_cast_fp16 = add(x = w_37_cast_fp16, y = pad_19_to_fp16)[name = string("op_2310_cast_fp16")]; + tensor w_39_cast_fp16 = softmax(axis = var_2185, x = var_2310_cast_fp16)[name = string("w_39_cast_fp16")]; + tensor transpose_354_perm_0 = const()[name = string("transpose_354_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_19_transpose_x_1 = const()[name = string("attn_19_transpose_x_1"), val = bool(false)]; + bool attn_19_transpose_y_1 = const()[name = string("attn_19_transpose_y_1"), val = bool(true)]; + tensor transpose_354_cast_fp16 = transpose(perm = transpose_354_perm_0, x = reshape_39_cast_fp16)[name = string("transpose_420")]; + tensor attn_19_cast_fp16 = matmul(transpose_x = attn_19_transpose_x_1, transpose_y = attn_19_transpose_y_1, x = transpose_354_cast_fp16, y = w_39_cast_fp16)[name = string("attn_19_cast_fp16")]; + tensor var_2314 = const()[name = string("op_2314"), val = tensor([1, 2048, 1, 1])]; + tensor input_93_cast_fp16 = reshape(shape = var_2314, x = attn_19_cast_fp16)[name = string("input_93_cast_fp16")]; + string attn_output_19_pad_type_0 = const()[name = string("attn_output_19_pad_type_0"), val = string("valid")]; + tensor attn_output_19_strides_0 = const()[name = string("attn_output_19_strides_0"), val = tensor([1, 1])]; + tensor attn_output_19_pad_0 = const()[name = string("attn_output_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_19_dilations_0 = const()[name = string("attn_output_19_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_19_groups_0 = const()[name = string("attn_output_19_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67165568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69262784))))[name = string("layers_4_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor attn_output_19_cast_fp16 = conv(dilations = attn_output_19_dilations_0, groups = attn_output_19_groups_0, pad = attn_output_19_pad_0, pad_type = attn_output_19_pad_type_0, strides = attn_output_19_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_93_cast_fp16)[name = string("attn_output_19_cast_fp16")]; + tensor x_75_cast_fp16 = add(x = x_69_cast_fp16, y = attn_output_19_cast_fp16)[name = string("x_75_cast_fp16")]; + tensor var_2328_cast_fp16 = mul(x = x_75_cast_fp16, y = x_75_cast_fp16)[name = string("op_2328_cast_fp16")]; + tensor variance_79_axes_0 = const()[name = string("variance_79_axes_0"), val = tensor([1])]; + bool variance_79_keep_dims_0 = const()[name = string("variance_79_keep_dims_0"), val = bool(true)]; + tensor variance_79_cast_fp16 = reduce_mean(axes = variance_79_axes_0, keep_dims = variance_79_keep_dims_0, x = var_2328_cast_fp16)[name = string("variance_79_cast_fp16")]; + fp16 var_2331_to_fp16 = const()[name = string("op_2331_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2332_cast_fp16 = add(x = variance_79_cast_fp16, y = var_2331_to_fp16)[name = string("op_2332_cast_fp16")]; + fp32 var_2333_epsilon_0 = const()[name = string("op_2333_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2333_cast_fp16 = rsqrt(epsilon = var_2333_epsilon_0, x = var_2332_cast_fp16)[name = string("op_2333_cast_fp16")]; + tensor var_2334_cast_fp16 = mul(x = x_75_cast_fp16, y = var_2333_cast_fp16)[name = string("op_2334_cast_fp16")]; + tensor layers_4_post_attention_layernorm_weight_to_fp16 = const()[name = string("layers_4_post_attention_layernorm_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69263360)))]; + tensor input_95_cast_fp16 = mul(x = var_2334_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_95_cast_fp16")]; + string input_97_pad_type_0 = const()[name = string("input_97_pad_type_0"), val = string("valid")]; + tensor input_97_strides_0 = const()[name = string("input_97_strides_0"), val = tensor([1, 1])]; + tensor input_97_pad_0 = const()[name = string("input_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_97_dilations_0 = const()[name = string("input_97_dilations_0"), val = tensor([1, 1])]; + int32 input_97_groups_0 = const()[name = string("input_97_groups_0"), val = int32(1)]; + tensor layers_4_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69265472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72411264))))[name = string("layers_4_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_97_cast_fp16 = conv(dilations = input_97_dilations_0, groups = input_97_groups_0, pad = input_97_pad_0, pad_type = input_97_pad_type_0, strides = input_97_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_95_cast_fp16)[name = string("input_97_cast_fp16")]; + tensor var_2342_cast_fp16 = silu(x = input_97_cast_fp16)[name = string("op_2342_cast_fp16")]; + string var_2348_pad_type_0 = const()[name = string("op_2348_pad_type_0"), val = string("valid")]; + tensor var_2348_strides_0 = const()[name = string("op_2348_strides_0"), val = tensor([1, 1])]; + tensor var_2348_pad_0 = const()[name = string("op_2348_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2348_dilations_0 = const()[name = string("op_2348_dilations_0"), val = tensor([1, 1])]; + int32 var_2348_groups_0 = const()[name = string("op_2348_groups_0"), val = int32(1)]; + tensor layers_4_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72411840))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75557632))))[name = string("layers_4_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_2348_cast_fp16 = conv(dilations = var_2348_dilations_0, groups = var_2348_groups_0, pad = var_2348_pad_0, pad_type = var_2348_pad_type_0, strides = var_2348_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_95_cast_fp16)[name = string("op_2348_cast_fp16")]; + tensor input_99_cast_fp16 = mul(x = var_2342_cast_fp16, y = var_2348_cast_fp16)[name = string("input_99_cast_fp16")]; + string h_19_pad_type_0 = const()[name = string("h_19_pad_type_0"), val = string("valid")]; + tensor h_19_strides_0 = const()[name = string("h_19_strides_0"), val = tensor([1, 1])]; + tensor h_19_pad_0 = const()[name = string("h_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_19_dilations_0 = const()[name = string("h_19_dilations_0"), val = tensor([1, 1])]; + int32 h_19_groups_0 = const()[name = string("h_19_groups_0"), val = int32(1)]; + tensor layers_4_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75558208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(78704000))))[name = string("layers_4_mlp_down_proj_weight_to_fp16_palettized")]; + tensor h_19_cast_fp16 = conv(dilations = h_19_dilations_0, groups = h_19_groups_0, pad = h_19_pad_0, pad_type = h_19_pad_type_0, strides = h_19_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_99_cast_fp16)[name = string("h_19_cast_fp16")]; + tensor inputs_1_cast_fp16 = add(x = x_75_cast_fp16, y = h_19_cast_fp16)[name = string("inputs_1_cast_fp16")]; + int32 var_2376 = const()[name = string("op_2376"), val = int32(1)]; + bool layer_key_caches_5_interleave_0 = const()[name = string("layer_key_caches_5_interleave_0"), val = bool(false)]; + tensor layer_key_caches_5_cast_fp16 = concat(axis = var_2376, interleave = layer_key_caches_5_interleave_0, values = (key_11_cast_fp16, key_13_cast_fp16, key_15_cast_fp16, key_17_cast_fp16, key_19_cast_fp16))[name = string("layer_key_caches_5_cast_fp16")]; + int32 var_2379 = const()[name = string("op_2379"), val = int32(1)]; + bool layer_value_caches_5_interleave_0 = const()[name = string("layer_value_caches_5_interleave_0"), val = bool(false)]; + tensor layer_value_caches_5_cast_fp16 = concat(axis = var_2379, interleave = layer_value_caches_5_interleave_0, values = (value_11_cast_fp16, value_13_cast_fp16, value_15_cast_fp16, value_17_cast_fp16, value_19_cast_fp16))[name = string("layer_value_caches_5_cast_fp16")]; + tensor inputs_sq_1_cast_fp16 = mul(x = inputs_1_cast_fp16, y = inputs_1_cast_fp16)[name = string("inputs_sq_1_cast_fp16")]; + tensor variance_81_axes_0 = const()[name = string("variance_81_axes_0"), val = tensor([1])]; + bool variance_81_keep_dims_0 = const()[name = string("variance_81_keep_dims_0"), val = bool(true)]; + tensor variance_81_cast_fp16 = reduce_mean(axes = variance_81_axes_0, keep_dims = variance_81_keep_dims_0, x = inputs_sq_1_cast_fp16)[name = string("variance_81_cast_fp16")]; + fp16 var_2399_to_fp16 = const()[name = string("op_2399_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2400_cast_fp16 = add(x = variance_81_cast_fp16, y = var_2399_to_fp16)[name = string("op_2400_cast_fp16")]; + fp32 var_2401_epsilon_0 = const()[name = string("op_2401_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2401_cast_fp16 = rsqrt(epsilon = var_2401_epsilon_0, x = var_2400_cast_fp16)[name = string("op_2401_cast_fp16")]; + tensor hidden_states_1_cast_fp16 = mul(x = inputs_1_cast_fp16, y = var_2401_cast_fp16)[name = string("hidden_states_1_cast_fp16")]; + tensor w_41_to_fp16 = const()[name = string("w_41_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(78704576)))]; + tensor input_101_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_1_cast_fp16)[name = string("input_101_cast_fp16")]; + string logits_1_pad_type_0 = const()[name = string("logits_1_pad_type_0"), val = string("valid")]; + tensor logits_1_strides_0 = const()[name = string("logits_1_strides_0"), val = tensor([1, 1])]; + tensor logits_1_pad_0 = const()[name = string("logits_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_1_dilations_0 = const()[name = string("logits_1_dilations_0"), val = tensor([1, 1])]; + int32 logits_1_groups_0 = const()[name = string("logits_1_groups_0"), val = int32(1)]; + tensor lm_heads_0_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(78706688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80803904))))[name = string("lm_heads_0_weight_to_fp16_palettized")]; + tensor logits_1_cast_fp16 = conv(dilations = logits_1_dilations_0, groups = logits_1_groups_0, pad = logits_1_pad_0, pad_type = logits_1_pad_type_0, strides = logits_1_strides_0, weight = lm_heads_0_weight_to_fp16_palettized, x = input_101_cast_fp16)[name = string("logits_1_cast_fp16")]; + tensor var_2419 = const()[name = string("op_2419"), val = tensor([1, 2048])]; + tensor logits_3_cast_fp16 = reshape(shape = var_2419, x = logits_1_cast_fp16)[name = string("logits_3_cast_fp16")]; + tensor scaled_logits_1_cast_fp16 = real_div(x = logits_3_cast_fp16, y = temperature)[name = string("scaled_logits_1_cast_fp16")]; + int32 var_2429 = const()[name = string("op_2429"), val = int32(100)]; + int32 top_values_1_axis_0 = const()[name = string("top_values_1_axis_0"), val = int32(1)]; + bool top_values_1_ascending_0 = const()[name = string("top_values_1_ascending_0"), val = bool(false)]; + bool top_values_1_sort_0 = const()[name = string("top_values_1_sort_0"), val = bool(true)]; + bool top_values_1_return_indices_0 = const()[name = string("top_values_1_return_indices_0"), val = bool(true)]; + string top_values_1_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_1_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_1_cast_fp16_cast_uint16_0, tensor top_values_1_cast_fp16_cast_uint16_1 = topk(ascending = top_values_1_ascending_0, axis = top_values_1_axis_0, k = var_2429, output_indices_dtype = top_values_1_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_1_return_indices_0, sort = top_values_1_sort_0, x = scaled_logits_1_cast_fp16)[name = string("top_values_1_cast_fp16_cast_uint16")]; + tensor top_k_mask_candidate_ranks_to_fp16 = const()[name = string("top_k_mask_candidate_ranks_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110175168)))]; + string top_k_float_to_fp16_dtype_0 = const()[name = string("top_k_float_to_fp16_dtype_0"), val = string("fp16")]; + tensor top_k_to_fp16 = cast(dtype = top_k_float_to_fp16_dtype_0, x = top_k)[name = string("cast_16")]; + tensor var_2433_cast_fp16 = less(x = top_k_mask_candidate_ranks_to_fp16, y = top_k_to_fp16)[name = string("op_2433_cast_fp16")]; + string candidate_mask_1_to_fp16_dtype_0 = const()[name = string("candidate_mask_1_to_fp16_dtype_0"), val = string("fp16")]; + tensor var_2433_cast_fp16_to_fp16 = cast(dtype = candidate_mask_1_to_fp16_dtype_0, x = var_2433_cast_fp16)[name = string("cast_15")]; + tensor var_2435_cast_fp16 = mul(x = top_values_1_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_2435_cast_fp16")]; + fp16 var_2423_to_fp16 = const()[name = string("op_2423_to_fp16"), val = fp16(0x1p+0)]; + tensor var_2436_cast_fp16 = sub(x = var_2423_to_fp16, y = var_2433_cast_fp16_to_fp16)[name = string("op_2436_cast_fp16")]; + fp16 var_2437_to_fp16 = const()[name = string("op_2437_to_fp16"), val = fp16(0x1.d4cp+14)]; + tensor var_2438_cast_fp16 = mul(x = var_2436_cast_fp16, y = var_2437_to_fp16)[name = string("op_2438_cast_fp16")]; + tensor var_2439_cast_fp16 = add(x = var_2435_cast_fp16, y = var_2438_cast_fp16)[name = string("op_2439_cast_fp16")]; + tensor reduce_min_0_axes_0 = const()[name = string("reduce_min_0_axes_0"), val = tensor([1])]; + bool reduce_min_0_keep_dims_0 = const()[name = string("reduce_min_0_keep_dims_0"), val = bool(true)]; + tensor reduce_min_0_cast_fp16 = reduce_min(axes = reduce_min_0_axes_0, keep_dims = reduce_min_0_keep_dims_0, x = var_2439_cast_fp16)[name = string("reduce_min_0_cast_fp16")]; + tensor var_2442_cast_fp16 = greater_equal(x = scaled_logits_1_cast_fp16, y = reduce_min_0_cast_fp16)[name = string("op_2442_cast_fp16")]; + fp16 var_2443_value_0_to_fp16 = const()[name = string("op_2443_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_2443_cast_fp16 = fill_like(ref_tensor = scaled_logits_1_cast_fp16, value = var_2443_value_0_to_fp16)[name = string("op_2443_cast_fp16")]; + tensor masked_logits_1_cast_fp16 = select(a = scaled_logits_1_cast_fp16, b = var_2443_cast_fp16, cond = var_2442_cast_fp16)[name = string("masked_logits_1_cast_fp16")]; + tensor var_2447_begin_0 = const()[name = string("op_2447_begin_0"), val = tensor([0, 0])]; + tensor var_2447_end_0 = const()[name = string("op_2447_end_0"), val = tensor([1, 2048])]; + tensor var_2447_end_mask_0 = const()[name = string("op_2447_end_mask_0"), val = tensor([false, true])]; + tensor var_2447_squeeze_mask_0 = const()[name = string("op_2447_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_2447_cast_fp16 = slice_by_index(begin = var_2447_begin_0, end = var_2447_end_0, end_mask = var_2447_end_mask_0, squeeze_mask = var_2447_squeeze_mask_0, x = gumbel)[name = string("op_2447_cast_fp16")]; + tensor var_2450 = const()[name = string("op_2450"), val = tensor([1, 2048])]; + tensor var_2451_cast_fp16 = reshape(shape = var_2450, x = var_2447_cast_fp16)[name = string("op_2451_cast_fp16")]; + tensor noisy_logits_1_cast_fp16 = add(x = masked_logits_1_cast_fp16, y = var_2451_cast_fp16)[name = string("noisy_logits_1_cast_fp16")]; + int32 code_1_axis_0 = const()[name = string("code_1_axis_0"), val = int32(1)]; + bool code_1_keep_dims_0 = const()[name = string("code_1_keep_dims_0"), val = bool(false)]; + string code_1_output_dtype_0 = const()[name = string("code_1_output_dtype_0"), val = string("int32")]; + tensor code_1_cast_fp16 = reduce_argmax(axis = code_1_axis_0, keep_dims = code_1_keep_dims_0, output_dtype = code_1_output_dtype_0, x = noisy_logits_1_cast_fp16)[name = string("code_1_cast_fp16")]; + int32 code_embed_1_axis_0 = const()[name = string("code_embed_1_axis_0"), val = int32(0)]; + int32 code_embed_1_batch_dims_0 = const()[name = string("code_embed_1_batch_dims_0"), val = int32(0)]; + bool code_embed_1_validate_indices_0 = const()[name = string("code_embed_1_validate_indices_0"), val = bool(false)]; + tensor codec_embedding_embedding_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110175488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141632832))))[name = string("codec_embedding_embedding_weight_to_fp16_palettized")]; + string code_1_cast_fp16_to_uint16_dtype_0 = const()[name = string("code_1_cast_fp16_to_uint16_dtype_0"), val = string("uint16")]; + tensor code_1_cast_fp16_to_uint16 = cast(dtype = code_1_cast_fp16_to_uint16_dtype_0, x = code_1_cast_fp16)[name = string("cast_14")]; + tensor code_embed_1_cast_fp16_cast_uint16 = gather(axis = code_embed_1_axis_0, batch_dims = code_embed_1_batch_dims_0, indices = code_1_cast_fp16_to_uint16, validate_indices = code_embed_1_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_1_cast_fp16_cast_uint16")]; + tensor var_2466 = const()[name = string("op_2466"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_3_cast_fp16 = reshape(shape = var_2466, x = code_embed_1_cast_fp16_cast_uint16)[name = string("code_embed_3_cast_fp16")]; + tensor key_cache_21_begin_0 = const()[name = string("key_cache_21_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_21_end_0 = const()[name = string("key_cache_21_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_21_end_mask_0 = const()[name = string("key_cache_21_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_21_cast_fp16 = slice_by_index(begin = key_cache_21_begin_0, end = key_cache_21_end_0, end_mask = key_cache_21_end_mask_0, x = layer_key_caches_5_cast_fp16)[name = string("key_cache_21_cast_fp16")]; + tensor value_cache_21_begin_0 = const()[name = string("value_cache_21_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_21_end_0 = const()[name = string("value_cache_21_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_21_end_mask_0 = const()[name = string("value_cache_21_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_21_cast_fp16 = slice_by_index(begin = value_cache_21_begin_0, end = value_cache_21_end_0, end_mask = value_cache_21_end_mask_0, x = layer_value_caches_5_cast_fp16)[name = string("value_cache_21_cast_fp16")]; + int32 var_2565 = const()[name = string("op_2565"), val = int32(2)]; + int32 var_2569 = const()[name = string("op_2569"), val = int32(3)]; + tensor var_2584_cast_fp16 = mul(x = code_embed_3_cast_fp16, y = code_embed_3_cast_fp16)[name = string("op_2584_cast_fp16")]; + tensor variance_83_axes_0 = const()[name = string("variance_83_axes_0"), val = tensor([1])]; + bool variance_83_keep_dims_0 = const()[name = string("variance_83_keep_dims_0"), val = bool(true)]; + tensor variance_83_cast_fp16 = reduce_mean(axes = variance_83_axes_0, keep_dims = variance_83_keep_dims_0, x = var_2584_cast_fp16)[name = string("variance_83_cast_fp16")]; + fp16 var_2587_to_fp16 = const()[name = string("op_2587_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2588_cast_fp16 = add(x = variance_83_cast_fp16, y = var_2587_to_fp16)[name = string("op_2588_cast_fp16")]; + fp32 var_2589_epsilon_0 = const()[name = string("op_2589_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2589_cast_fp16 = rsqrt(epsilon = var_2589_epsilon_0, x = var_2588_cast_fp16)[name = string("op_2589_cast_fp16")]; + tensor var_2590_cast_fp16 = mul(x = code_embed_3_cast_fp16, y = var_2589_cast_fp16)[name = string("op_2590_cast_fp16")]; + tensor input_105_cast_fp16 = mul(x = var_2590_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_105_cast_fp16")]; + string q_61_pad_type_0 = const()[name = string("q_61_pad_type_0"), val = string("valid")]; + tensor q_61_strides_0 = const()[name = string("q_61_strides_0"), val = tensor([1, 1])]; + tensor q_61_pad_0 = const()[name = string("q_61_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_61_dilations_0 = const()[name = string("q_61_dilations_0"), val = tensor([1, 1])]; + int32 q_61_groups_0 = const()[name = string("q_61_groups_0"), val = int32(1)]; + tensor q_61_cast_fp16 = conv(dilations = q_61_dilations_0, groups = q_61_groups_0, pad = q_61_pad_0, pad_type = q_61_pad_type_0, strides = q_61_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_105_cast_fp16)[name = string("q_61_cast_fp16")]; + string k_61_pad_type_0 = const()[name = string("k_61_pad_type_0"), val = string("valid")]; + tensor k_61_strides_0 = const()[name = string("k_61_strides_0"), val = tensor([1, 1])]; + tensor k_61_pad_0 = const()[name = string("k_61_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_61_dilations_0 = const()[name = string("k_61_dilations_0"), val = tensor([1, 1])]; + int32 k_61_groups_0 = const()[name = string("k_61_groups_0"), val = int32(1)]; + tensor k_61_cast_fp16 = conv(dilations = k_61_dilations_0, groups = k_61_groups_0, pad = k_61_pad_0, pad_type = k_61_pad_type_0, strides = k_61_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_105_cast_fp16)[name = string("k_61_cast_fp16")]; + string v_21_pad_type_0 = const()[name = string("v_21_pad_type_0"), val = string("valid")]; + tensor v_21_strides_0 = const()[name = string("v_21_strides_0"), val = tensor([1, 1])]; + tensor v_21_pad_0 = const()[name = string("v_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_21_dilations_0 = const()[name = string("v_21_dilations_0"), val = tensor([1, 1])]; + int32 v_21_groups_0 = const()[name = string("v_21_groups_0"), val = int32(1)]; + tensor v_21_cast_fp16 = conv(dilations = v_21_dilations_0, groups = v_21_groups_0, pad = v_21_pad_0, pad_type = v_21_pad_type_0, strides = v_21_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_105_cast_fp16)[name = string("v_21_cast_fp16")]; + tensor var_2624 = const()[name = string("op_2624"), val = tensor([16, 128, 1, 1])]; + tensor x_77_cast_fp16 = reshape(shape = var_2624, x = q_61_cast_fp16)[name = string("x_77_cast_fp16")]; + tensor var_2627_cast_fp16 = mul(x = x_77_cast_fp16, y = x_77_cast_fp16)[name = string("op_2627_cast_fp16")]; + tensor variance_85_axes_0 = const()[name = string("variance_85_axes_0"), val = tensor([1])]; + bool variance_85_keep_dims_0 = const()[name = string("variance_85_keep_dims_0"), val = bool(true)]; + tensor variance_85_cast_fp16 = reduce_mean(axes = variance_85_axes_0, keep_dims = variance_85_keep_dims_0, x = var_2627_cast_fp16)[name = string("variance_85_cast_fp16")]; + fp16 var_2630_to_fp16 = const()[name = string("op_2630_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2631_cast_fp16 = add(x = variance_85_cast_fp16, y = var_2630_to_fp16)[name = string("op_2631_cast_fp16")]; + fp32 var_2632_epsilon_0 = const()[name = string("op_2632_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2632_cast_fp16 = rsqrt(epsilon = var_2632_epsilon_0, x = var_2631_cast_fp16)[name = string("op_2632_cast_fp16")]; + tensor var_2633_cast_fp16 = mul(x = x_77_cast_fp16, y = var_2632_cast_fp16)[name = string("op_2633_cast_fp16")]; + tensor q_63_cast_fp16 = mul(x = var_2633_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_63_cast_fp16")]; + tensor var_2635 = const()[name = string("op_2635"), val = tensor([8, 128, 1, 1])]; + tensor x_79_cast_fp16 = reshape(shape = var_2635, x = k_61_cast_fp16)[name = string("x_79_cast_fp16")]; + tensor var_2638_cast_fp16 = mul(x = x_79_cast_fp16, y = x_79_cast_fp16)[name = string("op_2638_cast_fp16")]; + tensor variance_87_axes_0 = const()[name = string("variance_87_axes_0"), val = tensor([1])]; + bool variance_87_keep_dims_0 = const()[name = string("variance_87_keep_dims_0"), val = bool(true)]; + tensor variance_87_cast_fp16 = reduce_mean(axes = variance_87_axes_0, keep_dims = variance_87_keep_dims_0, x = var_2638_cast_fp16)[name = string("variance_87_cast_fp16")]; + fp16 var_2641_to_fp16 = const()[name = string("op_2641_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2642_cast_fp16 = add(x = variance_87_cast_fp16, y = var_2641_to_fp16)[name = string("op_2642_cast_fp16")]; + fp32 var_2643_epsilon_0 = const()[name = string("op_2643_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2643_cast_fp16 = rsqrt(epsilon = var_2643_epsilon_0, x = var_2642_cast_fp16)[name = string("op_2643_cast_fp16")]; + tensor var_2644_cast_fp16 = mul(x = x_79_cast_fp16, y = var_2643_cast_fp16)[name = string("op_2644_cast_fp16")]; + tensor k_63_cast_fp16 = mul(x = var_2644_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_63_cast_fp16")]; + tensor var_2646 = const()[name = string("op_2646"), val = tensor([1, 16, 128, 1])]; + tensor z_41_cast_fp16 = reshape(shape = var_2646, x = q_63_cast_fp16)[name = string("z_41_cast_fp16")]; + tensor var_2648 = const()[name = string("op_2648"), val = tensor([1, 8, 128, 1])]; + tensor z_43_cast_fp16 = reshape(shape = var_2648, x = k_63_cast_fp16)[name = string("z_43_cast_fp16")]; + tensor z1_41_begin_0 = const()[name = string("z1_41_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_41_end_0 = const()[name = string("z1_41_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_41_end_mask_0 = const()[name = string("z1_41_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_41_cast_fp16 = slice_by_index(begin = z1_41_begin_0, end = z1_41_end_0, end_mask = z1_41_end_mask_0, x = z_41_cast_fp16)[name = string("z1_41_cast_fp16")]; + tensor z2_41_begin_0 = const()[name = string("z2_41_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_41_end_0 = const()[name = string("z2_41_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_41_end_mask_0 = const()[name = string("z2_41_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_41_cast_fp16 = slice_by_index(begin = z2_41_begin_0, end = z2_41_end_0, end_mask = z2_41_end_mask_0, x = z_41_cast_fp16)[name = string("z2_41_cast_fp16")]; + tensor cos_21_to_fp16 = const()[name = string("cos_21_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141633408)))]; + tensor var_2656_cast_fp16 = mul(x = z_41_cast_fp16, y = cos_21_to_fp16)[name = string("op_2656_cast_fp16")]; + fp16 const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2657_cast_fp16 = mul(x = z2_41_cast_fp16, y = const_23_promoted_to_fp16)[name = string("op_2657_cast_fp16")]; + bool var_2659_interleave_0 = const()[name = string("op_2659_interleave_0"), val = bool(false)]; + tensor var_2659_cast_fp16 = concat(axis = var_2565, interleave = var_2659_interleave_0, values = (var_2657_cast_fp16, z1_41_cast_fp16))[name = string("op_2659_cast_fp16")]; + tensor sin_21_to_fp16 = const()[name = string("sin_21_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141633728)))]; + tensor var_2660_cast_fp16 = mul(x = var_2659_cast_fp16, y = sin_21_to_fp16)[name = string("op_2660_cast_fp16")]; + tensor q_65_cast_fp16 = add(x = var_2656_cast_fp16, y = var_2660_cast_fp16)[name = string("q_65_cast_fp16")]; + tensor z1_43_begin_0 = const()[name = string("z1_43_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_43_end_0 = const()[name = string("z1_43_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_43_end_mask_0 = const()[name = string("z1_43_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_43_cast_fp16 = slice_by_index(begin = z1_43_begin_0, end = z1_43_end_0, end_mask = z1_43_end_mask_0, x = z_43_cast_fp16)[name = string("z1_43_cast_fp16")]; + tensor z2_43_begin_0 = const()[name = string("z2_43_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_43_end_0 = const()[name = string("z2_43_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_43_end_mask_0 = const()[name = string("z2_43_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_43_cast_fp16 = slice_by_index(begin = z2_43_begin_0, end = z2_43_end_0, end_mask = z2_43_end_mask_0, x = z_43_cast_fp16)[name = string("z2_43_cast_fp16")]; + tensor var_2668_cast_fp16 = mul(x = z_43_cast_fp16, y = cos_21_to_fp16)[name = string("op_2668_cast_fp16")]; + fp16 const_24_promoted_to_fp16 = const()[name = string("const_24_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2669_cast_fp16 = mul(x = z2_43_cast_fp16, y = const_24_promoted_to_fp16)[name = string("op_2669_cast_fp16")]; + bool var_2671_interleave_0 = const()[name = string("op_2671_interleave_0"), val = bool(false)]; + tensor var_2671_cast_fp16 = concat(axis = var_2565, interleave = var_2671_interleave_0, values = (var_2669_cast_fp16, z1_43_cast_fp16))[name = string("op_2671_cast_fp16")]; + tensor var_2672_cast_fp16 = mul(x = var_2671_cast_fp16, y = sin_21_to_fp16)[name = string("op_2672_cast_fp16")]; + tensor k_65_cast_fp16 = add(x = var_2668_cast_fp16, y = var_2672_cast_fp16)[name = string("k_65_cast_fp16")]; + tensor var_2674 = const()[name = string("op_2674"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_21_cast_fp16 = reshape(shape = var_2674, x = k_65_cast_fp16)[name = string("cur_key_21_cast_fp16")]; + tensor var_2676_to_fp16 = const()[name = string("op_2676_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634048)))]; + tensor var_2677_cast_fp16 = mul(x = key_cache_21_cast_fp16, y = var_2676_to_fp16)[name = string("op_2677_cast_fp16")]; + tensor upd_21_to_fp16 = const()[name = string("upd_21_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634176)))]; + tensor var_2678_cast_fp16 = mul(x = cur_key_21_cast_fp16, y = upd_21_to_fp16)[name = string("op_2678_cast_fp16")]; + tensor key_21_cast_fp16 = add(x = var_2677_cast_fp16, y = var_2678_cast_fp16)[name = string("key_21_cast_fp16")]; + tensor var_2680_to_fp16 = const()[name = string("op_2680_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634048)))]; + tensor var_2681_cast_fp16 = mul(x = value_cache_21_cast_fp16, y = var_2680_to_fp16)[name = string("op_2681_cast_fp16")]; + tensor var_2682_cast_fp16 = mul(x = v_21_cast_fp16, y = upd_21_to_fp16)[name = string("op_2682_cast_fp16")]; + tensor value_21_cast_fp16 = add(x = var_2681_cast_fp16, y = var_2682_cast_fp16)[name = string("value_21_cast_fp16")]; + tensor var_2684 = const()[name = string("op_2684"), val = tensor([1, 8, 128, 16])]; + tensor kh_41_cast_fp16 = reshape(shape = var_2684, x = key_21_cast_fp16)[name = string("kh_41_cast_fp16")]; + tensor var_2686 = const()[name = string("op_2686"), val = tensor([1, 8, 128, 16])]; + tensor vh_41_cast_fp16 = reshape(shape = var_2686, x = value_21_cast_fp16)[name = string("vh_41_cast_fp16")]; + tensor transpose_40_perm_0 = const()[name = string("transpose_40_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_20_reps_0 = const()[name = string("tile_20_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_40_cast_fp16 = transpose(perm = transpose_40_perm_0, x = kh_41_cast_fp16)[name = string("transpose_419")]; + tensor tile_20_cast_fp16 = tile(reps = tile_20_reps_0, x = transpose_40_cast_fp16)[name = string("tile_20_cast_fp16")]; + tensor concat_53 = const()[name = string("concat_53"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_40_cast_fp16 = reshape(shape = concat_53, x = tile_20_cast_fp16)[name = string("reshape_40_cast_fp16")]; + tensor transpose_41_perm_0 = const()[name = string("transpose_41_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_54 = const()[name = string("concat_54"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_41_cast_fp16 = transpose(perm = transpose_41_perm_0, x = reshape_40_cast_fp16)[name = string("transpose_418")]; + tensor reshape_41_cast_fp16 = reshape(shape = concat_54, x = transpose_41_cast_fp16)[name = string("reshape_41_cast_fp16")]; + tensor transpose_42_perm_0 = const()[name = string("transpose_42_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_21_reps_0 = const()[name = string("tile_21_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_42_cast_fp16 = transpose(perm = transpose_42_perm_0, x = vh_41_cast_fp16)[name = string("transpose_417")]; + tensor tile_21_cast_fp16 = tile(reps = tile_21_reps_0, x = transpose_42_cast_fp16)[name = string("tile_21_cast_fp16")]; + tensor concat_55 = const()[name = string("concat_55"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_42_cast_fp16 = reshape(shape = concat_55, x = tile_21_cast_fp16)[name = string("reshape_42_cast_fp16")]; + tensor transpose_43_perm_0 = const()[name = string("transpose_43_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_56 = const()[name = string("concat_56"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_43_cast_fp16 = transpose(perm = transpose_43_perm_0, x = reshape_42_cast_fp16)[name = string("transpose_416")]; + tensor reshape_43_cast_fp16 = reshape(shape = concat_56, x = transpose_43_cast_fp16)[name = string("reshape_43_cast_fp16")]; + fp16 var_2690_to_fp16 = const()[name = string("op_2690_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_2691_cast_fp16 = mul(x = q_65_cast_fp16, y = var_2690_to_fp16)[name = string("op_2691_cast_fp16")]; + tensor transpose_357_perm_0 = const()[name = string("transpose_357_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_43_transpose_x_1 = const()[name = string("w_43_transpose_x_1"), val = bool(true)]; + bool w_43_transpose_y_1 = const()[name = string("w_43_transpose_y_1"), val = bool(false)]; + tensor transpose_357_cast_fp16 = transpose(perm = transpose_357_perm_0, x = reshape_41_cast_fp16)[name = string("transpose_415")]; + tensor w_43_cast_fp16 = matmul(transpose_x = w_43_transpose_x_1, transpose_y = w_43_transpose_y_1, x = var_2691_cast_fp16, y = transpose_357_cast_fp16)[name = string("w_43_cast_fp16")]; + tensor pad_21_to_fp16 = const()[name = string("pad_21_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634304)))]; + tensor var_2694_cast_fp16 = add(x = w_43_cast_fp16, y = pad_21_to_fp16)[name = string("op_2694_cast_fp16")]; + tensor w_45_cast_fp16 = softmax(axis = var_2569, x = var_2694_cast_fp16)[name = string("w_45_cast_fp16")]; + tensor transpose_358_perm_0 = const()[name = string("transpose_358_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_21_transpose_x_1 = const()[name = string("attn_21_transpose_x_1"), val = bool(false)]; + bool attn_21_transpose_y_1 = const()[name = string("attn_21_transpose_y_1"), val = bool(true)]; + tensor transpose_358_cast_fp16 = transpose(perm = transpose_358_perm_0, x = reshape_43_cast_fp16)[name = string("transpose_414")]; + tensor attn_21_cast_fp16 = matmul(transpose_x = attn_21_transpose_x_1, transpose_y = attn_21_transpose_y_1, x = transpose_358_cast_fp16, y = w_45_cast_fp16)[name = string("attn_21_cast_fp16")]; + tensor var_2698 = const()[name = string("op_2698"), val = tensor([1, 2048, 1, 1])]; + tensor input_107_cast_fp16 = reshape(shape = var_2698, x = attn_21_cast_fp16)[name = string("input_107_cast_fp16")]; + string attn_output_21_pad_type_0 = const()[name = string("attn_output_21_pad_type_0"), val = string("valid")]; + tensor attn_output_21_strides_0 = const()[name = string("attn_output_21_strides_0"), val = tensor([1, 1])]; + tensor attn_output_21_pad_0 = const()[name = string("attn_output_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_21_dilations_0 = const()[name = string("attn_output_21_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_21_groups_0 = const()[name = string("attn_output_21_groups_0"), val = int32(1)]; + tensor attn_output_21_cast_fp16 = conv(dilations = attn_output_21_dilations_0, groups = attn_output_21_groups_0, pad = attn_output_21_pad_0, pad_type = attn_output_21_pad_type_0, strides = attn_output_21_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_107_cast_fp16)[name = string("attn_output_21_cast_fp16")]; + tensor x_81_cast_fp16 = add(x = code_embed_3_cast_fp16, y = attn_output_21_cast_fp16)[name = string("x_81_cast_fp16")]; + tensor var_2712_cast_fp16 = mul(x = x_81_cast_fp16, y = x_81_cast_fp16)[name = string("op_2712_cast_fp16")]; + tensor variance_89_axes_0 = const()[name = string("variance_89_axes_0"), val = tensor([1])]; + bool variance_89_keep_dims_0 = const()[name = string("variance_89_keep_dims_0"), val = bool(true)]; + tensor variance_89_cast_fp16 = reduce_mean(axes = variance_89_axes_0, keep_dims = variance_89_keep_dims_0, x = var_2712_cast_fp16)[name = string("variance_89_cast_fp16")]; + fp16 var_2715_to_fp16 = const()[name = string("op_2715_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2716_cast_fp16 = add(x = variance_89_cast_fp16, y = var_2715_to_fp16)[name = string("op_2716_cast_fp16")]; + fp32 var_2717_epsilon_0 = const()[name = string("op_2717_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2717_cast_fp16 = rsqrt(epsilon = var_2717_epsilon_0, x = var_2716_cast_fp16)[name = string("op_2717_cast_fp16")]; + tensor var_2718_cast_fp16 = mul(x = x_81_cast_fp16, y = var_2717_cast_fp16)[name = string("op_2718_cast_fp16")]; + tensor input_109_cast_fp16 = mul(x = var_2718_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_109_cast_fp16")]; + string input_111_pad_type_0 = const()[name = string("input_111_pad_type_0"), val = string("valid")]; + tensor input_111_strides_0 = const()[name = string("input_111_strides_0"), val = tensor([1, 1])]; + tensor input_111_pad_0 = const()[name = string("input_111_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_111_dilations_0 = const()[name = string("input_111_dilations_0"), val = tensor([1, 1])]; + int32 input_111_groups_0 = const()[name = string("input_111_groups_0"), val = int32(1)]; + tensor input_111_cast_fp16 = conv(dilations = input_111_dilations_0, groups = input_111_groups_0, pad = input_111_pad_0, pad_type = input_111_pad_type_0, strides = input_111_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_109_cast_fp16)[name = string("input_111_cast_fp16")]; + tensor var_2726_cast_fp16 = silu(x = input_111_cast_fp16)[name = string("op_2726_cast_fp16")]; + string var_2732_pad_type_0 = const()[name = string("op_2732_pad_type_0"), val = string("valid")]; + tensor var_2732_strides_0 = const()[name = string("op_2732_strides_0"), val = tensor([1, 1])]; + tensor var_2732_pad_0 = const()[name = string("op_2732_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2732_dilations_0 = const()[name = string("op_2732_dilations_0"), val = tensor([1, 1])]; + int32 var_2732_groups_0 = const()[name = string("op_2732_groups_0"), val = int32(1)]; + tensor var_2732_cast_fp16 = conv(dilations = var_2732_dilations_0, groups = var_2732_groups_0, pad = var_2732_pad_0, pad_type = var_2732_pad_type_0, strides = var_2732_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_109_cast_fp16)[name = string("op_2732_cast_fp16")]; + tensor input_113_cast_fp16 = mul(x = var_2726_cast_fp16, y = var_2732_cast_fp16)[name = string("input_113_cast_fp16")]; + string h_21_pad_type_0 = const()[name = string("h_21_pad_type_0"), val = string("valid")]; + tensor h_21_strides_0 = const()[name = string("h_21_strides_0"), val = tensor([1, 1])]; + tensor h_21_pad_0 = const()[name = string("h_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_21_dilations_0 = const()[name = string("h_21_dilations_0"), val = tensor([1, 1])]; + int32 h_21_groups_0 = const()[name = string("h_21_groups_0"), val = int32(1)]; + tensor h_21_cast_fp16 = conv(dilations = h_21_dilations_0, groups = h_21_groups_0, pad = h_21_pad_0, pad_type = h_21_pad_type_0, strides = h_21_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_113_cast_fp16)[name = string("h_21_cast_fp16")]; + tensor x_83_cast_fp16 = add(x = x_81_cast_fp16, y = h_21_cast_fp16)[name = string("x_83_cast_fp16")]; + tensor key_cache_23_begin_0 = const()[name = string("key_cache_23_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_23_end_0 = const()[name = string("key_cache_23_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_23_end_mask_0 = const()[name = string("key_cache_23_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_23_cast_fp16 = slice_by_index(begin = key_cache_23_begin_0, end = key_cache_23_end_0, end_mask = key_cache_23_end_mask_0, x = layer_key_caches_5_cast_fp16)[name = string("key_cache_23_cast_fp16")]; + tensor value_cache_23_begin_0 = const()[name = string("value_cache_23_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_23_end_0 = const()[name = string("value_cache_23_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_23_end_mask_0 = const()[name = string("value_cache_23_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_23_cast_fp16 = slice_by_index(begin = value_cache_23_begin_0, end = value_cache_23_end_0, end_mask = value_cache_23_end_mask_0, x = layer_value_caches_5_cast_fp16)[name = string("value_cache_23_cast_fp16")]; + int32 var_2785 = const()[name = string("op_2785"), val = int32(2)]; + int32 var_2789 = const()[name = string("op_2789"), val = int32(3)]; + tensor var_2804_cast_fp16 = mul(x = x_83_cast_fp16, y = x_83_cast_fp16)[name = string("op_2804_cast_fp16")]; + tensor variance_91_axes_0 = const()[name = string("variance_91_axes_0"), val = tensor([1])]; + bool variance_91_keep_dims_0 = const()[name = string("variance_91_keep_dims_0"), val = bool(true)]; + tensor variance_91_cast_fp16 = reduce_mean(axes = variance_91_axes_0, keep_dims = variance_91_keep_dims_0, x = var_2804_cast_fp16)[name = string("variance_91_cast_fp16")]; + fp16 var_2807_to_fp16 = const()[name = string("op_2807_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2808_cast_fp16 = add(x = variance_91_cast_fp16, y = var_2807_to_fp16)[name = string("op_2808_cast_fp16")]; + fp32 var_2809_epsilon_0 = const()[name = string("op_2809_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2809_cast_fp16 = rsqrt(epsilon = var_2809_epsilon_0, x = var_2808_cast_fp16)[name = string("op_2809_cast_fp16")]; + tensor var_2810_cast_fp16 = mul(x = x_83_cast_fp16, y = var_2809_cast_fp16)[name = string("op_2810_cast_fp16")]; + tensor input_115_cast_fp16 = mul(x = var_2810_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_115_cast_fp16")]; + string q_67_pad_type_0 = const()[name = string("q_67_pad_type_0"), val = string("valid")]; + tensor q_67_strides_0 = const()[name = string("q_67_strides_0"), val = tensor([1, 1])]; + tensor q_67_pad_0 = const()[name = string("q_67_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_67_dilations_0 = const()[name = string("q_67_dilations_0"), val = tensor([1, 1])]; + int32 q_67_groups_0 = const()[name = string("q_67_groups_0"), val = int32(1)]; + tensor q_67_cast_fp16 = conv(dilations = q_67_dilations_0, groups = q_67_groups_0, pad = q_67_pad_0, pad_type = q_67_pad_type_0, strides = q_67_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_115_cast_fp16)[name = string("q_67_cast_fp16")]; + string k_67_pad_type_0 = const()[name = string("k_67_pad_type_0"), val = string("valid")]; + tensor k_67_strides_0 = const()[name = string("k_67_strides_0"), val = tensor([1, 1])]; + tensor k_67_pad_0 = const()[name = string("k_67_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_67_dilations_0 = const()[name = string("k_67_dilations_0"), val = tensor([1, 1])]; + int32 k_67_groups_0 = const()[name = string("k_67_groups_0"), val = int32(1)]; + tensor k_67_cast_fp16 = conv(dilations = k_67_dilations_0, groups = k_67_groups_0, pad = k_67_pad_0, pad_type = k_67_pad_type_0, strides = k_67_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_115_cast_fp16)[name = string("k_67_cast_fp16")]; + string v_23_pad_type_0 = const()[name = string("v_23_pad_type_0"), val = string("valid")]; + tensor v_23_strides_0 = const()[name = string("v_23_strides_0"), val = tensor([1, 1])]; + tensor v_23_pad_0 = const()[name = string("v_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_23_dilations_0 = const()[name = string("v_23_dilations_0"), val = tensor([1, 1])]; + int32 v_23_groups_0 = const()[name = string("v_23_groups_0"), val = int32(1)]; + tensor v_23_cast_fp16 = conv(dilations = v_23_dilations_0, groups = v_23_groups_0, pad = v_23_pad_0, pad_type = v_23_pad_type_0, strides = v_23_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_115_cast_fp16)[name = string("v_23_cast_fp16")]; + tensor var_2844 = const()[name = string("op_2844"), val = tensor([16, 128, 1, 1])]; + tensor x_85_cast_fp16 = reshape(shape = var_2844, x = q_67_cast_fp16)[name = string("x_85_cast_fp16")]; + tensor var_2847_cast_fp16 = mul(x = x_85_cast_fp16, y = x_85_cast_fp16)[name = string("op_2847_cast_fp16")]; + tensor variance_93_axes_0 = const()[name = string("variance_93_axes_0"), val = tensor([1])]; + bool variance_93_keep_dims_0 = const()[name = string("variance_93_keep_dims_0"), val = bool(true)]; + tensor variance_93_cast_fp16 = reduce_mean(axes = variance_93_axes_0, keep_dims = variance_93_keep_dims_0, x = var_2847_cast_fp16)[name = string("variance_93_cast_fp16")]; + fp16 var_2850_to_fp16 = const()[name = string("op_2850_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2851_cast_fp16 = add(x = variance_93_cast_fp16, y = var_2850_to_fp16)[name = string("op_2851_cast_fp16")]; + fp32 var_2852_epsilon_0 = const()[name = string("op_2852_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2852_cast_fp16 = rsqrt(epsilon = var_2852_epsilon_0, x = var_2851_cast_fp16)[name = string("op_2852_cast_fp16")]; + tensor var_2853_cast_fp16 = mul(x = x_85_cast_fp16, y = var_2852_cast_fp16)[name = string("op_2853_cast_fp16")]; + tensor q_69_cast_fp16 = mul(x = var_2853_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_69_cast_fp16")]; + tensor var_2855 = const()[name = string("op_2855"), val = tensor([8, 128, 1, 1])]; + tensor x_87_cast_fp16 = reshape(shape = var_2855, x = k_67_cast_fp16)[name = string("x_87_cast_fp16")]; + tensor var_2858_cast_fp16 = mul(x = x_87_cast_fp16, y = x_87_cast_fp16)[name = string("op_2858_cast_fp16")]; + tensor variance_95_axes_0 = const()[name = string("variance_95_axes_0"), val = tensor([1])]; + bool variance_95_keep_dims_0 = const()[name = string("variance_95_keep_dims_0"), val = bool(true)]; + tensor variance_95_cast_fp16 = reduce_mean(axes = variance_95_axes_0, keep_dims = variance_95_keep_dims_0, x = var_2858_cast_fp16)[name = string("variance_95_cast_fp16")]; + fp16 var_2861_to_fp16 = const()[name = string("op_2861_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2862_cast_fp16 = add(x = variance_95_cast_fp16, y = var_2861_to_fp16)[name = string("op_2862_cast_fp16")]; + fp32 var_2863_epsilon_0 = const()[name = string("op_2863_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2863_cast_fp16 = rsqrt(epsilon = var_2863_epsilon_0, x = var_2862_cast_fp16)[name = string("op_2863_cast_fp16")]; + tensor var_2864_cast_fp16 = mul(x = x_87_cast_fp16, y = var_2863_cast_fp16)[name = string("op_2864_cast_fp16")]; + tensor k_69_cast_fp16 = mul(x = var_2864_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_69_cast_fp16")]; + tensor var_2866 = const()[name = string("op_2866"), val = tensor([1, 16, 128, 1])]; + tensor z_45_cast_fp16 = reshape(shape = var_2866, x = q_69_cast_fp16)[name = string("z_45_cast_fp16")]; + tensor var_2868 = const()[name = string("op_2868"), val = tensor([1, 8, 128, 1])]; + tensor z_47_cast_fp16 = reshape(shape = var_2868, x = k_69_cast_fp16)[name = string("z_47_cast_fp16")]; + tensor z1_45_begin_0 = const()[name = string("z1_45_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_45_end_0 = const()[name = string("z1_45_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_45_end_mask_0 = const()[name = string("z1_45_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_45_cast_fp16 = slice_by_index(begin = z1_45_begin_0, end = z1_45_end_0, end_mask = z1_45_end_mask_0, x = z_45_cast_fp16)[name = string("z1_45_cast_fp16")]; + tensor z2_45_begin_0 = const()[name = string("z2_45_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_45_end_0 = const()[name = string("z2_45_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_45_end_mask_0 = const()[name = string("z2_45_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_45_cast_fp16 = slice_by_index(begin = z2_45_begin_0, end = z2_45_end_0, end_mask = z2_45_end_mask_0, x = z_45_cast_fp16)[name = string("z2_45_cast_fp16")]; + tensor var_2876_cast_fp16 = mul(x = z_45_cast_fp16, y = cos_21_to_fp16)[name = string("op_2876_cast_fp16")]; + fp16 const_25_promoted_to_fp16 = const()[name = string("const_25_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2877_cast_fp16 = mul(x = z2_45_cast_fp16, y = const_25_promoted_to_fp16)[name = string("op_2877_cast_fp16")]; + bool var_2879_interleave_0 = const()[name = string("op_2879_interleave_0"), val = bool(false)]; + tensor var_2879_cast_fp16 = concat(axis = var_2785, interleave = var_2879_interleave_0, values = (var_2877_cast_fp16, z1_45_cast_fp16))[name = string("op_2879_cast_fp16")]; + tensor var_2880_cast_fp16 = mul(x = var_2879_cast_fp16, y = sin_21_to_fp16)[name = string("op_2880_cast_fp16")]; + tensor q_71_cast_fp16 = add(x = var_2876_cast_fp16, y = var_2880_cast_fp16)[name = string("q_71_cast_fp16")]; + tensor z1_47_begin_0 = const()[name = string("z1_47_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_47_end_0 = const()[name = string("z1_47_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_47_end_mask_0 = const()[name = string("z1_47_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_47_cast_fp16 = slice_by_index(begin = z1_47_begin_0, end = z1_47_end_0, end_mask = z1_47_end_mask_0, x = z_47_cast_fp16)[name = string("z1_47_cast_fp16")]; + tensor z2_47_begin_0 = const()[name = string("z2_47_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_47_end_0 = const()[name = string("z2_47_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_47_end_mask_0 = const()[name = string("z2_47_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_47_cast_fp16 = slice_by_index(begin = z2_47_begin_0, end = z2_47_end_0, end_mask = z2_47_end_mask_0, x = z_47_cast_fp16)[name = string("z2_47_cast_fp16")]; + tensor var_2888_cast_fp16 = mul(x = z_47_cast_fp16, y = cos_21_to_fp16)[name = string("op_2888_cast_fp16")]; + fp16 const_26_promoted_to_fp16 = const()[name = string("const_26_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2889_cast_fp16 = mul(x = z2_47_cast_fp16, y = const_26_promoted_to_fp16)[name = string("op_2889_cast_fp16")]; + bool var_2891_interleave_0 = const()[name = string("op_2891_interleave_0"), val = bool(false)]; + tensor var_2891_cast_fp16 = concat(axis = var_2785, interleave = var_2891_interleave_0, values = (var_2889_cast_fp16, z1_47_cast_fp16))[name = string("op_2891_cast_fp16")]; + tensor var_2892_cast_fp16 = mul(x = var_2891_cast_fp16, y = sin_21_to_fp16)[name = string("op_2892_cast_fp16")]; + tensor k_71_cast_fp16 = add(x = var_2888_cast_fp16, y = var_2892_cast_fp16)[name = string("k_71_cast_fp16")]; + tensor var_2894 = const()[name = string("op_2894"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_23_cast_fp16 = reshape(shape = var_2894, x = k_71_cast_fp16)[name = string("cur_key_23_cast_fp16")]; + tensor var_2896_to_fp16 = const()[name = string("op_2896_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634048)))]; + tensor var_2897_cast_fp16 = mul(x = key_cache_23_cast_fp16, y = var_2896_to_fp16)[name = string("op_2897_cast_fp16")]; + tensor upd_23_to_fp16 = const()[name = string("upd_23_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634176)))]; + tensor var_2898_cast_fp16 = mul(x = cur_key_23_cast_fp16, y = upd_23_to_fp16)[name = string("op_2898_cast_fp16")]; + tensor key_23_cast_fp16 = add(x = var_2897_cast_fp16, y = var_2898_cast_fp16)[name = string("key_23_cast_fp16")]; + tensor var_2900_to_fp16 = const()[name = string("op_2900_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634048)))]; + tensor var_2901_cast_fp16 = mul(x = value_cache_23_cast_fp16, y = var_2900_to_fp16)[name = string("op_2901_cast_fp16")]; + tensor var_2902_cast_fp16 = mul(x = v_23_cast_fp16, y = upd_23_to_fp16)[name = string("op_2902_cast_fp16")]; + tensor value_23_cast_fp16 = add(x = var_2901_cast_fp16, y = var_2902_cast_fp16)[name = string("value_23_cast_fp16")]; + tensor var_2904 = const()[name = string("op_2904"), val = tensor([1, 8, 128, 16])]; + tensor kh_45_cast_fp16 = reshape(shape = var_2904, x = key_23_cast_fp16)[name = string("kh_45_cast_fp16")]; + tensor var_2906 = const()[name = string("op_2906"), val = tensor([1, 8, 128, 16])]; + tensor vh_45_cast_fp16 = reshape(shape = var_2906, x = value_23_cast_fp16)[name = string("vh_45_cast_fp16")]; + tensor transpose_44_perm_0 = const()[name = string("transpose_44_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_22_reps_0 = const()[name = string("tile_22_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_44_cast_fp16 = transpose(perm = transpose_44_perm_0, x = kh_45_cast_fp16)[name = string("transpose_413")]; + tensor tile_22_cast_fp16 = tile(reps = tile_22_reps_0, x = transpose_44_cast_fp16)[name = string("tile_22_cast_fp16")]; + tensor concat_57 = const()[name = string("concat_57"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_44_cast_fp16 = reshape(shape = concat_57, x = tile_22_cast_fp16)[name = string("reshape_44_cast_fp16")]; + tensor transpose_45_perm_0 = const()[name = string("transpose_45_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_58 = const()[name = string("concat_58"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_45_cast_fp16 = transpose(perm = transpose_45_perm_0, x = reshape_44_cast_fp16)[name = string("transpose_412")]; + tensor reshape_45_cast_fp16 = reshape(shape = concat_58, x = transpose_45_cast_fp16)[name = string("reshape_45_cast_fp16")]; + tensor transpose_46_perm_0 = const()[name = string("transpose_46_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_23_reps_0 = const()[name = string("tile_23_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_46_cast_fp16 = transpose(perm = transpose_46_perm_0, x = vh_45_cast_fp16)[name = string("transpose_411")]; + tensor tile_23_cast_fp16 = tile(reps = tile_23_reps_0, x = transpose_46_cast_fp16)[name = string("tile_23_cast_fp16")]; + tensor concat_59 = const()[name = string("concat_59"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_46_cast_fp16 = reshape(shape = concat_59, x = tile_23_cast_fp16)[name = string("reshape_46_cast_fp16")]; + tensor transpose_47_perm_0 = const()[name = string("transpose_47_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_60 = const()[name = string("concat_60"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_47_cast_fp16 = transpose(perm = transpose_47_perm_0, x = reshape_46_cast_fp16)[name = string("transpose_410")]; + tensor reshape_47_cast_fp16 = reshape(shape = concat_60, x = transpose_47_cast_fp16)[name = string("reshape_47_cast_fp16")]; + fp16 var_2910_to_fp16 = const()[name = string("op_2910_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_2911_cast_fp16 = mul(x = q_71_cast_fp16, y = var_2910_to_fp16)[name = string("op_2911_cast_fp16")]; + tensor transpose_361_perm_0 = const()[name = string("transpose_361_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_47_transpose_x_1 = const()[name = string("w_47_transpose_x_1"), val = bool(true)]; + bool w_47_transpose_y_1 = const()[name = string("w_47_transpose_y_1"), val = bool(false)]; + tensor transpose_361_cast_fp16 = transpose(perm = transpose_361_perm_0, x = reshape_45_cast_fp16)[name = string("transpose_409")]; + tensor w_47_cast_fp16 = matmul(transpose_x = w_47_transpose_x_1, transpose_y = w_47_transpose_y_1, x = var_2911_cast_fp16, y = transpose_361_cast_fp16)[name = string("w_47_cast_fp16")]; + tensor pad_23_to_fp16 = const()[name = string("pad_23_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634304)))]; + tensor var_2914_cast_fp16 = add(x = w_47_cast_fp16, y = pad_23_to_fp16)[name = string("op_2914_cast_fp16")]; + tensor w_49_cast_fp16 = softmax(axis = var_2789, x = var_2914_cast_fp16)[name = string("w_49_cast_fp16")]; + tensor transpose_362_perm_0 = const()[name = string("transpose_362_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_23_transpose_x_1 = const()[name = string("attn_23_transpose_x_1"), val = bool(false)]; + bool attn_23_transpose_y_1 = const()[name = string("attn_23_transpose_y_1"), val = bool(true)]; + tensor transpose_362_cast_fp16 = transpose(perm = transpose_362_perm_0, x = reshape_47_cast_fp16)[name = string("transpose_408")]; + tensor attn_23_cast_fp16 = matmul(transpose_x = attn_23_transpose_x_1, transpose_y = attn_23_transpose_y_1, x = transpose_362_cast_fp16, y = w_49_cast_fp16)[name = string("attn_23_cast_fp16")]; + tensor var_2918 = const()[name = string("op_2918"), val = tensor([1, 2048, 1, 1])]; + tensor input_117_cast_fp16 = reshape(shape = var_2918, x = attn_23_cast_fp16)[name = string("input_117_cast_fp16")]; + string attn_output_23_pad_type_0 = const()[name = string("attn_output_23_pad_type_0"), val = string("valid")]; + tensor attn_output_23_strides_0 = const()[name = string("attn_output_23_strides_0"), val = tensor([1, 1])]; + tensor attn_output_23_pad_0 = const()[name = string("attn_output_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_23_dilations_0 = const()[name = string("attn_output_23_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_23_groups_0 = const()[name = string("attn_output_23_groups_0"), val = int32(1)]; + tensor attn_output_23_cast_fp16 = conv(dilations = attn_output_23_dilations_0, groups = attn_output_23_groups_0, pad = attn_output_23_pad_0, pad_type = attn_output_23_pad_type_0, strides = attn_output_23_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_117_cast_fp16)[name = string("attn_output_23_cast_fp16")]; + tensor x_89_cast_fp16 = add(x = x_83_cast_fp16, y = attn_output_23_cast_fp16)[name = string("x_89_cast_fp16")]; + tensor var_2932_cast_fp16 = mul(x = x_89_cast_fp16, y = x_89_cast_fp16)[name = string("op_2932_cast_fp16")]; + tensor variance_97_axes_0 = const()[name = string("variance_97_axes_0"), val = tensor([1])]; + bool variance_97_keep_dims_0 = const()[name = string("variance_97_keep_dims_0"), val = bool(true)]; + tensor variance_97_cast_fp16 = reduce_mean(axes = variance_97_axes_0, keep_dims = variance_97_keep_dims_0, x = var_2932_cast_fp16)[name = string("variance_97_cast_fp16")]; + fp16 var_2935_to_fp16 = const()[name = string("op_2935_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2936_cast_fp16 = add(x = variance_97_cast_fp16, y = var_2935_to_fp16)[name = string("op_2936_cast_fp16")]; + fp32 var_2937_epsilon_0 = const()[name = string("op_2937_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2937_cast_fp16 = rsqrt(epsilon = var_2937_epsilon_0, x = var_2936_cast_fp16)[name = string("op_2937_cast_fp16")]; + tensor var_2938_cast_fp16 = mul(x = x_89_cast_fp16, y = var_2937_cast_fp16)[name = string("op_2938_cast_fp16")]; + tensor input_119_cast_fp16 = mul(x = var_2938_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_119_cast_fp16")]; + string input_121_pad_type_0 = const()[name = string("input_121_pad_type_0"), val = string("valid")]; + tensor input_121_strides_0 = const()[name = string("input_121_strides_0"), val = tensor([1, 1])]; + tensor input_121_pad_0 = const()[name = string("input_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_121_dilations_0 = const()[name = string("input_121_dilations_0"), val = tensor([1, 1])]; + int32 input_121_groups_0 = const()[name = string("input_121_groups_0"), val = int32(1)]; + tensor input_121_cast_fp16 = conv(dilations = input_121_dilations_0, groups = input_121_groups_0, pad = input_121_pad_0, pad_type = input_121_pad_type_0, strides = input_121_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_119_cast_fp16)[name = string("input_121_cast_fp16")]; + tensor var_2946_cast_fp16 = silu(x = input_121_cast_fp16)[name = string("op_2946_cast_fp16")]; + string var_2952_pad_type_0 = const()[name = string("op_2952_pad_type_0"), val = string("valid")]; + tensor var_2952_strides_0 = const()[name = string("op_2952_strides_0"), val = tensor([1, 1])]; + tensor var_2952_pad_0 = const()[name = string("op_2952_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2952_dilations_0 = const()[name = string("op_2952_dilations_0"), val = tensor([1, 1])]; + int32 var_2952_groups_0 = const()[name = string("op_2952_groups_0"), val = int32(1)]; + tensor var_2952_cast_fp16 = conv(dilations = var_2952_dilations_0, groups = var_2952_groups_0, pad = var_2952_pad_0, pad_type = var_2952_pad_type_0, strides = var_2952_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_119_cast_fp16)[name = string("op_2952_cast_fp16")]; + tensor input_123_cast_fp16 = mul(x = var_2946_cast_fp16, y = var_2952_cast_fp16)[name = string("input_123_cast_fp16")]; + string h_23_pad_type_0 = const()[name = string("h_23_pad_type_0"), val = string("valid")]; + tensor h_23_strides_0 = const()[name = string("h_23_strides_0"), val = tensor([1, 1])]; + tensor h_23_pad_0 = const()[name = string("h_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_23_dilations_0 = const()[name = string("h_23_dilations_0"), val = tensor([1, 1])]; + int32 h_23_groups_0 = const()[name = string("h_23_groups_0"), val = int32(1)]; + tensor h_23_cast_fp16 = conv(dilations = h_23_dilations_0, groups = h_23_groups_0, pad = h_23_pad_0, pad_type = h_23_pad_type_0, strides = h_23_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_123_cast_fp16)[name = string("h_23_cast_fp16")]; + tensor x_91_cast_fp16 = add(x = x_89_cast_fp16, y = h_23_cast_fp16)[name = string("x_91_cast_fp16")]; + tensor key_cache_25_begin_0 = const()[name = string("key_cache_25_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_25_end_0 = const()[name = string("key_cache_25_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_25_end_mask_0 = const()[name = string("key_cache_25_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_25_cast_fp16 = slice_by_index(begin = key_cache_25_begin_0, end = key_cache_25_end_0, end_mask = key_cache_25_end_mask_0, x = layer_key_caches_5_cast_fp16)[name = string("key_cache_25_cast_fp16")]; + tensor value_cache_25_begin_0 = const()[name = string("value_cache_25_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_25_end_0 = const()[name = string("value_cache_25_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_25_end_mask_0 = const()[name = string("value_cache_25_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_25_cast_fp16 = slice_by_index(begin = value_cache_25_begin_0, end = value_cache_25_end_0, end_mask = value_cache_25_end_mask_0, x = layer_value_caches_5_cast_fp16)[name = string("value_cache_25_cast_fp16")]; + int32 var_3005 = const()[name = string("op_3005"), val = int32(2)]; + int32 var_3009 = const()[name = string("op_3009"), val = int32(3)]; + tensor var_3024_cast_fp16 = mul(x = x_91_cast_fp16, y = x_91_cast_fp16)[name = string("op_3024_cast_fp16")]; + tensor variance_99_axes_0 = const()[name = string("variance_99_axes_0"), val = tensor([1])]; + bool variance_99_keep_dims_0 = const()[name = string("variance_99_keep_dims_0"), val = bool(true)]; + tensor variance_99_cast_fp16 = reduce_mean(axes = variance_99_axes_0, keep_dims = variance_99_keep_dims_0, x = var_3024_cast_fp16)[name = string("variance_99_cast_fp16")]; + fp16 var_3027_to_fp16 = const()[name = string("op_3027_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3028_cast_fp16 = add(x = variance_99_cast_fp16, y = var_3027_to_fp16)[name = string("op_3028_cast_fp16")]; + fp32 var_3029_epsilon_0 = const()[name = string("op_3029_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3029_cast_fp16 = rsqrt(epsilon = var_3029_epsilon_0, x = var_3028_cast_fp16)[name = string("op_3029_cast_fp16")]; + tensor var_3030_cast_fp16 = mul(x = x_91_cast_fp16, y = var_3029_cast_fp16)[name = string("op_3030_cast_fp16")]; + tensor input_125_cast_fp16 = mul(x = var_3030_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_125_cast_fp16")]; + string q_73_pad_type_0 = const()[name = string("q_73_pad_type_0"), val = string("valid")]; + tensor q_73_strides_0 = const()[name = string("q_73_strides_0"), val = tensor([1, 1])]; + tensor q_73_pad_0 = const()[name = string("q_73_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_73_dilations_0 = const()[name = string("q_73_dilations_0"), val = tensor([1, 1])]; + int32 q_73_groups_0 = const()[name = string("q_73_groups_0"), val = int32(1)]; + tensor q_73_cast_fp16 = conv(dilations = q_73_dilations_0, groups = q_73_groups_0, pad = q_73_pad_0, pad_type = q_73_pad_type_0, strides = q_73_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_125_cast_fp16)[name = string("q_73_cast_fp16")]; + string k_73_pad_type_0 = const()[name = string("k_73_pad_type_0"), val = string("valid")]; + tensor k_73_strides_0 = const()[name = string("k_73_strides_0"), val = tensor([1, 1])]; + tensor k_73_pad_0 = const()[name = string("k_73_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_73_dilations_0 = const()[name = string("k_73_dilations_0"), val = tensor([1, 1])]; + int32 k_73_groups_0 = const()[name = string("k_73_groups_0"), val = int32(1)]; + tensor k_73_cast_fp16 = conv(dilations = k_73_dilations_0, groups = k_73_groups_0, pad = k_73_pad_0, pad_type = k_73_pad_type_0, strides = k_73_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_125_cast_fp16)[name = string("k_73_cast_fp16")]; + string v_25_pad_type_0 = const()[name = string("v_25_pad_type_0"), val = string("valid")]; + tensor v_25_strides_0 = const()[name = string("v_25_strides_0"), val = tensor([1, 1])]; + tensor v_25_pad_0 = const()[name = string("v_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_25_dilations_0 = const()[name = string("v_25_dilations_0"), val = tensor([1, 1])]; + int32 v_25_groups_0 = const()[name = string("v_25_groups_0"), val = int32(1)]; + tensor v_25_cast_fp16 = conv(dilations = v_25_dilations_0, groups = v_25_groups_0, pad = v_25_pad_0, pad_type = v_25_pad_type_0, strides = v_25_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_125_cast_fp16)[name = string("v_25_cast_fp16")]; + tensor var_3064 = const()[name = string("op_3064"), val = tensor([16, 128, 1, 1])]; + tensor x_93_cast_fp16 = reshape(shape = var_3064, x = q_73_cast_fp16)[name = string("x_93_cast_fp16")]; + tensor var_3067_cast_fp16 = mul(x = x_93_cast_fp16, y = x_93_cast_fp16)[name = string("op_3067_cast_fp16")]; + tensor variance_101_axes_0 = const()[name = string("variance_101_axes_0"), val = tensor([1])]; + bool variance_101_keep_dims_0 = const()[name = string("variance_101_keep_dims_0"), val = bool(true)]; + tensor variance_101_cast_fp16 = reduce_mean(axes = variance_101_axes_0, keep_dims = variance_101_keep_dims_0, x = var_3067_cast_fp16)[name = string("variance_101_cast_fp16")]; + fp16 var_3070_to_fp16 = const()[name = string("op_3070_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3071_cast_fp16 = add(x = variance_101_cast_fp16, y = var_3070_to_fp16)[name = string("op_3071_cast_fp16")]; + fp32 var_3072_epsilon_0 = const()[name = string("op_3072_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3072_cast_fp16 = rsqrt(epsilon = var_3072_epsilon_0, x = var_3071_cast_fp16)[name = string("op_3072_cast_fp16")]; + tensor var_3073_cast_fp16 = mul(x = x_93_cast_fp16, y = var_3072_cast_fp16)[name = string("op_3073_cast_fp16")]; + tensor q_75_cast_fp16 = mul(x = var_3073_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_75_cast_fp16")]; + tensor var_3075 = const()[name = string("op_3075"), val = tensor([8, 128, 1, 1])]; + tensor x_95_cast_fp16 = reshape(shape = var_3075, x = k_73_cast_fp16)[name = string("x_95_cast_fp16")]; + tensor var_3078_cast_fp16 = mul(x = x_95_cast_fp16, y = x_95_cast_fp16)[name = string("op_3078_cast_fp16")]; + tensor variance_103_axes_0 = const()[name = string("variance_103_axes_0"), val = tensor([1])]; + bool variance_103_keep_dims_0 = const()[name = string("variance_103_keep_dims_0"), val = bool(true)]; + tensor variance_103_cast_fp16 = reduce_mean(axes = variance_103_axes_0, keep_dims = variance_103_keep_dims_0, x = var_3078_cast_fp16)[name = string("variance_103_cast_fp16")]; + fp16 var_3081_to_fp16 = const()[name = string("op_3081_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3082_cast_fp16 = add(x = variance_103_cast_fp16, y = var_3081_to_fp16)[name = string("op_3082_cast_fp16")]; + fp32 var_3083_epsilon_0 = const()[name = string("op_3083_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3083_cast_fp16 = rsqrt(epsilon = var_3083_epsilon_0, x = var_3082_cast_fp16)[name = string("op_3083_cast_fp16")]; + tensor var_3084_cast_fp16 = mul(x = x_95_cast_fp16, y = var_3083_cast_fp16)[name = string("op_3084_cast_fp16")]; + tensor k_75_cast_fp16 = mul(x = var_3084_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_75_cast_fp16")]; + tensor var_3086 = const()[name = string("op_3086"), val = tensor([1, 16, 128, 1])]; + tensor z_49_cast_fp16 = reshape(shape = var_3086, x = q_75_cast_fp16)[name = string("z_49_cast_fp16")]; + tensor var_3088 = const()[name = string("op_3088"), val = tensor([1, 8, 128, 1])]; + tensor z_51_cast_fp16 = reshape(shape = var_3088, x = k_75_cast_fp16)[name = string("z_51_cast_fp16")]; + tensor z1_49_begin_0 = const()[name = string("z1_49_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_49_end_0 = const()[name = string("z1_49_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_49_end_mask_0 = const()[name = string("z1_49_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_49_cast_fp16 = slice_by_index(begin = z1_49_begin_0, end = z1_49_end_0, end_mask = z1_49_end_mask_0, x = z_49_cast_fp16)[name = string("z1_49_cast_fp16")]; + tensor z2_49_begin_0 = const()[name = string("z2_49_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_49_end_0 = const()[name = string("z2_49_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_49_end_mask_0 = const()[name = string("z2_49_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_49_cast_fp16 = slice_by_index(begin = z2_49_begin_0, end = z2_49_end_0, end_mask = z2_49_end_mask_0, x = z_49_cast_fp16)[name = string("z2_49_cast_fp16")]; + tensor var_3096_cast_fp16 = mul(x = z_49_cast_fp16, y = cos_21_to_fp16)[name = string("op_3096_cast_fp16")]; + fp16 const_27_promoted_to_fp16 = const()[name = string("const_27_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3097_cast_fp16 = mul(x = z2_49_cast_fp16, y = const_27_promoted_to_fp16)[name = string("op_3097_cast_fp16")]; + bool var_3099_interleave_0 = const()[name = string("op_3099_interleave_0"), val = bool(false)]; + tensor var_3099_cast_fp16 = concat(axis = var_3005, interleave = var_3099_interleave_0, values = (var_3097_cast_fp16, z1_49_cast_fp16))[name = string("op_3099_cast_fp16")]; + tensor var_3100_cast_fp16 = mul(x = var_3099_cast_fp16, y = sin_21_to_fp16)[name = string("op_3100_cast_fp16")]; + tensor q_77_cast_fp16 = add(x = var_3096_cast_fp16, y = var_3100_cast_fp16)[name = string("q_77_cast_fp16")]; + tensor z1_51_begin_0 = const()[name = string("z1_51_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_51_end_0 = const()[name = string("z1_51_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_51_end_mask_0 = const()[name = string("z1_51_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_51_cast_fp16 = slice_by_index(begin = z1_51_begin_0, end = z1_51_end_0, end_mask = z1_51_end_mask_0, x = z_51_cast_fp16)[name = string("z1_51_cast_fp16")]; + tensor z2_51_begin_0 = const()[name = string("z2_51_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_51_end_0 = const()[name = string("z2_51_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_51_end_mask_0 = const()[name = string("z2_51_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_51_cast_fp16 = slice_by_index(begin = z2_51_begin_0, end = z2_51_end_0, end_mask = z2_51_end_mask_0, x = z_51_cast_fp16)[name = string("z2_51_cast_fp16")]; + tensor var_3108_cast_fp16 = mul(x = z_51_cast_fp16, y = cos_21_to_fp16)[name = string("op_3108_cast_fp16")]; + fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3109_cast_fp16 = mul(x = z2_51_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_3109_cast_fp16")]; + bool var_3111_interleave_0 = const()[name = string("op_3111_interleave_0"), val = bool(false)]; + tensor var_3111_cast_fp16 = concat(axis = var_3005, interleave = var_3111_interleave_0, values = (var_3109_cast_fp16, z1_51_cast_fp16))[name = string("op_3111_cast_fp16")]; + tensor var_3112_cast_fp16 = mul(x = var_3111_cast_fp16, y = sin_21_to_fp16)[name = string("op_3112_cast_fp16")]; + tensor k_77_cast_fp16 = add(x = var_3108_cast_fp16, y = var_3112_cast_fp16)[name = string("k_77_cast_fp16")]; + tensor var_3114 = const()[name = string("op_3114"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_25_cast_fp16 = reshape(shape = var_3114, x = k_77_cast_fp16)[name = string("cur_key_25_cast_fp16")]; + tensor var_3116_to_fp16 = const()[name = string("op_3116_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634048)))]; + tensor var_3117_cast_fp16 = mul(x = key_cache_25_cast_fp16, y = var_3116_to_fp16)[name = string("op_3117_cast_fp16")]; + tensor upd_25_to_fp16 = const()[name = string("upd_25_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634176)))]; + tensor var_3118_cast_fp16 = mul(x = cur_key_25_cast_fp16, y = upd_25_to_fp16)[name = string("op_3118_cast_fp16")]; + tensor key_25_cast_fp16 = add(x = var_3117_cast_fp16, y = var_3118_cast_fp16)[name = string("key_25_cast_fp16")]; + tensor var_3120_to_fp16 = const()[name = string("op_3120_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634048)))]; + tensor var_3121_cast_fp16 = mul(x = value_cache_25_cast_fp16, y = var_3120_to_fp16)[name = string("op_3121_cast_fp16")]; + tensor var_3122_cast_fp16 = mul(x = v_25_cast_fp16, y = upd_25_to_fp16)[name = string("op_3122_cast_fp16")]; + tensor value_25_cast_fp16 = add(x = var_3121_cast_fp16, y = var_3122_cast_fp16)[name = string("value_25_cast_fp16")]; + tensor var_3124 = const()[name = string("op_3124"), val = tensor([1, 8, 128, 16])]; + tensor kh_49_cast_fp16 = reshape(shape = var_3124, x = key_25_cast_fp16)[name = string("kh_49_cast_fp16")]; + tensor var_3126 = const()[name = string("op_3126"), val = tensor([1, 8, 128, 16])]; + tensor vh_49_cast_fp16 = reshape(shape = var_3126, x = value_25_cast_fp16)[name = string("vh_49_cast_fp16")]; + tensor transpose_48_perm_0 = const()[name = string("transpose_48_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_24_reps_0 = const()[name = string("tile_24_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_48_cast_fp16 = transpose(perm = transpose_48_perm_0, x = kh_49_cast_fp16)[name = string("transpose_407")]; + tensor tile_24_cast_fp16 = tile(reps = tile_24_reps_0, x = transpose_48_cast_fp16)[name = string("tile_24_cast_fp16")]; + tensor concat_61 = const()[name = string("concat_61"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_48_cast_fp16 = reshape(shape = concat_61, x = tile_24_cast_fp16)[name = string("reshape_48_cast_fp16")]; + tensor transpose_49_perm_0 = const()[name = string("transpose_49_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_62 = const()[name = string("concat_62"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_49_cast_fp16 = transpose(perm = transpose_49_perm_0, x = reshape_48_cast_fp16)[name = string("transpose_406")]; + tensor reshape_49_cast_fp16 = reshape(shape = concat_62, x = transpose_49_cast_fp16)[name = string("reshape_49_cast_fp16")]; + tensor transpose_50_perm_0 = const()[name = string("transpose_50_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_25_reps_0 = const()[name = string("tile_25_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_50_cast_fp16 = transpose(perm = transpose_50_perm_0, x = vh_49_cast_fp16)[name = string("transpose_405")]; + tensor tile_25_cast_fp16 = tile(reps = tile_25_reps_0, x = transpose_50_cast_fp16)[name = string("tile_25_cast_fp16")]; + tensor concat_63 = const()[name = string("concat_63"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_50_cast_fp16 = reshape(shape = concat_63, x = tile_25_cast_fp16)[name = string("reshape_50_cast_fp16")]; + tensor transpose_51_perm_0 = const()[name = string("transpose_51_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_64 = const()[name = string("concat_64"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_51_cast_fp16 = transpose(perm = transpose_51_perm_0, x = reshape_50_cast_fp16)[name = string("transpose_404")]; + tensor reshape_51_cast_fp16 = reshape(shape = concat_64, x = transpose_51_cast_fp16)[name = string("reshape_51_cast_fp16")]; + fp16 var_3130_to_fp16 = const()[name = string("op_3130_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_3131_cast_fp16 = mul(x = q_77_cast_fp16, y = var_3130_to_fp16)[name = string("op_3131_cast_fp16")]; + tensor transpose_365_perm_0 = const()[name = string("transpose_365_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_51_transpose_x_1 = const()[name = string("w_51_transpose_x_1"), val = bool(true)]; + bool w_51_transpose_y_1 = const()[name = string("w_51_transpose_y_1"), val = bool(false)]; + tensor transpose_365_cast_fp16 = transpose(perm = transpose_365_perm_0, x = reshape_49_cast_fp16)[name = string("transpose_403")]; + tensor w_51_cast_fp16 = matmul(transpose_x = w_51_transpose_x_1, transpose_y = w_51_transpose_y_1, x = var_3131_cast_fp16, y = transpose_365_cast_fp16)[name = string("w_51_cast_fp16")]; + tensor pad_25_to_fp16 = const()[name = string("pad_25_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634304)))]; + tensor var_3134_cast_fp16 = add(x = w_51_cast_fp16, y = pad_25_to_fp16)[name = string("op_3134_cast_fp16")]; + tensor w_53_cast_fp16 = softmax(axis = var_3009, x = var_3134_cast_fp16)[name = string("w_53_cast_fp16")]; + tensor transpose_366_perm_0 = const()[name = string("transpose_366_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_25_transpose_x_1 = const()[name = string("attn_25_transpose_x_1"), val = bool(false)]; + bool attn_25_transpose_y_1 = const()[name = string("attn_25_transpose_y_1"), val = bool(true)]; + tensor transpose_366_cast_fp16 = transpose(perm = transpose_366_perm_0, x = reshape_51_cast_fp16)[name = string("transpose_402")]; + tensor attn_25_cast_fp16 = matmul(transpose_x = attn_25_transpose_x_1, transpose_y = attn_25_transpose_y_1, x = transpose_366_cast_fp16, y = w_53_cast_fp16)[name = string("attn_25_cast_fp16")]; + tensor var_3138 = const()[name = string("op_3138"), val = tensor([1, 2048, 1, 1])]; + tensor input_127_cast_fp16 = reshape(shape = var_3138, x = attn_25_cast_fp16)[name = string("input_127_cast_fp16")]; + string attn_output_25_pad_type_0 = const()[name = string("attn_output_25_pad_type_0"), val = string("valid")]; + tensor attn_output_25_strides_0 = const()[name = string("attn_output_25_strides_0"), val = tensor([1, 1])]; + tensor attn_output_25_pad_0 = const()[name = string("attn_output_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_25_dilations_0 = const()[name = string("attn_output_25_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_25_groups_0 = const()[name = string("attn_output_25_groups_0"), val = int32(1)]; + tensor attn_output_25_cast_fp16 = conv(dilations = attn_output_25_dilations_0, groups = attn_output_25_groups_0, pad = attn_output_25_pad_0, pad_type = attn_output_25_pad_type_0, strides = attn_output_25_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_127_cast_fp16)[name = string("attn_output_25_cast_fp16")]; + tensor x_97_cast_fp16 = add(x = x_91_cast_fp16, y = attn_output_25_cast_fp16)[name = string("x_97_cast_fp16")]; + tensor var_3152_cast_fp16 = mul(x = x_97_cast_fp16, y = x_97_cast_fp16)[name = string("op_3152_cast_fp16")]; + tensor variance_105_axes_0 = const()[name = string("variance_105_axes_0"), val = tensor([1])]; + bool variance_105_keep_dims_0 = const()[name = string("variance_105_keep_dims_0"), val = bool(true)]; + tensor variance_105_cast_fp16 = reduce_mean(axes = variance_105_axes_0, keep_dims = variance_105_keep_dims_0, x = var_3152_cast_fp16)[name = string("variance_105_cast_fp16")]; + fp16 var_3155_to_fp16 = const()[name = string("op_3155_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3156_cast_fp16 = add(x = variance_105_cast_fp16, y = var_3155_to_fp16)[name = string("op_3156_cast_fp16")]; + fp32 var_3157_epsilon_0 = const()[name = string("op_3157_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3157_cast_fp16 = rsqrt(epsilon = var_3157_epsilon_0, x = var_3156_cast_fp16)[name = string("op_3157_cast_fp16")]; + tensor var_3158_cast_fp16 = mul(x = x_97_cast_fp16, y = var_3157_cast_fp16)[name = string("op_3158_cast_fp16")]; + tensor input_129_cast_fp16 = mul(x = var_3158_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_129_cast_fp16")]; + string input_131_pad_type_0 = const()[name = string("input_131_pad_type_0"), val = string("valid")]; + tensor input_131_strides_0 = const()[name = string("input_131_strides_0"), val = tensor([1, 1])]; + tensor input_131_pad_0 = const()[name = string("input_131_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_131_dilations_0 = const()[name = string("input_131_dilations_0"), val = tensor([1, 1])]; + int32 input_131_groups_0 = const()[name = string("input_131_groups_0"), val = int32(1)]; + tensor input_131_cast_fp16 = conv(dilations = input_131_dilations_0, groups = input_131_groups_0, pad = input_131_pad_0, pad_type = input_131_pad_type_0, strides = input_131_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_129_cast_fp16)[name = string("input_131_cast_fp16")]; + tensor var_3166_cast_fp16 = silu(x = input_131_cast_fp16)[name = string("op_3166_cast_fp16")]; + string var_3172_pad_type_0 = const()[name = string("op_3172_pad_type_0"), val = string("valid")]; + tensor var_3172_strides_0 = const()[name = string("op_3172_strides_0"), val = tensor([1, 1])]; + tensor var_3172_pad_0 = const()[name = string("op_3172_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3172_dilations_0 = const()[name = string("op_3172_dilations_0"), val = tensor([1, 1])]; + int32 var_3172_groups_0 = const()[name = string("op_3172_groups_0"), val = int32(1)]; + tensor var_3172_cast_fp16 = conv(dilations = var_3172_dilations_0, groups = var_3172_groups_0, pad = var_3172_pad_0, pad_type = var_3172_pad_type_0, strides = var_3172_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_129_cast_fp16)[name = string("op_3172_cast_fp16")]; + tensor input_133_cast_fp16 = mul(x = var_3166_cast_fp16, y = var_3172_cast_fp16)[name = string("input_133_cast_fp16")]; + string h_25_pad_type_0 = const()[name = string("h_25_pad_type_0"), val = string("valid")]; + tensor h_25_strides_0 = const()[name = string("h_25_strides_0"), val = tensor([1, 1])]; + tensor h_25_pad_0 = const()[name = string("h_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_25_dilations_0 = const()[name = string("h_25_dilations_0"), val = tensor([1, 1])]; + int32 h_25_groups_0 = const()[name = string("h_25_groups_0"), val = int32(1)]; + tensor h_25_cast_fp16 = conv(dilations = h_25_dilations_0, groups = h_25_groups_0, pad = h_25_pad_0, pad_type = h_25_pad_type_0, strides = h_25_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_133_cast_fp16)[name = string("h_25_cast_fp16")]; + tensor x_99_cast_fp16 = add(x = x_97_cast_fp16, y = h_25_cast_fp16)[name = string("x_99_cast_fp16")]; + tensor key_cache_27_begin_0 = const()[name = string("key_cache_27_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_27_end_0 = const()[name = string("key_cache_27_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_27_end_mask_0 = const()[name = string("key_cache_27_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_27_cast_fp16 = slice_by_index(begin = key_cache_27_begin_0, end = key_cache_27_end_0, end_mask = key_cache_27_end_mask_0, x = layer_key_caches_5_cast_fp16)[name = string("key_cache_27_cast_fp16")]; + tensor value_cache_27_begin_0 = const()[name = string("value_cache_27_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_27_end_0 = const()[name = string("value_cache_27_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_27_end_mask_0 = const()[name = string("value_cache_27_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_27_cast_fp16 = slice_by_index(begin = value_cache_27_begin_0, end = value_cache_27_end_0, end_mask = value_cache_27_end_mask_0, x = layer_value_caches_5_cast_fp16)[name = string("value_cache_27_cast_fp16")]; + int32 var_3225 = const()[name = string("op_3225"), val = int32(2)]; + int32 var_3229 = const()[name = string("op_3229"), val = int32(3)]; + tensor var_3244_cast_fp16 = mul(x = x_99_cast_fp16, y = x_99_cast_fp16)[name = string("op_3244_cast_fp16")]; + tensor variance_107_axes_0 = const()[name = string("variance_107_axes_0"), val = tensor([1])]; + bool variance_107_keep_dims_0 = const()[name = string("variance_107_keep_dims_0"), val = bool(true)]; + tensor variance_107_cast_fp16 = reduce_mean(axes = variance_107_axes_0, keep_dims = variance_107_keep_dims_0, x = var_3244_cast_fp16)[name = string("variance_107_cast_fp16")]; + fp16 var_3247_to_fp16 = const()[name = string("op_3247_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3248_cast_fp16 = add(x = variance_107_cast_fp16, y = var_3247_to_fp16)[name = string("op_3248_cast_fp16")]; + fp32 var_3249_epsilon_0 = const()[name = string("op_3249_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3249_cast_fp16 = rsqrt(epsilon = var_3249_epsilon_0, x = var_3248_cast_fp16)[name = string("op_3249_cast_fp16")]; + tensor var_3250_cast_fp16 = mul(x = x_99_cast_fp16, y = var_3249_cast_fp16)[name = string("op_3250_cast_fp16")]; + tensor input_135_cast_fp16 = mul(x = var_3250_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_135_cast_fp16")]; + string q_79_pad_type_0 = const()[name = string("q_79_pad_type_0"), val = string("valid")]; + tensor q_79_strides_0 = const()[name = string("q_79_strides_0"), val = tensor([1, 1])]; + tensor q_79_pad_0 = const()[name = string("q_79_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_79_dilations_0 = const()[name = string("q_79_dilations_0"), val = tensor([1, 1])]; + int32 q_79_groups_0 = const()[name = string("q_79_groups_0"), val = int32(1)]; + tensor q_79_cast_fp16 = conv(dilations = q_79_dilations_0, groups = q_79_groups_0, pad = q_79_pad_0, pad_type = q_79_pad_type_0, strides = q_79_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_135_cast_fp16)[name = string("q_79_cast_fp16")]; + string k_79_pad_type_0 = const()[name = string("k_79_pad_type_0"), val = string("valid")]; + tensor k_79_strides_0 = const()[name = string("k_79_strides_0"), val = tensor([1, 1])]; + tensor k_79_pad_0 = const()[name = string("k_79_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_79_dilations_0 = const()[name = string("k_79_dilations_0"), val = tensor([1, 1])]; + int32 k_79_groups_0 = const()[name = string("k_79_groups_0"), val = int32(1)]; + tensor k_79_cast_fp16 = conv(dilations = k_79_dilations_0, groups = k_79_groups_0, pad = k_79_pad_0, pad_type = k_79_pad_type_0, strides = k_79_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_135_cast_fp16)[name = string("k_79_cast_fp16")]; + string v_27_pad_type_0 = const()[name = string("v_27_pad_type_0"), val = string("valid")]; + tensor v_27_strides_0 = const()[name = string("v_27_strides_0"), val = tensor([1, 1])]; + tensor v_27_pad_0 = const()[name = string("v_27_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_27_dilations_0 = const()[name = string("v_27_dilations_0"), val = tensor([1, 1])]; + int32 v_27_groups_0 = const()[name = string("v_27_groups_0"), val = int32(1)]; + tensor v_27_cast_fp16 = conv(dilations = v_27_dilations_0, groups = v_27_groups_0, pad = v_27_pad_0, pad_type = v_27_pad_type_0, strides = v_27_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_135_cast_fp16)[name = string("v_27_cast_fp16")]; + tensor var_3284 = const()[name = string("op_3284"), val = tensor([16, 128, 1, 1])]; + tensor x_101_cast_fp16 = reshape(shape = var_3284, x = q_79_cast_fp16)[name = string("x_101_cast_fp16")]; + tensor var_3287_cast_fp16 = mul(x = x_101_cast_fp16, y = x_101_cast_fp16)[name = string("op_3287_cast_fp16")]; + tensor variance_109_axes_0 = const()[name = string("variance_109_axes_0"), val = tensor([1])]; + bool variance_109_keep_dims_0 = const()[name = string("variance_109_keep_dims_0"), val = bool(true)]; + tensor variance_109_cast_fp16 = reduce_mean(axes = variance_109_axes_0, keep_dims = variance_109_keep_dims_0, x = var_3287_cast_fp16)[name = string("variance_109_cast_fp16")]; + fp16 var_3290_to_fp16 = const()[name = string("op_3290_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3291_cast_fp16 = add(x = variance_109_cast_fp16, y = var_3290_to_fp16)[name = string("op_3291_cast_fp16")]; + fp32 var_3292_epsilon_0 = const()[name = string("op_3292_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3292_cast_fp16 = rsqrt(epsilon = var_3292_epsilon_0, x = var_3291_cast_fp16)[name = string("op_3292_cast_fp16")]; + tensor var_3293_cast_fp16 = mul(x = x_101_cast_fp16, y = var_3292_cast_fp16)[name = string("op_3293_cast_fp16")]; + tensor q_81_cast_fp16 = mul(x = var_3293_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_81_cast_fp16")]; + tensor var_3295 = const()[name = string("op_3295"), val = tensor([8, 128, 1, 1])]; + tensor x_103_cast_fp16 = reshape(shape = var_3295, x = k_79_cast_fp16)[name = string("x_103_cast_fp16")]; + tensor var_3298_cast_fp16 = mul(x = x_103_cast_fp16, y = x_103_cast_fp16)[name = string("op_3298_cast_fp16")]; + tensor variance_111_axes_0 = const()[name = string("variance_111_axes_0"), val = tensor([1])]; + bool variance_111_keep_dims_0 = const()[name = string("variance_111_keep_dims_0"), val = bool(true)]; + tensor variance_111_cast_fp16 = reduce_mean(axes = variance_111_axes_0, keep_dims = variance_111_keep_dims_0, x = var_3298_cast_fp16)[name = string("variance_111_cast_fp16")]; + fp16 var_3301_to_fp16 = const()[name = string("op_3301_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3302_cast_fp16 = add(x = variance_111_cast_fp16, y = var_3301_to_fp16)[name = string("op_3302_cast_fp16")]; + fp32 var_3303_epsilon_0 = const()[name = string("op_3303_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3303_cast_fp16 = rsqrt(epsilon = var_3303_epsilon_0, x = var_3302_cast_fp16)[name = string("op_3303_cast_fp16")]; + tensor var_3304_cast_fp16 = mul(x = x_103_cast_fp16, y = var_3303_cast_fp16)[name = string("op_3304_cast_fp16")]; + tensor k_81_cast_fp16 = mul(x = var_3304_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_81_cast_fp16")]; + tensor var_3306 = const()[name = string("op_3306"), val = tensor([1, 16, 128, 1])]; + tensor z_53_cast_fp16 = reshape(shape = var_3306, x = q_81_cast_fp16)[name = string("z_53_cast_fp16")]; + tensor var_3308 = const()[name = string("op_3308"), val = tensor([1, 8, 128, 1])]; + tensor z_55_cast_fp16 = reshape(shape = var_3308, x = k_81_cast_fp16)[name = string("z_55_cast_fp16")]; + tensor z1_53_begin_0 = const()[name = string("z1_53_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_53_end_0 = const()[name = string("z1_53_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_53_end_mask_0 = const()[name = string("z1_53_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_53_cast_fp16 = slice_by_index(begin = z1_53_begin_0, end = z1_53_end_0, end_mask = z1_53_end_mask_0, x = z_53_cast_fp16)[name = string("z1_53_cast_fp16")]; + tensor z2_53_begin_0 = const()[name = string("z2_53_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_53_end_0 = const()[name = string("z2_53_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_53_end_mask_0 = const()[name = string("z2_53_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_53_cast_fp16 = slice_by_index(begin = z2_53_begin_0, end = z2_53_end_0, end_mask = z2_53_end_mask_0, x = z_53_cast_fp16)[name = string("z2_53_cast_fp16")]; + tensor var_3316_cast_fp16 = mul(x = z_53_cast_fp16, y = cos_21_to_fp16)[name = string("op_3316_cast_fp16")]; + fp16 const_29_promoted_to_fp16 = const()[name = string("const_29_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3317_cast_fp16 = mul(x = z2_53_cast_fp16, y = const_29_promoted_to_fp16)[name = string("op_3317_cast_fp16")]; + bool var_3319_interleave_0 = const()[name = string("op_3319_interleave_0"), val = bool(false)]; + tensor var_3319_cast_fp16 = concat(axis = var_3225, interleave = var_3319_interleave_0, values = (var_3317_cast_fp16, z1_53_cast_fp16))[name = string("op_3319_cast_fp16")]; + tensor var_3320_cast_fp16 = mul(x = var_3319_cast_fp16, y = sin_21_to_fp16)[name = string("op_3320_cast_fp16")]; + tensor q_83_cast_fp16 = add(x = var_3316_cast_fp16, y = var_3320_cast_fp16)[name = string("q_83_cast_fp16")]; + tensor z1_55_begin_0 = const()[name = string("z1_55_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_55_end_0 = const()[name = string("z1_55_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_55_end_mask_0 = const()[name = string("z1_55_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_55_cast_fp16 = slice_by_index(begin = z1_55_begin_0, end = z1_55_end_0, end_mask = z1_55_end_mask_0, x = z_55_cast_fp16)[name = string("z1_55_cast_fp16")]; + tensor z2_55_begin_0 = const()[name = string("z2_55_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_55_end_0 = const()[name = string("z2_55_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_55_end_mask_0 = const()[name = string("z2_55_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_55_cast_fp16 = slice_by_index(begin = z2_55_begin_0, end = z2_55_end_0, end_mask = z2_55_end_mask_0, x = z_55_cast_fp16)[name = string("z2_55_cast_fp16")]; + tensor var_3328_cast_fp16 = mul(x = z_55_cast_fp16, y = cos_21_to_fp16)[name = string("op_3328_cast_fp16")]; + fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3329_cast_fp16 = mul(x = z2_55_cast_fp16, y = const_30_promoted_to_fp16)[name = string("op_3329_cast_fp16")]; + bool var_3331_interleave_0 = const()[name = string("op_3331_interleave_0"), val = bool(false)]; + tensor var_3331_cast_fp16 = concat(axis = var_3225, interleave = var_3331_interleave_0, values = (var_3329_cast_fp16, z1_55_cast_fp16))[name = string("op_3331_cast_fp16")]; + tensor var_3332_cast_fp16 = mul(x = var_3331_cast_fp16, y = sin_21_to_fp16)[name = string("op_3332_cast_fp16")]; + tensor k_83_cast_fp16 = add(x = var_3328_cast_fp16, y = var_3332_cast_fp16)[name = string("k_83_cast_fp16")]; + tensor var_3334 = const()[name = string("op_3334"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_27_cast_fp16 = reshape(shape = var_3334, x = k_83_cast_fp16)[name = string("cur_key_27_cast_fp16")]; + tensor var_3336_to_fp16 = const()[name = string("op_3336_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634048)))]; + tensor var_3337_cast_fp16 = mul(x = key_cache_27_cast_fp16, y = var_3336_to_fp16)[name = string("op_3337_cast_fp16")]; + tensor upd_27_to_fp16 = const()[name = string("upd_27_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634176)))]; + tensor var_3338_cast_fp16 = mul(x = cur_key_27_cast_fp16, y = upd_27_to_fp16)[name = string("op_3338_cast_fp16")]; + tensor key_27_cast_fp16 = add(x = var_3337_cast_fp16, y = var_3338_cast_fp16)[name = string("key_27_cast_fp16")]; + tensor var_3340_to_fp16 = const()[name = string("op_3340_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634048)))]; + tensor var_3341_cast_fp16 = mul(x = value_cache_27_cast_fp16, y = var_3340_to_fp16)[name = string("op_3341_cast_fp16")]; + tensor var_3342_cast_fp16 = mul(x = v_27_cast_fp16, y = upd_27_to_fp16)[name = string("op_3342_cast_fp16")]; + tensor value_27_cast_fp16 = add(x = var_3341_cast_fp16, y = var_3342_cast_fp16)[name = string("value_27_cast_fp16")]; + tensor var_3344 = const()[name = string("op_3344"), val = tensor([1, 8, 128, 16])]; + tensor kh_53_cast_fp16 = reshape(shape = var_3344, x = key_27_cast_fp16)[name = string("kh_53_cast_fp16")]; + tensor var_3346 = const()[name = string("op_3346"), val = tensor([1, 8, 128, 16])]; + tensor vh_53_cast_fp16 = reshape(shape = var_3346, x = value_27_cast_fp16)[name = string("vh_53_cast_fp16")]; + tensor transpose_52_perm_0 = const()[name = string("transpose_52_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_26_reps_0 = const()[name = string("tile_26_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_52_cast_fp16 = transpose(perm = transpose_52_perm_0, x = kh_53_cast_fp16)[name = string("transpose_401")]; + tensor tile_26_cast_fp16 = tile(reps = tile_26_reps_0, x = transpose_52_cast_fp16)[name = string("tile_26_cast_fp16")]; + tensor concat_65 = const()[name = string("concat_65"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_52_cast_fp16 = reshape(shape = concat_65, x = tile_26_cast_fp16)[name = string("reshape_52_cast_fp16")]; + tensor transpose_53_perm_0 = const()[name = string("transpose_53_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_66 = const()[name = string("concat_66"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_53_cast_fp16 = transpose(perm = transpose_53_perm_0, x = reshape_52_cast_fp16)[name = string("transpose_400")]; + tensor reshape_53_cast_fp16 = reshape(shape = concat_66, x = transpose_53_cast_fp16)[name = string("reshape_53_cast_fp16")]; + tensor transpose_54_perm_0 = const()[name = string("transpose_54_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_27_reps_0 = const()[name = string("tile_27_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_54_cast_fp16 = transpose(perm = transpose_54_perm_0, x = vh_53_cast_fp16)[name = string("transpose_399")]; + tensor tile_27_cast_fp16 = tile(reps = tile_27_reps_0, x = transpose_54_cast_fp16)[name = string("tile_27_cast_fp16")]; + tensor concat_67 = const()[name = string("concat_67"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_54_cast_fp16 = reshape(shape = concat_67, x = tile_27_cast_fp16)[name = string("reshape_54_cast_fp16")]; + tensor transpose_55_perm_0 = const()[name = string("transpose_55_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_68 = const()[name = string("concat_68"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_55_cast_fp16 = transpose(perm = transpose_55_perm_0, x = reshape_54_cast_fp16)[name = string("transpose_398")]; + tensor reshape_55_cast_fp16 = reshape(shape = concat_68, x = transpose_55_cast_fp16)[name = string("reshape_55_cast_fp16")]; + fp16 var_3350_to_fp16 = const()[name = string("op_3350_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_3351_cast_fp16 = mul(x = q_83_cast_fp16, y = var_3350_to_fp16)[name = string("op_3351_cast_fp16")]; + tensor transpose_369_perm_0 = const()[name = string("transpose_369_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_55_transpose_x_1 = const()[name = string("w_55_transpose_x_1"), val = bool(true)]; + bool w_55_transpose_y_1 = const()[name = string("w_55_transpose_y_1"), val = bool(false)]; + tensor transpose_369_cast_fp16 = transpose(perm = transpose_369_perm_0, x = reshape_53_cast_fp16)[name = string("transpose_397")]; + tensor w_55_cast_fp16 = matmul(transpose_x = w_55_transpose_x_1, transpose_y = w_55_transpose_y_1, x = var_3351_cast_fp16, y = transpose_369_cast_fp16)[name = string("w_55_cast_fp16")]; + tensor pad_27_to_fp16 = const()[name = string("pad_27_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634304)))]; + tensor var_3354_cast_fp16 = add(x = w_55_cast_fp16, y = pad_27_to_fp16)[name = string("op_3354_cast_fp16")]; + tensor w_57_cast_fp16 = softmax(axis = var_3229, x = var_3354_cast_fp16)[name = string("w_57_cast_fp16")]; + tensor transpose_370_perm_0 = const()[name = string("transpose_370_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_27_transpose_x_1 = const()[name = string("attn_27_transpose_x_1"), val = bool(false)]; + bool attn_27_transpose_y_1 = const()[name = string("attn_27_transpose_y_1"), val = bool(true)]; + tensor transpose_370_cast_fp16 = transpose(perm = transpose_370_perm_0, x = reshape_55_cast_fp16)[name = string("transpose_396")]; + tensor attn_27_cast_fp16 = matmul(transpose_x = attn_27_transpose_x_1, transpose_y = attn_27_transpose_y_1, x = transpose_370_cast_fp16, y = w_57_cast_fp16)[name = string("attn_27_cast_fp16")]; + tensor var_3358 = const()[name = string("op_3358"), val = tensor([1, 2048, 1, 1])]; + tensor input_137_cast_fp16 = reshape(shape = var_3358, x = attn_27_cast_fp16)[name = string("input_137_cast_fp16")]; + string attn_output_27_pad_type_0 = const()[name = string("attn_output_27_pad_type_0"), val = string("valid")]; + tensor attn_output_27_strides_0 = const()[name = string("attn_output_27_strides_0"), val = tensor([1, 1])]; + tensor attn_output_27_pad_0 = const()[name = string("attn_output_27_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_27_dilations_0 = const()[name = string("attn_output_27_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_27_groups_0 = const()[name = string("attn_output_27_groups_0"), val = int32(1)]; + tensor attn_output_27_cast_fp16 = conv(dilations = attn_output_27_dilations_0, groups = attn_output_27_groups_0, pad = attn_output_27_pad_0, pad_type = attn_output_27_pad_type_0, strides = attn_output_27_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_137_cast_fp16)[name = string("attn_output_27_cast_fp16")]; + tensor x_105_cast_fp16 = add(x = x_99_cast_fp16, y = attn_output_27_cast_fp16)[name = string("x_105_cast_fp16")]; + tensor var_3372_cast_fp16 = mul(x = x_105_cast_fp16, y = x_105_cast_fp16)[name = string("op_3372_cast_fp16")]; + tensor variance_113_axes_0 = const()[name = string("variance_113_axes_0"), val = tensor([1])]; + bool variance_113_keep_dims_0 = const()[name = string("variance_113_keep_dims_0"), val = bool(true)]; + tensor variance_113_cast_fp16 = reduce_mean(axes = variance_113_axes_0, keep_dims = variance_113_keep_dims_0, x = var_3372_cast_fp16)[name = string("variance_113_cast_fp16")]; + fp16 var_3375_to_fp16 = const()[name = string("op_3375_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3376_cast_fp16 = add(x = variance_113_cast_fp16, y = var_3375_to_fp16)[name = string("op_3376_cast_fp16")]; + fp32 var_3377_epsilon_0 = const()[name = string("op_3377_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3377_cast_fp16 = rsqrt(epsilon = var_3377_epsilon_0, x = var_3376_cast_fp16)[name = string("op_3377_cast_fp16")]; + tensor var_3378_cast_fp16 = mul(x = x_105_cast_fp16, y = var_3377_cast_fp16)[name = string("op_3378_cast_fp16")]; + tensor input_139_cast_fp16 = mul(x = var_3378_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_139_cast_fp16")]; + string input_141_pad_type_0 = const()[name = string("input_141_pad_type_0"), val = string("valid")]; + tensor input_141_strides_0 = const()[name = string("input_141_strides_0"), val = tensor([1, 1])]; + tensor input_141_pad_0 = const()[name = string("input_141_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_141_dilations_0 = const()[name = string("input_141_dilations_0"), val = tensor([1, 1])]; + int32 input_141_groups_0 = const()[name = string("input_141_groups_0"), val = int32(1)]; + tensor input_141_cast_fp16 = conv(dilations = input_141_dilations_0, groups = input_141_groups_0, pad = input_141_pad_0, pad_type = input_141_pad_type_0, strides = input_141_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_139_cast_fp16)[name = string("input_141_cast_fp16")]; + tensor var_3386_cast_fp16 = silu(x = input_141_cast_fp16)[name = string("op_3386_cast_fp16")]; + string var_3392_pad_type_0 = const()[name = string("op_3392_pad_type_0"), val = string("valid")]; + tensor var_3392_strides_0 = const()[name = string("op_3392_strides_0"), val = tensor([1, 1])]; + tensor var_3392_pad_0 = const()[name = string("op_3392_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3392_dilations_0 = const()[name = string("op_3392_dilations_0"), val = tensor([1, 1])]; + int32 var_3392_groups_0 = const()[name = string("op_3392_groups_0"), val = int32(1)]; + tensor var_3392_cast_fp16 = conv(dilations = var_3392_dilations_0, groups = var_3392_groups_0, pad = var_3392_pad_0, pad_type = var_3392_pad_type_0, strides = var_3392_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_139_cast_fp16)[name = string("op_3392_cast_fp16")]; + tensor input_143_cast_fp16 = mul(x = var_3386_cast_fp16, y = var_3392_cast_fp16)[name = string("input_143_cast_fp16")]; + string h_27_pad_type_0 = const()[name = string("h_27_pad_type_0"), val = string("valid")]; + tensor h_27_strides_0 = const()[name = string("h_27_strides_0"), val = tensor([1, 1])]; + tensor h_27_pad_0 = const()[name = string("h_27_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_27_dilations_0 = const()[name = string("h_27_dilations_0"), val = tensor([1, 1])]; + int32 h_27_groups_0 = const()[name = string("h_27_groups_0"), val = int32(1)]; + tensor h_27_cast_fp16 = conv(dilations = h_27_dilations_0, groups = h_27_groups_0, pad = h_27_pad_0, pad_type = h_27_pad_type_0, strides = h_27_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_143_cast_fp16)[name = string("h_27_cast_fp16")]; + tensor x_107_cast_fp16 = add(x = x_105_cast_fp16, y = h_27_cast_fp16)[name = string("x_107_cast_fp16")]; + tensor key_cache_29_begin_0 = const()[name = string("key_cache_29_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_29_end_0 = const()[name = string("key_cache_29_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_29_end_mask_0 = const()[name = string("key_cache_29_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_29_cast_fp16 = slice_by_index(begin = key_cache_29_begin_0, end = key_cache_29_end_0, end_mask = key_cache_29_end_mask_0, x = layer_key_caches_5_cast_fp16)[name = string("key_cache_29_cast_fp16")]; + tensor value_cache_29_begin_0 = const()[name = string("value_cache_29_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_29_end_0 = const()[name = string("value_cache_29_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_29_end_mask_0 = const()[name = string("value_cache_29_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_29_cast_fp16 = slice_by_index(begin = value_cache_29_begin_0, end = value_cache_29_end_0, end_mask = value_cache_29_end_mask_0, x = layer_value_caches_5_cast_fp16)[name = string("value_cache_29_cast_fp16")]; + int32 var_3445 = const()[name = string("op_3445"), val = int32(2)]; + int32 var_3449 = const()[name = string("op_3449"), val = int32(3)]; + tensor var_3464_cast_fp16 = mul(x = x_107_cast_fp16, y = x_107_cast_fp16)[name = string("op_3464_cast_fp16")]; + tensor variance_115_axes_0 = const()[name = string("variance_115_axes_0"), val = tensor([1])]; + bool variance_115_keep_dims_0 = const()[name = string("variance_115_keep_dims_0"), val = bool(true)]; + tensor variance_115_cast_fp16 = reduce_mean(axes = variance_115_axes_0, keep_dims = variance_115_keep_dims_0, x = var_3464_cast_fp16)[name = string("variance_115_cast_fp16")]; + fp16 var_3467_to_fp16 = const()[name = string("op_3467_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3468_cast_fp16 = add(x = variance_115_cast_fp16, y = var_3467_to_fp16)[name = string("op_3468_cast_fp16")]; + fp32 var_3469_epsilon_0 = const()[name = string("op_3469_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3469_cast_fp16 = rsqrt(epsilon = var_3469_epsilon_0, x = var_3468_cast_fp16)[name = string("op_3469_cast_fp16")]; + tensor var_3470_cast_fp16 = mul(x = x_107_cast_fp16, y = var_3469_cast_fp16)[name = string("op_3470_cast_fp16")]; + tensor input_145_cast_fp16 = mul(x = var_3470_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_145_cast_fp16")]; + string q_85_pad_type_0 = const()[name = string("q_85_pad_type_0"), val = string("valid")]; + tensor q_85_strides_0 = const()[name = string("q_85_strides_0"), val = tensor([1, 1])]; + tensor q_85_pad_0 = const()[name = string("q_85_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_85_dilations_0 = const()[name = string("q_85_dilations_0"), val = tensor([1, 1])]; + int32 q_85_groups_0 = const()[name = string("q_85_groups_0"), val = int32(1)]; + tensor q_85_cast_fp16 = conv(dilations = q_85_dilations_0, groups = q_85_groups_0, pad = q_85_pad_0, pad_type = q_85_pad_type_0, strides = q_85_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_145_cast_fp16)[name = string("q_85_cast_fp16")]; + string k_85_pad_type_0 = const()[name = string("k_85_pad_type_0"), val = string("valid")]; + tensor k_85_strides_0 = const()[name = string("k_85_strides_0"), val = tensor([1, 1])]; + tensor k_85_pad_0 = const()[name = string("k_85_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_85_dilations_0 = const()[name = string("k_85_dilations_0"), val = tensor([1, 1])]; + int32 k_85_groups_0 = const()[name = string("k_85_groups_0"), val = int32(1)]; + tensor k_85_cast_fp16 = conv(dilations = k_85_dilations_0, groups = k_85_groups_0, pad = k_85_pad_0, pad_type = k_85_pad_type_0, strides = k_85_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_145_cast_fp16)[name = string("k_85_cast_fp16")]; + string v_29_pad_type_0 = const()[name = string("v_29_pad_type_0"), val = string("valid")]; + tensor v_29_strides_0 = const()[name = string("v_29_strides_0"), val = tensor([1, 1])]; + tensor v_29_pad_0 = const()[name = string("v_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_29_dilations_0 = const()[name = string("v_29_dilations_0"), val = tensor([1, 1])]; + int32 v_29_groups_0 = const()[name = string("v_29_groups_0"), val = int32(1)]; + tensor v_29_cast_fp16 = conv(dilations = v_29_dilations_0, groups = v_29_groups_0, pad = v_29_pad_0, pad_type = v_29_pad_type_0, strides = v_29_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_145_cast_fp16)[name = string("v_29_cast_fp16")]; + tensor var_3504 = const()[name = string("op_3504"), val = tensor([16, 128, 1, 1])]; + tensor x_109_cast_fp16 = reshape(shape = var_3504, x = q_85_cast_fp16)[name = string("x_109_cast_fp16")]; + tensor var_3507_cast_fp16 = mul(x = x_109_cast_fp16, y = x_109_cast_fp16)[name = string("op_3507_cast_fp16")]; + tensor variance_117_axes_0 = const()[name = string("variance_117_axes_0"), val = tensor([1])]; + bool variance_117_keep_dims_0 = const()[name = string("variance_117_keep_dims_0"), val = bool(true)]; + tensor variance_117_cast_fp16 = reduce_mean(axes = variance_117_axes_0, keep_dims = variance_117_keep_dims_0, x = var_3507_cast_fp16)[name = string("variance_117_cast_fp16")]; + fp16 var_3510_to_fp16 = const()[name = string("op_3510_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3511_cast_fp16 = add(x = variance_117_cast_fp16, y = var_3510_to_fp16)[name = string("op_3511_cast_fp16")]; + fp32 var_3512_epsilon_0 = const()[name = string("op_3512_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3512_cast_fp16 = rsqrt(epsilon = var_3512_epsilon_0, x = var_3511_cast_fp16)[name = string("op_3512_cast_fp16")]; + tensor var_3513_cast_fp16 = mul(x = x_109_cast_fp16, y = var_3512_cast_fp16)[name = string("op_3513_cast_fp16")]; + tensor q_87_cast_fp16 = mul(x = var_3513_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_87_cast_fp16")]; + tensor var_3515 = const()[name = string("op_3515"), val = tensor([8, 128, 1, 1])]; + tensor x_111_cast_fp16 = reshape(shape = var_3515, x = k_85_cast_fp16)[name = string("x_111_cast_fp16")]; + tensor var_3518_cast_fp16 = mul(x = x_111_cast_fp16, y = x_111_cast_fp16)[name = string("op_3518_cast_fp16")]; + tensor variance_119_axes_0 = const()[name = string("variance_119_axes_0"), val = tensor([1])]; + bool variance_119_keep_dims_0 = const()[name = string("variance_119_keep_dims_0"), val = bool(true)]; + tensor variance_119_cast_fp16 = reduce_mean(axes = variance_119_axes_0, keep_dims = variance_119_keep_dims_0, x = var_3518_cast_fp16)[name = string("variance_119_cast_fp16")]; + fp16 var_3521_to_fp16 = const()[name = string("op_3521_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3522_cast_fp16 = add(x = variance_119_cast_fp16, y = var_3521_to_fp16)[name = string("op_3522_cast_fp16")]; + fp32 var_3523_epsilon_0 = const()[name = string("op_3523_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3523_cast_fp16 = rsqrt(epsilon = var_3523_epsilon_0, x = var_3522_cast_fp16)[name = string("op_3523_cast_fp16")]; + tensor var_3524_cast_fp16 = mul(x = x_111_cast_fp16, y = var_3523_cast_fp16)[name = string("op_3524_cast_fp16")]; + tensor k_87_cast_fp16 = mul(x = var_3524_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_87_cast_fp16")]; + tensor var_3526 = const()[name = string("op_3526"), val = tensor([1, 16, 128, 1])]; + tensor z_57_cast_fp16 = reshape(shape = var_3526, x = q_87_cast_fp16)[name = string("z_57_cast_fp16")]; + tensor var_3528 = const()[name = string("op_3528"), val = tensor([1, 8, 128, 1])]; + tensor z_59_cast_fp16 = reshape(shape = var_3528, x = k_87_cast_fp16)[name = string("z_59_cast_fp16")]; + tensor z1_57_begin_0 = const()[name = string("z1_57_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_57_end_0 = const()[name = string("z1_57_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_57_end_mask_0 = const()[name = string("z1_57_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_57_cast_fp16 = slice_by_index(begin = z1_57_begin_0, end = z1_57_end_0, end_mask = z1_57_end_mask_0, x = z_57_cast_fp16)[name = string("z1_57_cast_fp16")]; + tensor z2_57_begin_0 = const()[name = string("z2_57_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_57_end_0 = const()[name = string("z2_57_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_57_end_mask_0 = const()[name = string("z2_57_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_57_cast_fp16 = slice_by_index(begin = z2_57_begin_0, end = z2_57_end_0, end_mask = z2_57_end_mask_0, x = z_57_cast_fp16)[name = string("z2_57_cast_fp16")]; + tensor var_3536_cast_fp16 = mul(x = z_57_cast_fp16, y = cos_21_to_fp16)[name = string("op_3536_cast_fp16")]; + fp16 const_31_promoted_to_fp16 = const()[name = string("const_31_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3537_cast_fp16 = mul(x = z2_57_cast_fp16, y = const_31_promoted_to_fp16)[name = string("op_3537_cast_fp16")]; + bool var_3539_interleave_0 = const()[name = string("op_3539_interleave_0"), val = bool(false)]; + tensor var_3539_cast_fp16 = concat(axis = var_3445, interleave = var_3539_interleave_0, values = (var_3537_cast_fp16, z1_57_cast_fp16))[name = string("op_3539_cast_fp16")]; + tensor var_3540_cast_fp16 = mul(x = var_3539_cast_fp16, y = sin_21_to_fp16)[name = string("op_3540_cast_fp16")]; + tensor q_89_cast_fp16 = add(x = var_3536_cast_fp16, y = var_3540_cast_fp16)[name = string("q_89_cast_fp16")]; + tensor z1_59_begin_0 = const()[name = string("z1_59_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_59_end_0 = const()[name = string("z1_59_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_59_end_mask_0 = const()[name = string("z1_59_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_59_cast_fp16 = slice_by_index(begin = z1_59_begin_0, end = z1_59_end_0, end_mask = z1_59_end_mask_0, x = z_59_cast_fp16)[name = string("z1_59_cast_fp16")]; + tensor z2_59_begin_0 = const()[name = string("z2_59_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_59_end_0 = const()[name = string("z2_59_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_59_end_mask_0 = const()[name = string("z2_59_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_59_cast_fp16 = slice_by_index(begin = z2_59_begin_0, end = z2_59_end_0, end_mask = z2_59_end_mask_0, x = z_59_cast_fp16)[name = string("z2_59_cast_fp16")]; + tensor var_3548_cast_fp16 = mul(x = z_59_cast_fp16, y = cos_21_to_fp16)[name = string("op_3548_cast_fp16")]; + fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3549_cast_fp16 = mul(x = z2_59_cast_fp16, y = const_32_promoted_to_fp16)[name = string("op_3549_cast_fp16")]; + bool var_3551_interleave_0 = const()[name = string("op_3551_interleave_0"), val = bool(false)]; + tensor var_3551_cast_fp16 = concat(axis = var_3445, interleave = var_3551_interleave_0, values = (var_3549_cast_fp16, z1_59_cast_fp16))[name = string("op_3551_cast_fp16")]; + tensor var_3552_cast_fp16 = mul(x = var_3551_cast_fp16, y = sin_21_to_fp16)[name = string("op_3552_cast_fp16")]; + tensor k_89_cast_fp16 = add(x = var_3548_cast_fp16, y = var_3552_cast_fp16)[name = string("k_89_cast_fp16")]; + tensor var_3554 = const()[name = string("op_3554"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_29_cast_fp16 = reshape(shape = var_3554, x = k_89_cast_fp16)[name = string("cur_key_29_cast_fp16")]; + tensor var_3556_to_fp16 = const()[name = string("op_3556_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634048)))]; + tensor var_3557_cast_fp16 = mul(x = key_cache_29_cast_fp16, y = var_3556_to_fp16)[name = string("op_3557_cast_fp16")]; + tensor upd_29_to_fp16 = const()[name = string("upd_29_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634176)))]; + tensor var_3558_cast_fp16 = mul(x = cur_key_29_cast_fp16, y = upd_29_to_fp16)[name = string("op_3558_cast_fp16")]; + tensor key_29_cast_fp16 = add(x = var_3557_cast_fp16, y = var_3558_cast_fp16)[name = string("key_29_cast_fp16")]; + tensor var_3560_to_fp16 = const()[name = string("op_3560_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634048)))]; + tensor var_3561_cast_fp16 = mul(x = value_cache_29_cast_fp16, y = var_3560_to_fp16)[name = string("op_3561_cast_fp16")]; + tensor var_3562_cast_fp16 = mul(x = v_29_cast_fp16, y = upd_29_to_fp16)[name = string("op_3562_cast_fp16")]; + tensor value_29_cast_fp16 = add(x = var_3561_cast_fp16, y = var_3562_cast_fp16)[name = string("value_29_cast_fp16")]; + tensor var_3564 = const()[name = string("op_3564"), val = tensor([1, 8, 128, 16])]; + tensor kh_57_cast_fp16 = reshape(shape = var_3564, x = key_29_cast_fp16)[name = string("kh_57_cast_fp16")]; + tensor var_3566 = const()[name = string("op_3566"), val = tensor([1, 8, 128, 16])]; + tensor vh_57_cast_fp16 = reshape(shape = var_3566, x = value_29_cast_fp16)[name = string("vh_57_cast_fp16")]; + tensor transpose_56_perm_0 = const()[name = string("transpose_56_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_28_reps_0 = const()[name = string("tile_28_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_56_cast_fp16 = transpose(perm = transpose_56_perm_0, x = kh_57_cast_fp16)[name = string("transpose_395")]; + tensor tile_28_cast_fp16 = tile(reps = tile_28_reps_0, x = transpose_56_cast_fp16)[name = string("tile_28_cast_fp16")]; + tensor concat_69 = const()[name = string("concat_69"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_56_cast_fp16 = reshape(shape = concat_69, x = tile_28_cast_fp16)[name = string("reshape_56_cast_fp16")]; + tensor transpose_57_perm_0 = const()[name = string("transpose_57_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_70 = const()[name = string("concat_70"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_57_cast_fp16 = transpose(perm = transpose_57_perm_0, x = reshape_56_cast_fp16)[name = string("transpose_394")]; + tensor reshape_57_cast_fp16 = reshape(shape = concat_70, x = transpose_57_cast_fp16)[name = string("reshape_57_cast_fp16")]; + tensor transpose_58_perm_0 = const()[name = string("transpose_58_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_29_reps_0 = const()[name = string("tile_29_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_58_cast_fp16 = transpose(perm = transpose_58_perm_0, x = vh_57_cast_fp16)[name = string("transpose_393")]; + tensor tile_29_cast_fp16 = tile(reps = tile_29_reps_0, x = transpose_58_cast_fp16)[name = string("tile_29_cast_fp16")]; + tensor concat_71 = const()[name = string("concat_71"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_58_cast_fp16 = reshape(shape = concat_71, x = tile_29_cast_fp16)[name = string("reshape_58_cast_fp16")]; + tensor transpose_59_perm_0 = const()[name = string("transpose_59_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_72 = const()[name = string("concat_72"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_59_cast_fp16 = transpose(perm = transpose_59_perm_0, x = reshape_58_cast_fp16)[name = string("transpose_392")]; + tensor reshape_59_cast_fp16 = reshape(shape = concat_72, x = transpose_59_cast_fp16)[name = string("reshape_59_cast_fp16")]; + fp16 var_3570_to_fp16 = const()[name = string("op_3570_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_3571_cast_fp16 = mul(x = q_89_cast_fp16, y = var_3570_to_fp16)[name = string("op_3571_cast_fp16")]; + tensor transpose_373_perm_0 = const()[name = string("transpose_373_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_59_transpose_x_1 = const()[name = string("w_59_transpose_x_1"), val = bool(true)]; + bool w_59_transpose_y_1 = const()[name = string("w_59_transpose_y_1"), val = bool(false)]; + tensor transpose_373_cast_fp16 = transpose(perm = transpose_373_perm_0, x = reshape_57_cast_fp16)[name = string("transpose_391")]; + tensor w_59_cast_fp16 = matmul(transpose_x = w_59_transpose_x_1, transpose_y = w_59_transpose_y_1, x = var_3571_cast_fp16, y = transpose_373_cast_fp16)[name = string("w_59_cast_fp16")]; + tensor pad_29_to_fp16 = const()[name = string("pad_29_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634304)))]; + tensor var_3574_cast_fp16 = add(x = w_59_cast_fp16, y = pad_29_to_fp16)[name = string("op_3574_cast_fp16")]; + tensor w_61_cast_fp16 = softmax(axis = var_3449, x = var_3574_cast_fp16)[name = string("w_61_cast_fp16")]; + tensor transpose_374_perm_0 = const()[name = string("transpose_374_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_29_transpose_x_1 = const()[name = string("attn_29_transpose_x_1"), val = bool(false)]; + bool attn_29_transpose_y_1 = const()[name = string("attn_29_transpose_y_1"), val = bool(true)]; + tensor transpose_374_cast_fp16 = transpose(perm = transpose_374_perm_0, x = reshape_59_cast_fp16)[name = string("transpose_390")]; + tensor attn_29_cast_fp16 = matmul(transpose_x = attn_29_transpose_x_1, transpose_y = attn_29_transpose_y_1, x = transpose_374_cast_fp16, y = w_61_cast_fp16)[name = string("attn_29_cast_fp16")]; + tensor var_3578 = const()[name = string("op_3578"), val = tensor([1, 2048, 1, 1])]; + tensor input_147_cast_fp16 = reshape(shape = var_3578, x = attn_29_cast_fp16)[name = string("input_147_cast_fp16")]; + string attn_output_29_pad_type_0 = const()[name = string("attn_output_29_pad_type_0"), val = string("valid")]; + tensor attn_output_29_strides_0 = const()[name = string("attn_output_29_strides_0"), val = tensor([1, 1])]; + tensor attn_output_29_pad_0 = const()[name = string("attn_output_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_29_dilations_0 = const()[name = string("attn_output_29_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_29_groups_0 = const()[name = string("attn_output_29_groups_0"), val = int32(1)]; + tensor attn_output_29_cast_fp16 = conv(dilations = attn_output_29_dilations_0, groups = attn_output_29_groups_0, pad = attn_output_29_pad_0, pad_type = attn_output_29_pad_type_0, strides = attn_output_29_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_147_cast_fp16)[name = string("attn_output_29_cast_fp16")]; + tensor x_113_cast_fp16 = add(x = x_107_cast_fp16, y = attn_output_29_cast_fp16)[name = string("x_113_cast_fp16")]; + tensor var_3592_cast_fp16 = mul(x = x_113_cast_fp16, y = x_113_cast_fp16)[name = string("op_3592_cast_fp16")]; + tensor variance_121_axes_0 = const()[name = string("variance_121_axes_0"), val = tensor([1])]; + bool variance_121_keep_dims_0 = const()[name = string("variance_121_keep_dims_0"), val = bool(true)]; + tensor variance_121_cast_fp16 = reduce_mean(axes = variance_121_axes_0, keep_dims = variance_121_keep_dims_0, x = var_3592_cast_fp16)[name = string("variance_121_cast_fp16")]; + fp16 var_3595_to_fp16 = const()[name = string("op_3595_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3596_cast_fp16 = add(x = variance_121_cast_fp16, y = var_3595_to_fp16)[name = string("op_3596_cast_fp16")]; + fp32 var_3597_epsilon_0 = const()[name = string("op_3597_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3597_cast_fp16 = rsqrt(epsilon = var_3597_epsilon_0, x = var_3596_cast_fp16)[name = string("op_3597_cast_fp16")]; + tensor var_3598_cast_fp16 = mul(x = x_113_cast_fp16, y = var_3597_cast_fp16)[name = string("op_3598_cast_fp16")]; + tensor input_149_cast_fp16 = mul(x = var_3598_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_149_cast_fp16")]; + string input_151_pad_type_0 = const()[name = string("input_151_pad_type_0"), val = string("valid")]; + tensor input_151_strides_0 = const()[name = string("input_151_strides_0"), val = tensor([1, 1])]; + tensor input_151_pad_0 = const()[name = string("input_151_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_151_dilations_0 = const()[name = string("input_151_dilations_0"), val = tensor([1, 1])]; + int32 input_151_groups_0 = const()[name = string("input_151_groups_0"), val = int32(1)]; + tensor input_151_cast_fp16 = conv(dilations = input_151_dilations_0, groups = input_151_groups_0, pad = input_151_pad_0, pad_type = input_151_pad_type_0, strides = input_151_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_149_cast_fp16)[name = string("input_151_cast_fp16")]; + tensor var_3606_cast_fp16 = silu(x = input_151_cast_fp16)[name = string("op_3606_cast_fp16")]; + string var_3612_pad_type_0 = const()[name = string("op_3612_pad_type_0"), val = string("valid")]; + tensor var_3612_strides_0 = const()[name = string("op_3612_strides_0"), val = tensor([1, 1])]; + tensor var_3612_pad_0 = const()[name = string("op_3612_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3612_dilations_0 = const()[name = string("op_3612_dilations_0"), val = tensor([1, 1])]; + int32 var_3612_groups_0 = const()[name = string("op_3612_groups_0"), val = int32(1)]; + tensor var_3612_cast_fp16 = conv(dilations = var_3612_dilations_0, groups = var_3612_groups_0, pad = var_3612_pad_0, pad_type = var_3612_pad_type_0, strides = var_3612_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_149_cast_fp16)[name = string("op_3612_cast_fp16")]; + tensor input_153_cast_fp16 = mul(x = var_3606_cast_fp16, y = var_3612_cast_fp16)[name = string("input_153_cast_fp16")]; + string h_29_pad_type_0 = const()[name = string("h_29_pad_type_0"), val = string("valid")]; + tensor h_29_strides_0 = const()[name = string("h_29_strides_0"), val = tensor([1, 1])]; + tensor h_29_pad_0 = const()[name = string("h_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_29_dilations_0 = const()[name = string("h_29_dilations_0"), val = tensor([1, 1])]; + int32 h_29_groups_0 = const()[name = string("h_29_groups_0"), val = int32(1)]; + tensor h_29_cast_fp16 = conv(dilations = h_29_dilations_0, groups = h_29_groups_0, pad = h_29_pad_0, pad_type = h_29_pad_type_0, strides = h_29_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_153_cast_fp16)[name = string("h_29_cast_fp16")]; + tensor inputs_3_cast_fp16 = add(x = x_113_cast_fp16, y = h_29_cast_fp16)[name = string("inputs_3_cast_fp16")]; + int32 var_3640 = const()[name = string("op_3640"), val = int32(1)]; + bool layer_key_caches_7_interleave_0 = const()[name = string("layer_key_caches_7_interleave_0"), val = bool(false)]; + tensor layer_key_caches_7_cast_fp16 = concat(axis = var_3640, interleave = layer_key_caches_7_interleave_0, values = (key_21_cast_fp16, key_23_cast_fp16, key_25_cast_fp16, key_27_cast_fp16, key_29_cast_fp16))[name = string("layer_key_caches_7_cast_fp16")]; + int32 var_3643 = const()[name = string("op_3643"), val = int32(1)]; + bool layer_value_caches_7_interleave_0 = const()[name = string("layer_value_caches_7_interleave_0"), val = bool(false)]; + tensor layer_value_caches_7_cast_fp16 = concat(axis = var_3643, interleave = layer_value_caches_7_interleave_0, values = (value_21_cast_fp16, value_23_cast_fp16, value_25_cast_fp16, value_27_cast_fp16, value_29_cast_fp16))[name = string("layer_value_caches_7_cast_fp16")]; + tensor inputs_sq_3_cast_fp16 = mul(x = inputs_3_cast_fp16, y = inputs_3_cast_fp16)[name = string("inputs_sq_3_cast_fp16")]; + tensor variance_123_axes_0 = const()[name = string("variance_123_axes_0"), val = tensor([1])]; + bool variance_123_keep_dims_0 = const()[name = string("variance_123_keep_dims_0"), val = bool(true)]; + tensor variance_123_cast_fp16 = reduce_mean(axes = variance_123_axes_0, keep_dims = variance_123_keep_dims_0, x = inputs_sq_3_cast_fp16)[name = string("variance_123_cast_fp16")]; + fp16 var_3653_to_fp16 = const()[name = string("op_3653_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3654_cast_fp16 = add(x = variance_123_cast_fp16, y = var_3653_to_fp16)[name = string("op_3654_cast_fp16")]; + fp32 var_3655_epsilon_0 = const()[name = string("op_3655_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3655_cast_fp16 = rsqrt(epsilon = var_3655_epsilon_0, x = var_3654_cast_fp16)[name = string("op_3655_cast_fp16")]; + tensor hidden_states_3_cast_fp16 = mul(x = inputs_3_cast_fp16, y = var_3655_cast_fp16)[name = string("hidden_states_3_cast_fp16")]; + tensor input_155_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_3_cast_fp16)[name = string("input_155_cast_fp16")]; + string logits_5_pad_type_0 = const()[name = string("logits_5_pad_type_0"), val = string("valid")]; + tensor logits_5_strides_0 = const()[name = string("logits_5_strides_0"), val = tensor([1, 1])]; + tensor logits_5_pad_0 = const()[name = string("logits_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_5_dilations_0 = const()[name = string("logits_5_dilations_0"), val = tensor([1, 1])]; + int32 logits_5_groups_0 = const()[name = string("logits_5_groups_0"), val = int32(1)]; + tensor lm_heads_1_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80804480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82901696))))[name = string("lm_heads_1_weight_to_fp16_palettized")]; + tensor logits_5_cast_fp16 = conv(dilations = logits_5_dilations_0, groups = logits_5_groups_0, pad = logits_5_pad_0, pad_type = logits_5_pad_type_0, strides = logits_5_strides_0, weight = lm_heads_1_weight_to_fp16_palettized, x = input_155_cast_fp16)[name = string("logits_5_cast_fp16")]; + tensor var_3673 = const()[name = string("op_3673"), val = tensor([1, 2048])]; + tensor logits_7_cast_fp16 = reshape(shape = var_3673, x = logits_5_cast_fp16)[name = string("logits_7_cast_fp16")]; + tensor scaled_logits_3_cast_fp16 = real_div(x = logits_7_cast_fp16, y = temperature)[name = string("scaled_logits_3_cast_fp16")]; + int32 var_3683 = const()[name = string("op_3683"), val = int32(100)]; + int32 top_values_3_axis_0 = const()[name = string("top_values_3_axis_0"), val = int32(1)]; + bool top_values_3_ascending_0 = const()[name = string("top_values_3_ascending_0"), val = bool(false)]; + bool top_values_3_sort_0 = const()[name = string("top_values_3_sort_0"), val = bool(true)]; + bool top_values_3_return_indices_0 = const()[name = string("top_values_3_return_indices_0"), val = bool(true)]; + string top_values_3_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_3_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_3_cast_fp16_cast_uint16_0, tensor top_values_3_cast_fp16_cast_uint16_1 = topk(ascending = top_values_3_ascending_0, axis = top_values_3_axis_0, k = var_3683, output_indices_dtype = top_values_3_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_3_return_indices_0, sort = top_values_3_sort_0, x = scaled_logits_3_cast_fp16)[name = string("top_values_3_cast_fp16_cast_uint16")]; + tensor var_3689_cast_fp16 = mul(x = top_values_3_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_3689_cast_fp16")]; + tensor var_3693_cast_fp16 = add(x = var_3689_cast_fp16, y = var_2438_cast_fp16)[name = string("op_3693_cast_fp16")]; + tensor reduce_min_1_axes_0 = const()[name = string("reduce_min_1_axes_0"), val = tensor([1])]; + bool reduce_min_1_keep_dims_0 = const()[name = string("reduce_min_1_keep_dims_0"), val = bool(true)]; + tensor reduce_min_1_cast_fp16 = reduce_min(axes = reduce_min_1_axes_0, keep_dims = reduce_min_1_keep_dims_0, x = var_3693_cast_fp16)[name = string("reduce_min_1_cast_fp16")]; + tensor var_3696_cast_fp16 = greater_equal(x = scaled_logits_3_cast_fp16, y = reduce_min_1_cast_fp16)[name = string("op_3696_cast_fp16")]; + fp16 var_3697_value_0_to_fp16 = const()[name = string("op_3697_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_3697_cast_fp16 = fill_like(ref_tensor = scaled_logits_3_cast_fp16, value = var_3697_value_0_to_fp16)[name = string("op_3697_cast_fp16")]; + tensor masked_logits_3_cast_fp16 = select(a = scaled_logits_3_cast_fp16, b = var_3697_cast_fp16, cond = var_3696_cast_fp16)[name = string("masked_logits_3_cast_fp16")]; + tensor var_3701_begin_0 = const()[name = string("op_3701_begin_0"), val = tensor([1, 0])]; + tensor var_3701_end_0 = const()[name = string("op_3701_end_0"), val = tensor([2, 2048])]; + tensor var_3701_end_mask_0 = const()[name = string("op_3701_end_mask_0"), val = tensor([false, true])]; + tensor var_3701_squeeze_mask_0 = const()[name = string("op_3701_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_3701_cast_fp16 = slice_by_index(begin = var_3701_begin_0, end = var_3701_end_0, end_mask = var_3701_end_mask_0, squeeze_mask = var_3701_squeeze_mask_0, x = gumbel)[name = string("op_3701_cast_fp16")]; + tensor var_3704 = const()[name = string("op_3704"), val = tensor([1, 2048])]; + tensor var_3705_cast_fp16 = reshape(shape = var_3704, x = var_3701_cast_fp16)[name = string("op_3705_cast_fp16")]; + tensor noisy_logits_3_cast_fp16 = add(x = masked_logits_3_cast_fp16, y = var_3705_cast_fp16)[name = string("noisy_logits_3_cast_fp16")]; + int32 code_3_axis_0 = const()[name = string("code_3_axis_0"), val = int32(1)]; + bool code_3_keep_dims_0 = const()[name = string("code_3_keep_dims_0"), val = bool(false)]; + string code_3_output_dtype_0 = const()[name = string("code_3_output_dtype_0"), val = string("int32")]; + tensor code_3_cast_fp16 = reduce_argmax(axis = code_3_axis_0, keep_dims = code_3_keep_dims_0, output_dtype = code_3_output_dtype_0, x = noisy_logits_3_cast_fp16)[name = string("code_3_cast_fp16")]; + int32 var_3716 = const()[name = string("op_3716"), val = int32(2048)]; + tensor input_157 = add(x = code_3_cast_fp16, y = var_3716)[name = string("input_157")]; + int32 code_embed_5_axis_0 = const()[name = string("code_embed_5_axis_0"), val = int32(0)]; + int32 code_embed_5_batch_dims_0 = const()[name = string("code_embed_5_batch_dims_0"), val = int32(0)]; + bool code_embed_5_validate_indices_0 = const()[name = string("code_embed_5_validate_indices_0"), val = bool(false)]; + string input_157_to_uint16_dtype_0 = const()[name = string("input_157_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_157_to_uint16 = cast(dtype = input_157_to_uint16_dtype_0, x = input_157)[name = string("cast_13")]; + tensor code_embed_5_cast_fp16_cast_uint16 = gather(axis = code_embed_5_axis_0, batch_dims = code_embed_5_batch_dims_0, indices = input_157_to_uint16, validate_indices = code_embed_5_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_5_cast_fp16_cast_uint16")]; + tensor var_3720 = const()[name = string("op_3720"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_7_cast_fp16 = reshape(shape = var_3720, x = code_embed_5_cast_fp16_cast_uint16)[name = string("code_embed_7_cast_fp16")]; + tensor embed_sum_5_cast_fp16 = add(x = code_embed_3_cast_fp16, y = code_embed_7_cast_fp16)[name = string("embed_sum_5_cast_fp16")]; + tensor key_cache_31_begin_0 = const()[name = string("key_cache_31_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_31_end_0 = const()[name = string("key_cache_31_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_31_end_mask_0 = const()[name = string("key_cache_31_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_31_cast_fp16 = slice_by_index(begin = key_cache_31_begin_0, end = key_cache_31_end_0, end_mask = key_cache_31_end_mask_0, x = layer_key_caches_7_cast_fp16)[name = string("key_cache_31_cast_fp16")]; + tensor value_cache_31_begin_0 = const()[name = string("value_cache_31_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_31_end_0 = const()[name = string("value_cache_31_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_31_end_mask_0 = const()[name = string("value_cache_31_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_31_cast_fp16 = slice_by_index(begin = value_cache_31_begin_0, end = value_cache_31_end_0, end_mask = value_cache_31_end_mask_0, x = layer_value_caches_7_cast_fp16)[name = string("value_cache_31_cast_fp16")]; + int32 var_3819 = const()[name = string("op_3819"), val = int32(2)]; + int32 var_3823 = const()[name = string("op_3823"), val = int32(3)]; + tensor var_3838_cast_fp16 = mul(x = code_embed_7_cast_fp16, y = code_embed_7_cast_fp16)[name = string("op_3838_cast_fp16")]; + tensor variance_125_axes_0 = const()[name = string("variance_125_axes_0"), val = tensor([1])]; + bool variance_125_keep_dims_0 = const()[name = string("variance_125_keep_dims_0"), val = bool(true)]; + tensor variance_125_cast_fp16 = reduce_mean(axes = variance_125_axes_0, keep_dims = variance_125_keep_dims_0, x = var_3838_cast_fp16)[name = string("variance_125_cast_fp16")]; + fp16 var_3841_to_fp16 = const()[name = string("op_3841_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3842_cast_fp16 = add(x = variance_125_cast_fp16, y = var_3841_to_fp16)[name = string("op_3842_cast_fp16")]; + fp32 var_3843_epsilon_0 = const()[name = string("op_3843_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3843_cast_fp16 = rsqrt(epsilon = var_3843_epsilon_0, x = var_3842_cast_fp16)[name = string("op_3843_cast_fp16")]; + tensor var_3844_cast_fp16 = mul(x = code_embed_7_cast_fp16, y = var_3843_cast_fp16)[name = string("op_3844_cast_fp16")]; + tensor input_159_cast_fp16 = mul(x = var_3844_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_159_cast_fp16")]; + string q_91_pad_type_0 = const()[name = string("q_91_pad_type_0"), val = string("valid")]; + tensor q_91_strides_0 = const()[name = string("q_91_strides_0"), val = tensor([1, 1])]; + tensor q_91_pad_0 = const()[name = string("q_91_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_91_dilations_0 = const()[name = string("q_91_dilations_0"), val = tensor([1, 1])]; + int32 q_91_groups_0 = const()[name = string("q_91_groups_0"), val = int32(1)]; + tensor q_91_cast_fp16 = conv(dilations = q_91_dilations_0, groups = q_91_groups_0, pad = q_91_pad_0, pad_type = q_91_pad_type_0, strides = q_91_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_159_cast_fp16)[name = string("q_91_cast_fp16")]; + string k_91_pad_type_0 = const()[name = string("k_91_pad_type_0"), val = string("valid")]; + tensor k_91_strides_0 = const()[name = string("k_91_strides_0"), val = tensor([1, 1])]; + tensor k_91_pad_0 = const()[name = string("k_91_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_91_dilations_0 = const()[name = string("k_91_dilations_0"), val = tensor([1, 1])]; + int32 k_91_groups_0 = const()[name = string("k_91_groups_0"), val = int32(1)]; + tensor k_91_cast_fp16 = conv(dilations = k_91_dilations_0, groups = k_91_groups_0, pad = k_91_pad_0, pad_type = k_91_pad_type_0, strides = k_91_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_159_cast_fp16)[name = string("k_91_cast_fp16")]; + string v_31_pad_type_0 = const()[name = string("v_31_pad_type_0"), val = string("valid")]; + tensor v_31_strides_0 = const()[name = string("v_31_strides_0"), val = tensor([1, 1])]; + tensor v_31_pad_0 = const()[name = string("v_31_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_31_dilations_0 = const()[name = string("v_31_dilations_0"), val = tensor([1, 1])]; + int32 v_31_groups_0 = const()[name = string("v_31_groups_0"), val = int32(1)]; + tensor v_31_cast_fp16 = conv(dilations = v_31_dilations_0, groups = v_31_groups_0, pad = v_31_pad_0, pad_type = v_31_pad_type_0, strides = v_31_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_159_cast_fp16)[name = string("v_31_cast_fp16")]; + tensor var_3878 = const()[name = string("op_3878"), val = tensor([16, 128, 1, 1])]; + tensor x_115_cast_fp16 = reshape(shape = var_3878, x = q_91_cast_fp16)[name = string("x_115_cast_fp16")]; + tensor var_3881_cast_fp16 = mul(x = x_115_cast_fp16, y = x_115_cast_fp16)[name = string("op_3881_cast_fp16")]; + tensor variance_127_axes_0 = const()[name = string("variance_127_axes_0"), val = tensor([1])]; + bool variance_127_keep_dims_0 = const()[name = string("variance_127_keep_dims_0"), val = bool(true)]; + tensor variance_127_cast_fp16 = reduce_mean(axes = variance_127_axes_0, keep_dims = variance_127_keep_dims_0, x = var_3881_cast_fp16)[name = string("variance_127_cast_fp16")]; + fp16 var_3884_to_fp16 = const()[name = string("op_3884_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3885_cast_fp16 = add(x = variance_127_cast_fp16, y = var_3884_to_fp16)[name = string("op_3885_cast_fp16")]; + fp32 var_3886_epsilon_0 = const()[name = string("op_3886_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3886_cast_fp16 = rsqrt(epsilon = var_3886_epsilon_0, x = var_3885_cast_fp16)[name = string("op_3886_cast_fp16")]; + tensor var_3887_cast_fp16 = mul(x = x_115_cast_fp16, y = var_3886_cast_fp16)[name = string("op_3887_cast_fp16")]; + tensor q_93_cast_fp16 = mul(x = var_3887_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_93_cast_fp16")]; + tensor var_3889 = const()[name = string("op_3889"), val = tensor([8, 128, 1, 1])]; + tensor x_117_cast_fp16 = reshape(shape = var_3889, x = k_91_cast_fp16)[name = string("x_117_cast_fp16")]; + tensor var_3892_cast_fp16 = mul(x = x_117_cast_fp16, y = x_117_cast_fp16)[name = string("op_3892_cast_fp16")]; + tensor variance_129_axes_0 = const()[name = string("variance_129_axes_0"), val = tensor([1])]; + bool variance_129_keep_dims_0 = const()[name = string("variance_129_keep_dims_0"), val = bool(true)]; + tensor variance_129_cast_fp16 = reduce_mean(axes = variance_129_axes_0, keep_dims = variance_129_keep_dims_0, x = var_3892_cast_fp16)[name = string("variance_129_cast_fp16")]; + fp16 var_3895_to_fp16 = const()[name = string("op_3895_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3896_cast_fp16 = add(x = variance_129_cast_fp16, y = var_3895_to_fp16)[name = string("op_3896_cast_fp16")]; + fp32 var_3897_epsilon_0 = const()[name = string("op_3897_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3897_cast_fp16 = rsqrt(epsilon = var_3897_epsilon_0, x = var_3896_cast_fp16)[name = string("op_3897_cast_fp16")]; + tensor var_3898_cast_fp16 = mul(x = x_117_cast_fp16, y = var_3897_cast_fp16)[name = string("op_3898_cast_fp16")]; + tensor k_93_cast_fp16 = mul(x = var_3898_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_93_cast_fp16")]; + tensor var_3900 = const()[name = string("op_3900"), val = tensor([1, 16, 128, 1])]; + tensor z_61_cast_fp16 = reshape(shape = var_3900, x = q_93_cast_fp16)[name = string("z_61_cast_fp16")]; + tensor var_3902 = const()[name = string("op_3902"), val = tensor([1, 8, 128, 1])]; + tensor z_63_cast_fp16 = reshape(shape = var_3902, x = k_93_cast_fp16)[name = string("z_63_cast_fp16")]; + tensor z1_61_begin_0 = const()[name = string("z1_61_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_61_end_0 = const()[name = string("z1_61_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_61_end_mask_0 = const()[name = string("z1_61_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_61_cast_fp16 = slice_by_index(begin = z1_61_begin_0, end = z1_61_end_0, end_mask = z1_61_end_mask_0, x = z_61_cast_fp16)[name = string("z1_61_cast_fp16")]; + tensor z2_61_begin_0 = const()[name = string("z2_61_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_61_end_0 = const()[name = string("z2_61_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_61_end_mask_0 = const()[name = string("z2_61_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_61_cast_fp16 = slice_by_index(begin = z2_61_begin_0, end = z2_61_end_0, end_mask = z2_61_end_mask_0, x = z_61_cast_fp16)[name = string("z2_61_cast_fp16")]; + tensor cos_31_to_fp16 = const()[name = string("cos_31_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634432)))]; + tensor var_3910_cast_fp16 = mul(x = z_61_cast_fp16, y = cos_31_to_fp16)[name = string("op_3910_cast_fp16")]; + fp16 const_34_promoted_to_fp16 = const()[name = string("const_34_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3911_cast_fp16 = mul(x = z2_61_cast_fp16, y = const_34_promoted_to_fp16)[name = string("op_3911_cast_fp16")]; + bool var_3913_interleave_0 = const()[name = string("op_3913_interleave_0"), val = bool(false)]; + tensor var_3913_cast_fp16 = concat(axis = var_3819, interleave = var_3913_interleave_0, values = (var_3911_cast_fp16, z1_61_cast_fp16))[name = string("op_3913_cast_fp16")]; + tensor sin_31_to_fp16 = const()[name = string("sin_31_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141634752)))]; + tensor var_3914_cast_fp16 = mul(x = var_3913_cast_fp16, y = sin_31_to_fp16)[name = string("op_3914_cast_fp16")]; + tensor q_95_cast_fp16 = add(x = var_3910_cast_fp16, y = var_3914_cast_fp16)[name = string("q_95_cast_fp16")]; + tensor z1_63_begin_0 = const()[name = string("z1_63_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_63_end_0 = const()[name = string("z1_63_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_63_end_mask_0 = const()[name = string("z1_63_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_63_cast_fp16 = slice_by_index(begin = z1_63_begin_0, end = z1_63_end_0, end_mask = z1_63_end_mask_0, x = z_63_cast_fp16)[name = string("z1_63_cast_fp16")]; + tensor z2_63_begin_0 = const()[name = string("z2_63_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_63_end_0 = const()[name = string("z2_63_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_63_end_mask_0 = const()[name = string("z2_63_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_63_cast_fp16 = slice_by_index(begin = z2_63_begin_0, end = z2_63_end_0, end_mask = z2_63_end_mask_0, x = z_63_cast_fp16)[name = string("z2_63_cast_fp16")]; + tensor var_3922_cast_fp16 = mul(x = z_63_cast_fp16, y = cos_31_to_fp16)[name = string("op_3922_cast_fp16")]; + fp16 const_35_promoted_to_fp16 = const()[name = string("const_35_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3923_cast_fp16 = mul(x = z2_63_cast_fp16, y = const_35_promoted_to_fp16)[name = string("op_3923_cast_fp16")]; + bool var_3925_interleave_0 = const()[name = string("op_3925_interleave_0"), val = bool(false)]; + tensor var_3925_cast_fp16 = concat(axis = var_3819, interleave = var_3925_interleave_0, values = (var_3923_cast_fp16, z1_63_cast_fp16))[name = string("op_3925_cast_fp16")]; + tensor var_3926_cast_fp16 = mul(x = var_3925_cast_fp16, y = sin_31_to_fp16)[name = string("op_3926_cast_fp16")]; + tensor k_95_cast_fp16 = add(x = var_3922_cast_fp16, y = var_3926_cast_fp16)[name = string("k_95_cast_fp16")]; + tensor var_3928 = const()[name = string("op_3928"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_31_cast_fp16 = reshape(shape = var_3928, x = k_95_cast_fp16)[name = string("cur_key_31_cast_fp16")]; + tensor var_3930_to_fp16 = const()[name = string("op_3930_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635072)))]; + tensor var_3931_cast_fp16 = mul(x = key_cache_31_cast_fp16, y = var_3930_to_fp16)[name = string("op_3931_cast_fp16")]; + tensor upd_31_to_fp16 = const()[name = string("upd_31_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635200)))]; + tensor var_3932_cast_fp16 = mul(x = cur_key_31_cast_fp16, y = upd_31_to_fp16)[name = string("op_3932_cast_fp16")]; + tensor key_31_cast_fp16 = add(x = var_3931_cast_fp16, y = var_3932_cast_fp16)[name = string("key_31_cast_fp16")]; + tensor var_3934_to_fp16 = const()[name = string("op_3934_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635072)))]; + tensor var_3935_cast_fp16 = mul(x = value_cache_31_cast_fp16, y = var_3934_to_fp16)[name = string("op_3935_cast_fp16")]; + tensor var_3936_cast_fp16 = mul(x = v_31_cast_fp16, y = upd_31_to_fp16)[name = string("op_3936_cast_fp16")]; + tensor value_31_cast_fp16 = add(x = var_3935_cast_fp16, y = var_3936_cast_fp16)[name = string("value_31_cast_fp16")]; + tensor var_3938 = const()[name = string("op_3938"), val = tensor([1, 8, 128, 16])]; + tensor kh_61_cast_fp16 = reshape(shape = var_3938, x = key_31_cast_fp16)[name = string("kh_61_cast_fp16")]; + tensor var_3940 = const()[name = string("op_3940"), val = tensor([1, 8, 128, 16])]; + tensor vh_61_cast_fp16 = reshape(shape = var_3940, x = value_31_cast_fp16)[name = string("vh_61_cast_fp16")]; + tensor transpose_60_perm_0 = const()[name = string("transpose_60_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_30_reps_0 = const()[name = string("tile_30_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_60_cast_fp16 = transpose(perm = transpose_60_perm_0, x = kh_61_cast_fp16)[name = string("transpose_389")]; + tensor tile_30_cast_fp16 = tile(reps = tile_30_reps_0, x = transpose_60_cast_fp16)[name = string("tile_30_cast_fp16")]; + tensor concat_78 = const()[name = string("concat_78"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_60_cast_fp16 = reshape(shape = concat_78, x = tile_30_cast_fp16)[name = string("reshape_60_cast_fp16")]; + tensor transpose_61_perm_0 = const()[name = string("transpose_61_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_79 = const()[name = string("concat_79"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_61_cast_fp16 = transpose(perm = transpose_61_perm_0, x = reshape_60_cast_fp16)[name = string("transpose_388")]; + tensor reshape_61_cast_fp16 = reshape(shape = concat_79, x = transpose_61_cast_fp16)[name = string("reshape_61_cast_fp16")]; + tensor transpose_62_perm_0 = const()[name = string("transpose_62_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_31_reps_0 = const()[name = string("tile_31_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_62_cast_fp16 = transpose(perm = transpose_62_perm_0, x = vh_61_cast_fp16)[name = string("transpose_387")]; + tensor tile_31_cast_fp16 = tile(reps = tile_31_reps_0, x = transpose_62_cast_fp16)[name = string("tile_31_cast_fp16")]; + tensor concat_80 = const()[name = string("concat_80"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_62_cast_fp16 = reshape(shape = concat_80, x = tile_31_cast_fp16)[name = string("reshape_62_cast_fp16")]; + tensor transpose_63_perm_0 = const()[name = string("transpose_63_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_81 = const()[name = string("concat_81"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_63_cast_fp16 = transpose(perm = transpose_63_perm_0, x = reshape_62_cast_fp16)[name = string("transpose_386")]; + tensor reshape_63_cast_fp16 = reshape(shape = concat_81, x = transpose_63_cast_fp16)[name = string("reshape_63_cast_fp16")]; + fp16 var_3944_to_fp16 = const()[name = string("op_3944_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_3945_cast_fp16 = mul(x = q_95_cast_fp16, y = var_3944_to_fp16)[name = string("op_3945_cast_fp16")]; + tensor transpose_377_perm_0 = const()[name = string("transpose_377_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_65_transpose_x_1 = const()[name = string("w_65_transpose_x_1"), val = bool(true)]; + bool w_65_transpose_y_1 = const()[name = string("w_65_transpose_y_1"), val = bool(false)]; + tensor transpose_377_cast_fp16 = transpose(perm = transpose_377_perm_0, x = reshape_61_cast_fp16)[name = string("transpose_385")]; + tensor w_65_cast_fp16 = matmul(transpose_x = w_65_transpose_x_1, transpose_y = w_65_transpose_y_1, x = var_3945_cast_fp16, y = transpose_377_cast_fp16)[name = string("w_65_cast_fp16")]; + tensor pad_31_to_fp16 = const()[name = string("pad_31_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635328)))]; + tensor var_3948_cast_fp16 = add(x = w_65_cast_fp16, y = pad_31_to_fp16)[name = string("op_3948_cast_fp16")]; + tensor w_67_cast_fp16 = softmax(axis = var_3823, x = var_3948_cast_fp16)[name = string("w_67_cast_fp16")]; + tensor transpose_378_perm_0 = const()[name = string("transpose_378_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_31_transpose_x_1 = const()[name = string("attn_31_transpose_x_1"), val = bool(false)]; + bool attn_31_transpose_y_1 = const()[name = string("attn_31_transpose_y_1"), val = bool(true)]; + tensor transpose_378_cast_fp16 = transpose(perm = transpose_378_perm_0, x = reshape_63_cast_fp16)[name = string("transpose_384")]; + tensor attn_31_cast_fp16 = matmul(transpose_x = attn_31_transpose_x_1, transpose_y = attn_31_transpose_y_1, x = transpose_378_cast_fp16, y = w_67_cast_fp16)[name = string("attn_31_cast_fp16")]; + tensor var_3952 = const()[name = string("op_3952"), val = tensor([1, 2048, 1, 1])]; + tensor input_161_cast_fp16 = reshape(shape = var_3952, x = attn_31_cast_fp16)[name = string("input_161_cast_fp16")]; + string attn_output_31_pad_type_0 = const()[name = string("attn_output_31_pad_type_0"), val = string("valid")]; + tensor attn_output_31_strides_0 = const()[name = string("attn_output_31_strides_0"), val = tensor([1, 1])]; + tensor attn_output_31_pad_0 = const()[name = string("attn_output_31_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_31_dilations_0 = const()[name = string("attn_output_31_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_31_groups_0 = const()[name = string("attn_output_31_groups_0"), val = int32(1)]; + tensor attn_output_31_cast_fp16 = conv(dilations = attn_output_31_dilations_0, groups = attn_output_31_groups_0, pad = attn_output_31_pad_0, pad_type = attn_output_31_pad_type_0, strides = attn_output_31_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_161_cast_fp16)[name = string("attn_output_31_cast_fp16")]; + tensor x_119_cast_fp16 = add(x = code_embed_7_cast_fp16, y = attn_output_31_cast_fp16)[name = string("x_119_cast_fp16")]; + tensor var_3966_cast_fp16 = mul(x = x_119_cast_fp16, y = x_119_cast_fp16)[name = string("op_3966_cast_fp16")]; + tensor variance_131_axes_0 = const()[name = string("variance_131_axes_0"), val = tensor([1])]; + bool variance_131_keep_dims_0 = const()[name = string("variance_131_keep_dims_0"), val = bool(true)]; + tensor variance_131_cast_fp16 = reduce_mean(axes = variance_131_axes_0, keep_dims = variance_131_keep_dims_0, x = var_3966_cast_fp16)[name = string("variance_131_cast_fp16")]; + fp16 var_3969_to_fp16 = const()[name = string("op_3969_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3970_cast_fp16 = add(x = variance_131_cast_fp16, y = var_3969_to_fp16)[name = string("op_3970_cast_fp16")]; + fp32 var_3971_epsilon_0 = const()[name = string("op_3971_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3971_cast_fp16 = rsqrt(epsilon = var_3971_epsilon_0, x = var_3970_cast_fp16)[name = string("op_3971_cast_fp16")]; + tensor var_3972_cast_fp16 = mul(x = x_119_cast_fp16, y = var_3971_cast_fp16)[name = string("op_3972_cast_fp16")]; + tensor input_163_cast_fp16 = mul(x = var_3972_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_163_cast_fp16")]; + string input_165_pad_type_0 = const()[name = string("input_165_pad_type_0"), val = string("valid")]; + tensor input_165_strides_0 = const()[name = string("input_165_strides_0"), val = tensor([1, 1])]; + tensor input_165_pad_0 = const()[name = string("input_165_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_165_dilations_0 = const()[name = string("input_165_dilations_0"), val = tensor([1, 1])]; + int32 input_165_groups_0 = const()[name = string("input_165_groups_0"), val = int32(1)]; + tensor input_165_cast_fp16 = conv(dilations = input_165_dilations_0, groups = input_165_groups_0, pad = input_165_pad_0, pad_type = input_165_pad_type_0, strides = input_165_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_163_cast_fp16)[name = string("input_165_cast_fp16")]; + tensor var_3980_cast_fp16 = silu(x = input_165_cast_fp16)[name = string("op_3980_cast_fp16")]; + string var_3986_pad_type_0 = const()[name = string("op_3986_pad_type_0"), val = string("valid")]; + tensor var_3986_strides_0 = const()[name = string("op_3986_strides_0"), val = tensor([1, 1])]; + tensor var_3986_pad_0 = const()[name = string("op_3986_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3986_dilations_0 = const()[name = string("op_3986_dilations_0"), val = tensor([1, 1])]; + int32 var_3986_groups_0 = const()[name = string("op_3986_groups_0"), val = int32(1)]; + tensor var_3986_cast_fp16 = conv(dilations = var_3986_dilations_0, groups = var_3986_groups_0, pad = var_3986_pad_0, pad_type = var_3986_pad_type_0, strides = var_3986_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_163_cast_fp16)[name = string("op_3986_cast_fp16")]; + tensor input_167_cast_fp16 = mul(x = var_3980_cast_fp16, y = var_3986_cast_fp16)[name = string("input_167_cast_fp16")]; + string h_31_pad_type_0 = const()[name = string("h_31_pad_type_0"), val = string("valid")]; + tensor h_31_strides_0 = const()[name = string("h_31_strides_0"), val = tensor([1, 1])]; + tensor h_31_pad_0 = const()[name = string("h_31_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_31_dilations_0 = const()[name = string("h_31_dilations_0"), val = tensor([1, 1])]; + int32 h_31_groups_0 = const()[name = string("h_31_groups_0"), val = int32(1)]; + tensor h_31_cast_fp16 = conv(dilations = h_31_dilations_0, groups = h_31_groups_0, pad = h_31_pad_0, pad_type = h_31_pad_type_0, strides = h_31_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_167_cast_fp16)[name = string("h_31_cast_fp16")]; + tensor x_121_cast_fp16 = add(x = x_119_cast_fp16, y = h_31_cast_fp16)[name = string("x_121_cast_fp16")]; + tensor key_cache_33_begin_0 = const()[name = string("key_cache_33_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_33_end_0 = const()[name = string("key_cache_33_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_33_end_mask_0 = const()[name = string("key_cache_33_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_33_cast_fp16 = slice_by_index(begin = key_cache_33_begin_0, end = key_cache_33_end_0, end_mask = key_cache_33_end_mask_0, x = layer_key_caches_7_cast_fp16)[name = string("key_cache_33_cast_fp16")]; + tensor value_cache_33_begin_0 = const()[name = string("value_cache_33_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_33_end_0 = const()[name = string("value_cache_33_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_33_end_mask_0 = const()[name = string("value_cache_33_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_33_cast_fp16 = slice_by_index(begin = value_cache_33_begin_0, end = value_cache_33_end_0, end_mask = value_cache_33_end_mask_0, x = layer_value_caches_7_cast_fp16)[name = string("value_cache_33_cast_fp16")]; + int32 var_4039 = const()[name = string("op_4039"), val = int32(2)]; + int32 var_4043 = const()[name = string("op_4043"), val = int32(3)]; + tensor var_4058_cast_fp16 = mul(x = x_121_cast_fp16, y = x_121_cast_fp16)[name = string("op_4058_cast_fp16")]; + tensor variance_133_axes_0 = const()[name = string("variance_133_axes_0"), val = tensor([1])]; + bool variance_133_keep_dims_0 = const()[name = string("variance_133_keep_dims_0"), val = bool(true)]; + tensor variance_133_cast_fp16 = reduce_mean(axes = variance_133_axes_0, keep_dims = variance_133_keep_dims_0, x = var_4058_cast_fp16)[name = string("variance_133_cast_fp16")]; + fp16 var_4061_to_fp16 = const()[name = string("op_4061_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4062_cast_fp16 = add(x = variance_133_cast_fp16, y = var_4061_to_fp16)[name = string("op_4062_cast_fp16")]; + fp32 var_4063_epsilon_0 = const()[name = string("op_4063_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4063_cast_fp16 = rsqrt(epsilon = var_4063_epsilon_0, x = var_4062_cast_fp16)[name = string("op_4063_cast_fp16")]; + tensor var_4064_cast_fp16 = mul(x = x_121_cast_fp16, y = var_4063_cast_fp16)[name = string("op_4064_cast_fp16")]; + tensor input_169_cast_fp16 = mul(x = var_4064_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_169_cast_fp16")]; + string q_97_pad_type_0 = const()[name = string("q_97_pad_type_0"), val = string("valid")]; + tensor q_97_strides_0 = const()[name = string("q_97_strides_0"), val = tensor([1, 1])]; + tensor q_97_pad_0 = const()[name = string("q_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_97_dilations_0 = const()[name = string("q_97_dilations_0"), val = tensor([1, 1])]; + int32 q_97_groups_0 = const()[name = string("q_97_groups_0"), val = int32(1)]; + tensor q_97_cast_fp16 = conv(dilations = q_97_dilations_0, groups = q_97_groups_0, pad = q_97_pad_0, pad_type = q_97_pad_type_0, strides = q_97_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_169_cast_fp16)[name = string("q_97_cast_fp16")]; + string k_97_pad_type_0 = const()[name = string("k_97_pad_type_0"), val = string("valid")]; + tensor k_97_strides_0 = const()[name = string("k_97_strides_0"), val = tensor([1, 1])]; + tensor k_97_pad_0 = const()[name = string("k_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_97_dilations_0 = const()[name = string("k_97_dilations_0"), val = tensor([1, 1])]; + int32 k_97_groups_0 = const()[name = string("k_97_groups_0"), val = int32(1)]; + tensor k_97_cast_fp16 = conv(dilations = k_97_dilations_0, groups = k_97_groups_0, pad = k_97_pad_0, pad_type = k_97_pad_type_0, strides = k_97_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_169_cast_fp16)[name = string("k_97_cast_fp16")]; + string v_33_pad_type_0 = const()[name = string("v_33_pad_type_0"), val = string("valid")]; + tensor v_33_strides_0 = const()[name = string("v_33_strides_0"), val = tensor([1, 1])]; + tensor v_33_pad_0 = const()[name = string("v_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_33_dilations_0 = const()[name = string("v_33_dilations_0"), val = tensor([1, 1])]; + int32 v_33_groups_0 = const()[name = string("v_33_groups_0"), val = int32(1)]; + tensor v_33_cast_fp16 = conv(dilations = v_33_dilations_0, groups = v_33_groups_0, pad = v_33_pad_0, pad_type = v_33_pad_type_0, strides = v_33_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_169_cast_fp16)[name = string("v_33_cast_fp16")]; + tensor var_4098 = const()[name = string("op_4098"), val = tensor([16, 128, 1, 1])]; + tensor x_123_cast_fp16 = reshape(shape = var_4098, x = q_97_cast_fp16)[name = string("x_123_cast_fp16")]; + tensor var_4101_cast_fp16 = mul(x = x_123_cast_fp16, y = x_123_cast_fp16)[name = string("op_4101_cast_fp16")]; + tensor variance_135_axes_0 = const()[name = string("variance_135_axes_0"), val = tensor([1])]; + bool variance_135_keep_dims_0 = const()[name = string("variance_135_keep_dims_0"), val = bool(true)]; + tensor variance_135_cast_fp16 = reduce_mean(axes = variance_135_axes_0, keep_dims = variance_135_keep_dims_0, x = var_4101_cast_fp16)[name = string("variance_135_cast_fp16")]; + fp16 var_4104_to_fp16 = const()[name = string("op_4104_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4105_cast_fp16 = add(x = variance_135_cast_fp16, y = var_4104_to_fp16)[name = string("op_4105_cast_fp16")]; + fp32 var_4106_epsilon_0 = const()[name = string("op_4106_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4106_cast_fp16 = rsqrt(epsilon = var_4106_epsilon_0, x = var_4105_cast_fp16)[name = string("op_4106_cast_fp16")]; + tensor var_4107_cast_fp16 = mul(x = x_123_cast_fp16, y = var_4106_cast_fp16)[name = string("op_4107_cast_fp16")]; + tensor q_99_cast_fp16 = mul(x = var_4107_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_99_cast_fp16")]; + tensor var_4109 = const()[name = string("op_4109"), val = tensor([8, 128, 1, 1])]; + tensor x_125_cast_fp16 = reshape(shape = var_4109, x = k_97_cast_fp16)[name = string("x_125_cast_fp16")]; + tensor var_4112_cast_fp16 = mul(x = x_125_cast_fp16, y = x_125_cast_fp16)[name = string("op_4112_cast_fp16")]; + tensor variance_137_axes_0 = const()[name = string("variance_137_axes_0"), val = tensor([1])]; + bool variance_137_keep_dims_0 = const()[name = string("variance_137_keep_dims_0"), val = bool(true)]; + tensor variance_137_cast_fp16 = reduce_mean(axes = variance_137_axes_0, keep_dims = variance_137_keep_dims_0, x = var_4112_cast_fp16)[name = string("variance_137_cast_fp16")]; + fp16 var_4115_to_fp16 = const()[name = string("op_4115_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4116_cast_fp16 = add(x = variance_137_cast_fp16, y = var_4115_to_fp16)[name = string("op_4116_cast_fp16")]; + fp32 var_4117_epsilon_0 = const()[name = string("op_4117_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4117_cast_fp16 = rsqrt(epsilon = var_4117_epsilon_0, x = var_4116_cast_fp16)[name = string("op_4117_cast_fp16")]; + tensor var_4118_cast_fp16 = mul(x = x_125_cast_fp16, y = var_4117_cast_fp16)[name = string("op_4118_cast_fp16")]; + tensor k_99_cast_fp16 = mul(x = var_4118_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_99_cast_fp16")]; + tensor var_4120 = const()[name = string("op_4120"), val = tensor([1, 16, 128, 1])]; + tensor z_65_cast_fp16 = reshape(shape = var_4120, x = q_99_cast_fp16)[name = string("z_65_cast_fp16")]; + tensor var_4122 = const()[name = string("op_4122"), val = tensor([1, 8, 128, 1])]; + tensor z_67_cast_fp16 = reshape(shape = var_4122, x = k_99_cast_fp16)[name = string("z_67_cast_fp16")]; + tensor z1_65_begin_0 = const()[name = string("z1_65_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_65_end_0 = const()[name = string("z1_65_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_65_end_mask_0 = const()[name = string("z1_65_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_65_cast_fp16 = slice_by_index(begin = z1_65_begin_0, end = z1_65_end_0, end_mask = z1_65_end_mask_0, x = z_65_cast_fp16)[name = string("z1_65_cast_fp16")]; + tensor z2_65_begin_0 = const()[name = string("z2_65_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_65_end_0 = const()[name = string("z2_65_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_65_end_mask_0 = const()[name = string("z2_65_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_65_cast_fp16 = slice_by_index(begin = z2_65_begin_0, end = z2_65_end_0, end_mask = z2_65_end_mask_0, x = z_65_cast_fp16)[name = string("z2_65_cast_fp16")]; + tensor var_4130_cast_fp16 = mul(x = z_65_cast_fp16, y = cos_31_to_fp16)[name = string("op_4130_cast_fp16")]; + fp16 const_36_promoted_to_fp16 = const()[name = string("const_36_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4131_cast_fp16 = mul(x = z2_65_cast_fp16, y = const_36_promoted_to_fp16)[name = string("op_4131_cast_fp16")]; + bool var_4133_interleave_0 = const()[name = string("op_4133_interleave_0"), val = bool(false)]; + tensor var_4133_cast_fp16 = concat(axis = var_4039, interleave = var_4133_interleave_0, values = (var_4131_cast_fp16, z1_65_cast_fp16))[name = string("op_4133_cast_fp16")]; + tensor var_4134_cast_fp16 = mul(x = var_4133_cast_fp16, y = sin_31_to_fp16)[name = string("op_4134_cast_fp16")]; + tensor q_101_cast_fp16 = add(x = var_4130_cast_fp16, y = var_4134_cast_fp16)[name = string("q_101_cast_fp16")]; + tensor z1_67_begin_0 = const()[name = string("z1_67_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_67_end_0 = const()[name = string("z1_67_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_67_end_mask_0 = const()[name = string("z1_67_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_67_cast_fp16 = slice_by_index(begin = z1_67_begin_0, end = z1_67_end_0, end_mask = z1_67_end_mask_0, x = z_67_cast_fp16)[name = string("z1_67_cast_fp16")]; + tensor z2_67_begin_0 = const()[name = string("z2_67_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_67_end_0 = const()[name = string("z2_67_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_67_end_mask_0 = const()[name = string("z2_67_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_67_cast_fp16 = slice_by_index(begin = z2_67_begin_0, end = z2_67_end_0, end_mask = z2_67_end_mask_0, x = z_67_cast_fp16)[name = string("z2_67_cast_fp16")]; + tensor var_4142_cast_fp16 = mul(x = z_67_cast_fp16, y = cos_31_to_fp16)[name = string("op_4142_cast_fp16")]; + fp16 const_37_promoted_to_fp16 = const()[name = string("const_37_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4143_cast_fp16 = mul(x = z2_67_cast_fp16, y = const_37_promoted_to_fp16)[name = string("op_4143_cast_fp16")]; + bool var_4145_interleave_0 = const()[name = string("op_4145_interleave_0"), val = bool(false)]; + tensor var_4145_cast_fp16 = concat(axis = var_4039, interleave = var_4145_interleave_0, values = (var_4143_cast_fp16, z1_67_cast_fp16))[name = string("op_4145_cast_fp16")]; + tensor var_4146_cast_fp16 = mul(x = var_4145_cast_fp16, y = sin_31_to_fp16)[name = string("op_4146_cast_fp16")]; + tensor k_101_cast_fp16 = add(x = var_4142_cast_fp16, y = var_4146_cast_fp16)[name = string("k_101_cast_fp16")]; + tensor var_4148 = const()[name = string("op_4148"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_33_cast_fp16 = reshape(shape = var_4148, x = k_101_cast_fp16)[name = string("cur_key_33_cast_fp16")]; + tensor var_4150_to_fp16 = const()[name = string("op_4150_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635072)))]; + tensor var_4151_cast_fp16 = mul(x = key_cache_33_cast_fp16, y = var_4150_to_fp16)[name = string("op_4151_cast_fp16")]; + tensor upd_33_to_fp16 = const()[name = string("upd_33_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635200)))]; + tensor var_4152_cast_fp16 = mul(x = cur_key_33_cast_fp16, y = upd_33_to_fp16)[name = string("op_4152_cast_fp16")]; + tensor key_33_cast_fp16 = add(x = var_4151_cast_fp16, y = var_4152_cast_fp16)[name = string("key_33_cast_fp16")]; + tensor var_4154_to_fp16 = const()[name = string("op_4154_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635072)))]; + tensor var_4155_cast_fp16 = mul(x = value_cache_33_cast_fp16, y = var_4154_to_fp16)[name = string("op_4155_cast_fp16")]; + tensor var_4156_cast_fp16 = mul(x = v_33_cast_fp16, y = upd_33_to_fp16)[name = string("op_4156_cast_fp16")]; + tensor value_33_cast_fp16 = add(x = var_4155_cast_fp16, y = var_4156_cast_fp16)[name = string("value_33_cast_fp16")]; + tensor var_4158 = const()[name = string("op_4158"), val = tensor([1, 8, 128, 16])]; + tensor kh_65_cast_fp16 = reshape(shape = var_4158, x = key_33_cast_fp16)[name = string("kh_65_cast_fp16")]; + tensor var_4160 = const()[name = string("op_4160"), val = tensor([1, 8, 128, 16])]; + tensor vh_65_cast_fp16 = reshape(shape = var_4160, x = value_33_cast_fp16)[name = string("vh_65_cast_fp16")]; + tensor transpose_64_perm_0 = const()[name = string("transpose_64_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_32_reps_0 = const()[name = string("tile_32_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_64_cast_fp16 = transpose(perm = transpose_64_perm_0, x = kh_65_cast_fp16)[name = string("transpose_383")]; + tensor tile_32_cast_fp16 = tile(reps = tile_32_reps_0, x = transpose_64_cast_fp16)[name = string("tile_32_cast_fp16")]; + tensor concat_82 = const()[name = string("concat_82"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_64_cast_fp16 = reshape(shape = concat_82, x = tile_32_cast_fp16)[name = string("reshape_64_cast_fp16")]; + tensor transpose_65_perm_0 = const()[name = string("transpose_65_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_83 = const()[name = string("concat_83"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_65_cast_fp16 = transpose(perm = transpose_65_perm_0, x = reshape_64_cast_fp16)[name = string("transpose_382")]; + tensor reshape_65_cast_fp16 = reshape(shape = concat_83, x = transpose_65_cast_fp16)[name = string("reshape_65_cast_fp16")]; + tensor transpose_66_perm_0 = const()[name = string("transpose_66_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_33_reps_0 = const()[name = string("tile_33_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_66_cast_fp16 = transpose(perm = transpose_66_perm_0, x = vh_65_cast_fp16)[name = string("transpose_381")]; + tensor tile_33_cast_fp16 = tile(reps = tile_33_reps_0, x = transpose_66_cast_fp16)[name = string("tile_33_cast_fp16")]; + tensor concat_84 = const()[name = string("concat_84"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_66_cast_fp16 = reshape(shape = concat_84, x = tile_33_cast_fp16)[name = string("reshape_66_cast_fp16")]; + tensor transpose_67_perm_0 = const()[name = string("transpose_67_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_85 = const()[name = string("concat_85"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_67_cast_fp16 = transpose(perm = transpose_67_perm_0, x = reshape_66_cast_fp16)[name = string("transpose_380")]; + tensor reshape_67_cast_fp16 = reshape(shape = concat_85, x = transpose_67_cast_fp16)[name = string("reshape_67_cast_fp16")]; + fp16 var_4164_to_fp16 = const()[name = string("op_4164_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_4165_cast_fp16 = mul(x = q_101_cast_fp16, y = var_4164_to_fp16)[name = string("op_4165_cast_fp16")]; + tensor transpose_381_perm_0 = const()[name = string("transpose_381_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_69_transpose_x_1 = const()[name = string("w_69_transpose_x_1"), val = bool(true)]; + bool w_69_transpose_y_1 = const()[name = string("w_69_transpose_y_1"), val = bool(false)]; + tensor transpose_381_cast_fp16 = transpose(perm = transpose_381_perm_0, x = reshape_65_cast_fp16)[name = string("transpose_379")]; + tensor w_69_cast_fp16 = matmul(transpose_x = w_69_transpose_x_1, transpose_y = w_69_transpose_y_1, x = var_4165_cast_fp16, y = transpose_381_cast_fp16)[name = string("w_69_cast_fp16")]; + tensor pad_33_to_fp16 = const()[name = string("pad_33_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635328)))]; + tensor var_4168_cast_fp16 = add(x = w_69_cast_fp16, y = pad_33_to_fp16)[name = string("op_4168_cast_fp16")]; + tensor w_71_cast_fp16 = softmax(axis = var_4043, x = var_4168_cast_fp16)[name = string("w_71_cast_fp16")]; + tensor transpose_382_perm_0 = const()[name = string("transpose_382_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_33_transpose_x_1 = const()[name = string("attn_33_transpose_x_1"), val = bool(false)]; + bool attn_33_transpose_y_1 = const()[name = string("attn_33_transpose_y_1"), val = bool(true)]; + tensor transpose_382_cast_fp16 = transpose(perm = transpose_382_perm_0, x = reshape_67_cast_fp16)[name = string("transpose_378")]; + tensor attn_33_cast_fp16 = matmul(transpose_x = attn_33_transpose_x_1, transpose_y = attn_33_transpose_y_1, x = transpose_382_cast_fp16, y = w_71_cast_fp16)[name = string("attn_33_cast_fp16")]; + tensor var_4172 = const()[name = string("op_4172"), val = tensor([1, 2048, 1, 1])]; + tensor input_171_cast_fp16 = reshape(shape = var_4172, x = attn_33_cast_fp16)[name = string("input_171_cast_fp16")]; + string attn_output_33_pad_type_0 = const()[name = string("attn_output_33_pad_type_0"), val = string("valid")]; + tensor attn_output_33_strides_0 = const()[name = string("attn_output_33_strides_0"), val = tensor([1, 1])]; + tensor attn_output_33_pad_0 = const()[name = string("attn_output_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_33_dilations_0 = const()[name = string("attn_output_33_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_33_groups_0 = const()[name = string("attn_output_33_groups_0"), val = int32(1)]; + tensor attn_output_33_cast_fp16 = conv(dilations = attn_output_33_dilations_0, groups = attn_output_33_groups_0, pad = attn_output_33_pad_0, pad_type = attn_output_33_pad_type_0, strides = attn_output_33_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_171_cast_fp16)[name = string("attn_output_33_cast_fp16")]; + tensor x_127_cast_fp16 = add(x = x_121_cast_fp16, y = attn_output_33_cast_fp16)[name = string("x_127_cast_fp16")]; + tensor var_4186_cast_fp16 = mul(x = x_127_cast_fp16, y = x_127_cast_fp16)[name = string("op_4186_cast_fp16")]; + tensor variance_139_axes_0 = const()[name = string("variance_139_axes_0"), val = tensor([1])]; + bool variance_139_keep_dims_0 = const()[name = string("variance_139_keep_dims_0"), val = bool(true)]; + tensor variance_139_cast_fp16 = reduce_mean(axes = variance_139_axes_0, keep_dims = variance_139_keep_dims_0, x = var_4186_cast_fp16)[name = string("variance_139_cast_fp16")]; + fp16 var_4189_to_fp16 = const()[name = string("op_4189_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4190_cast_fp16 = add(x = variance_139_cast_fp16, y = var_4189_to_fp16)[name = string("op_4190_cast_fp16")]; + fp32 var_4191_epsilon_0 = const()[name = string("op_4191_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4191_cast_fp16 = rsqrt(epsilon = var_4191_epsilon_0, x = var_4190_cast_fp16)[name = string("op_4191_cast_fp16")]; + tensor var_4192_cast_fp16 = mul(x = x_127_cast_fp16, y = var_4191_cast_fp16)[name = string("op_4192_cast_fp16")]; + tensor input_173_cast_fp16 = mul(x = var_4192_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_173_cast_fp16")]; + string input_175_pad_type_0 = const()[name = string("input_175_pad_type_0"), val = string("valid")]; + tensor input_175_strides_0 = const()[name = string("input_175_strides_0"), val = tensor([1, 1])]; + tensor input_175_pad_0 = const()[name = string("input_175_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_175_dilations_0 = const()[name = string("input_175_dilations_0"), val = tensor([1, 1])]; + int32 input_175_groups_0 = const()[name = string("input_175_groups_0"), val = int32(1)]; + tensor input_175_cast_fp16 = conv(dilations = input_175_dilations_0, groups = input_175_groups_0, pad = input_175_pad_0, pad_type = input_175_pad_type_0, strides = input_175_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_173_cast_fp16)[name = string("input_175_cast_fp16")]; + tensor var_4200_cast_fp16 = silu(x = input_175_cast_fp16)[name = string("op_4200_cast_fp16")]; + string var_4206_pad_type_0 = const()[name = string("op_4206_pad_type_0"), val = string("valid")]; + tensor var_4206_strides_0 = const()[name = string("op_4206_strides_0"), val = tensor([1, 1])]; + tensor var_4206_pad_0 = const()[name = string("op_4206_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4206_dilations_0 = const()[name = string("op_4206_dilations_0"), val = tensor([1, 1])]; + int32 var_4206_groups_0 = const()[name = string("op_4206_groups_0"), val = int32(1)]; + tensor var_4206_cast_fp16 = conv(dilations = var_4206_dilations_0, groups = var_4206_groups_0, pad = var_4206_pad_0, pad_type = var_4206_pad_type_0, strides = var_4206_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_173_cast_fp16)[name = string("op_4206_cast_fp16")]; + tensor input_177_cast_fp16 = mul(x = var_4200_cast_fp16, y = var_4206_cast_fp16)[name = string("input_177_cast_fp16")]; + string h_33_pad_type_0 = const()[name = string("h_33_pad_type_0"), val = string("valid")]; + tensor h_33_strides_0 = const()[name = string("h_33_strides_0"), val = tensor([1, 1])]; + tensor h_33_pad_0 = const()[name = string("h_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_33_dilations_0 = const()[name = string("h_33_dilations_0"), val = tensor([1, 1])]; + int32 h_33_groups_0 = const()[name = string("h_33_groups_0"), val = int32(1)]; + tensor h_33_cast_fp16 = conv(dilations = h_33_dilations_0, groups = h_33_groups_0, pad = h_33_pad_0, pad_type = h_33_pad_type_0, strides = h_33_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_177_cast_fp16)[name = string("h_33_cast_fp16")]; + tensor x_129_cast_fp16 = add(x = x_127_cast_fp16, y = h_33_cast_fp16)[name = string("x_129_cast_fp16")]; + tensor key_cache_35_begin_0 = const()[name = string("key_cache_35_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_35_end_0 = const()[name = string("key_cache_35_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_35_end_mask_0 = const()[name = string("key_cache_35_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_35_cast_fp16 = slice_by_index(begin = key_cache_35_begin_0, end = key_cache_35_end_0, end_mask = key_cache_35_end_mask_0, x = layer_key_caches_7_cast_fp16)[name = string("key_cache_35_cast_fp16")]; + tensor value_cache_35_begin_0 = const()[name = string("value_cache_35_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_35_end_0 = const()[name = string("value_cache_35_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_35_end_mask_0 = const()[name = string("value_cache_35_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_35_cast_fp16 = slice_by_index(begin = value_cache_35_begin_0, end = value_cache_35_end_0, end_mask = value_cache_35_end_mask_0, x = layer_value_caches_7_cast_fp16)[name = string("value_cache_35_cast_fp16")]; + int32 var_4259 = const()[name = string("op_4259"), val = int32(2)]; + int32 var_4263 = const()[name = string("op_4263"), val = int32(3)]; + tensor var_4278_cast_fp16 = mul(x = x_129_cast_fp16, y = x_129_cast_fp16)[name = string("op_4278_cast_fp16")]; + tensor variance_141_axes_0 = const()[name = string("variance_141_axes_0"), val = tensor([1])]; + bool variance_141_keep_dims_0 = const()[name = string("variance_141_keep_dims_0"), val = bool(true)]; + tensor variance_141_cast_fp16 = reduce_mean(axes = variance_141_axes_0, keep_dims = variance_141_keep_dims_0, x = var_4278_cast_fp16)[name = string("variance_141_cast_fp16")]; + fp16 var_4281_to_fp16 = const()[name = string("op_4281_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4282_cast_fp16 = add(x = variance_141_cast_fp16, y = var_4281_to_fp16)[name = string("op_4282_cast_fp16")]; + fp32 var_4283_epsilon_0 = const()[name = string("op_4283_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4283_cast_fp16 = rsqrt(epsilon = var_4283_epsilon_0, x = var_4282_cast_fp16)[name = string("op_4283_cast_fp16")]; + tensor var_4284_cast_fp16 = mul(x = x_129_cast_fp16, y = var_4283_cast_fp16)[name = string("op_4284_cast_fp16")]; + tensor input_179_cast_fp16 = mul(x = var_4284_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_179_cast_fp16")]; + string q_103_pad_type_0 = const()[name = string("q_103_pad_type_0"), val = string("valid")]; + tensor q_103_strides_0 = const()[name = string("q_103_strides_0"), val = tensor([1, 1])]; + tensor q_103_pad_0 = const()[name = string("q_103_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_103_dilations_0 = const()[name = string("q_103_dilations_0"), val = tensor([1, 1])]; + int32 q_103_groups_0 = const()[name = string("q_103_groups_0"), val = int32(1)]; + tensor q_103_cast_fp16 = conv(dilations = q_103_dilations_0, groups = q_103_groups_0, pad = q_103_pad_0, pad_type = q_103_pad_type_0, strides = q_103_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_179_cast_fp16)[name = string("q_103_cast_fp16")]; + string k_103_pad_type_0 = const()[name = string("k_103_pad_type_0"), val = string("valid")]; + tensor k_103_strides_0 = const()[name = string("k_103_strides_0"), val = tensor([1, 1])]; + tensor k_103_pad_0 = const()[name = string("k_103_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_103_dilations_0 = const()[name = string("k_103_dilations_0"), val = tensor([1, 1])]; + int32 k_103_groups_0 = const()[name = string("k_103_groups_0"), val = int32(1)]; + tensor k_103_cast_fp16 = conv(dilations = k_103_dilations_0, groups = k_103_groups_0, pad = k_103_pad_0, pad_type = k_103_pad_type_0, strides = k_103_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_179_cast_fp16)[name = string("k_103_cast_fp16")]; + string v_35_pad_type_0 = const()[name = string("v_35_pad_type_0"), val = string("valid")]; + tensor v_35_strides_0 = const()[name = string("v_35_strides_0"), val = tensor([1, 1])]; + tensor v_35_pad_0 = const()[name = string("v_35_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_35_dilations_0 = const()[name = string("v_35_dilations_0"), val = tensor([1, 1])]; + int32 v_35_groups_0 = const()[name = string("v_35_groups_0"), val = int32(1)]; + tensor v_35_cast_fp16 = conv(dilations = v_35_dilations_0, groups = v_35_groups_0, pad = v_35_pad_0, pad_type = v_35_pad_type_0, strides = v_35_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_179_cast_fp16)[name = string("v_35_cast_fp16")]; + tensor var_4318 = const()[name = string("op_4318"), val = tensor([16, 128, 1, 1])]; + tensor x_131_cast_fp16 = reshape(shape = var_4318, x = q_103_cast_fp16)[name = string("x_131_cast_fp16")]; + tensor var_4321_cast_fp16 = mul(x = x_131_cast_fp16, y = x_131_cast_fp16)[name = string("op_4321_cast_fp16")]; + tensor variance_143_axes_0 = const()[name = string("variance_143_axes_0"), val = tensor([1])]; + bool variance_143_keep_dims_0 = const()[name = string("variance_143_keep_dims_0"), val = bool(true)]; + tensor variance_143_cast_fp16 = reduce_mean(axes = variance_143_axes_0, keep_dims = variance_143_keep_dims_0, x = var_4321_cast_fp16)[name = string("variance_143_cast_fp16")]; + fp16 var_4324_to_fp16 = const()[name = string("op_4324_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4325_cast_fp16 = add(x = variance_143_cast_fp16, y = var_4324_to_fp16)[name = string("op_4325_cast_fp16")]; + fp32 var_4326_epsilon_0 = const()[name = string("op_4326_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4326_cast_fp16 = rsqrt(epsilon = var_4326_epsilon_0, x = var_4325_cast_fp16)[name = string("op_4326_cast_fp16")]; + tensor var_4327_cast_fp16 = mul(x = x_131_cast_fp16, y = var_4326_cast_fp16)[name = string("op_4327_cast_fp16")]; + tensor q_105_cast_fp16 = mul(x = var_4327_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_105_cast_fp16")]; + tensor var_4329 = const()[name = string("op_4329"), val = tensor([8, 128, 1, 1])]; + tensor x_133_cast_fp16 = reshape(shape = var_4329, x = k_103_cast_fp16)[name = string("x_133_cast_fp16")]; + tensor var_4332_cast_fp16 = mul(x = x_133_cast_fp16, y = x_133_cast_fp16)[name = string("op_4332_cast_fp16")]; + tensor variance_145_axes_0 = const()[name = string("variance_145_axes_0"), val = tensor([1])]; + bool variance_145_keep_dims_0 = const()[name = string("variance_145_keep_dims_0"), val = bool(true)]; + tensor variance_145_cast_fp16 = reduce_mean(axes = variance_145_axes_0, keep_dims = variance_145_keep_dims_0, x = var_4332_cast_fp16)[name = string("variance_145_cast_fp16")]; + fp16 var_4335_to_fp16 = const()[name = string("op_4335_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4336_cast_fp16 = add(x = variance_145_cast_fp16, y = var_4335_to_fp16)[name = string("op_4336_cast_fp16")]; + fp32 var_4337_epsilon_0 = const()[name = string("op_4337_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4337_cast_fp16 = rsqrt(epsilon = var_4337_epsilon_0, x = var_4336_cast_fp16)[name = string("op_4337_cast_fp16")]; + tensor var_4338_cast_fp16 = mul(x = x_133_cast_fp16, y = var_4337_cast_fp16)[name = string("op_4338_cast_fp16")]; + tensor k_105_cast_fp16 = mul(x = var_4338_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_105_cast_fp16")]; + tensor var_4340 = const()[name = string("op_4340"), val = tensor([1, 16, 128, 1])]; + tensor z_69_cast_fp16 = reshape(shape = var_4340, x = q_105_cast_fp16)[name = string("z_69_cast_fp16")]; + tensor var_4342 = const()[name = string("op_4342"), val = tensor([1, 8, 128, 1])]; + tensor z_71_cast_fp16 = reshape(shape = var_4342, x = k_105_cast_fp16)[name = string("z_71_cast_fp16")]; + tensor z1_69_begin_0 = const()[name = string("z1_69_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_69_end_0 = const()[name = string("z1_69_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_69_end_mask_0 = const()[name = string("z1_69_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_69_cast_fp16 = slice_by_index(begin = z1_69_begin_0, end = z1_69_end_0, end_mask = z1_69_end_mask_0, x = z_69_cast_fp16)[name = string("z1_69_cast_fp16")]; + tensor z2_69_begin_0 = const()[name = string("z2_69_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_69_end_0 = const()[name = string("z2_69_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_69_end_mask_0 = const()[name = string("z2_69_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_69_cast_fp16 = slice_by_index(begin = z2_69_begin_0, end = z2_69_end_0, end_mask = z2_69_end_mask_0, x = z_69_cast_fp16)[name = string("z2_69_cast_fp16")]; + tensor var_4350_cast_fp16 = mul(x = z_69_cast_fp16, y = cos_31_to_fp16)[name = string("op_4350_cast_fp16")]; + fp16 const_38_promoted_to_fp16 = const()[name = string("const_38_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4351_cast_fp16 = mul(x = z2_69_cast_fp16, y = const_38_promoted_to_fp16)[name = string("op_4351_cast_fp16")]; + bool var_4353_interleave_0 = const()[name = string("op_4353_interleave_0"), val = bool(false)]; + tensor var_4353_cast_fp16 = concat(axis = var_4259, interleave = var_4353_interleave_0, values = (var_4351_cast_fp16, z1_69_cast_fp16))[name = string("op_4353_cast_fp16")]; + tensor var_4354_cast_fp16 = mul(x = var_4353_cast_fp16, y = sin_31_to_fp16)[name = string("op_4354_cast_fp16")]; + tensor q_107_cast_fp16 = add(x = var_4350_cast_fp16, y = var_4354_cast_fp16)[name = string("q_107_cast_fp16")]; + tensor z1_71_begin_0 = const()[name = string("z1_71_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_71_end_0 = const()[name = string("z1_71_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_71_end_mask_0 = const()[name = string("z1_71_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_71_cast_fp16 = slice_by_index(begin = z1_71_begin_0, end = z1_71_end_0, end_mask = z1_71_end_mask_0, x = z_71_cast_fp16)[name = string("z1_71_cast_fp16")]; + tensor z2_71_begin_0 = const()[name = string("z2_71_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_71_end_0 = const()[name = string("z2_71_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_71_end_mask_0 = const()[name = string("z2_71_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_71_cast_fp16 = slice_by_index(begin = z2_71_begin_0, end = z2_71_end_0, end_mask = z2_71_end_mask_0, x = z_71_cast_fp16)[name = string("z2_71_cast_fp16")]; + tensor var_4362_cast_fp16 = mul(x = z_71_cast_fp16, y = cos_31_to_fp16)[name = string("op_4362_cast_fp16")]; + fp16 const_39_promoted_to_fp16 = const()[name = string("const_39_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4363_cast_fp16 = mul(x = z2_71_cast_fp16, y = const_39_promoted_to_fp16)[name = string("op_4363_cast_fp16")]; + bool var_4365_interleave_0 = const()[name = string("op_4365_interleave_0"), val = bool(false)]; + tensor var_4365_cast_fp16 = concat(axis = var_4259, interleave = var_4365_interleave_0, values = (var_4363_cast_fp16, z1_71_cast_fp16))[name = string("op_4365_cast_fp16")]; + tensor var_4366_cast_fp16 = mul(x = var_4365_cast_fp16, y = sin_31_to_fp16)[name = string("op_4366_cast_fp16")]; + tensor k_107_cast_fp16 = add(x = var_4362_cast_fp16, y = var_4366_cast_fp16)[name = string("k_107_cast_fp16")]; + tensor var_4368 = const()[name = string("op_4368"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_35_cast_fp16 = reshape(shape = var_4368, x = k_107_cast_fp16)[name = string("cur_key_35_cast_fp16")]; + tensor var_4370_to_fp16 = const()[name = string("op_4370_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635072)))]; + tensor var_4371_cast_fp16 = mul(x = key_cache_35_cast_fp16, y = var_4370_to_fp16)[name = string("op_4371_cast_fp16")]; + tensor upd_35_to_fp16 = const()[name = string("upd_35_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635200)))]; + tensor var_4372_cast_fp16 = mul(x = cur_key_35_cast_fp16, y = upd_35_to_fp16)[name = string("op_4372_cast_fp16")]; + tensor key_35_cast_fp16 = add(x = var_4371_cast_fp16, y = var_4372_cast_fp16)[name = string("key_35_cast_fp16")]; + tensor var_4374_to_fp16 = const()[name = string("op_4374_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635072)))]; + tensor var_4375_cast_fp16 = mul(x = value_cache_35_cast_fp16, y = var_4374_to_fp16)[name = string("op_4375_cast_fp16")]; + tensor var_4376_cast_fp16 = mul(x = v_35_cast_fp16, y = upd_35_to_fp16)[name = string("op_4376_cast_fp16")]; + tensor value_35_cast_fp16 = add(x = var_4375_cast_fp16, y = var_4376_cast_fp16)[name = string("value_35_cast_fp16")]; + tensor var_4378 = const()[name = string("op_4378"), val = tensor([1, 8, 128, 16])]; + tensor kh_69_cast_fp16 = reshape(shape = var_4378, x = key_35_cast_fp16)[name = string("kh_69_cast_fp16")]; + tensor var_4380 = const()[name = string("op_4380"), val = tensor([1, 8, 128, 16])]; + tensor vh_69_cast_fp16 = reshape(shape = var_4380, x = value_35_cast_fp16)[name = string("vh_69_cast_fp16")]; + tensor transpose_68_perm_0 = const()[name = string("transpose_68_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_34_reps_0 = const()[name = string("tile_34_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_68_cast_fp16 = transpose(perm = transpose_68_perm_0, x = kh_69_cast_fp16)[name = string("transpose_377")]; + tensor tile_34_cast_fp16 = tile(reps = tile_34_reps_0, x = transpose_68_cast_fp16)[name = string("tile_34_cast_fp16")]; + tensor concat_86 = const()[name = string("concat_86"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_68_cast_fp16 = reshape(shape = concat_86, x = tile_34_cast_fp16)[name = string("reshape_68_cast_fp16")]; + tensor transpose_69_perm_0 = const()[name = string("transpose_69_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_87 = const()[name = string("concat_87"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_69_cast_fp16 = transpose(perm = transpose_69_perm_0, x = reshape_68_cast_fp16)[name = string("transpose_376")]; + tensor reshape_69_cast_fp16 = reshape(shape = concat_87, x = transpose_69_cast_fp16)[name = string("reshape_69_cast_fp16")]; + tensor transpose_70_perm_0 = const()[name = string("transpose_70_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_35_reps_0 = const()[name = string("tile_35_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_70_cast_fp16 = transpose(perm = transpose_70_perm_0, x = vh_69_cast_fp16)[name = string("transpose_375")]; + tensor tile_35_cast_fp16 = tile(reps = tile_35_reps_0, x = transpose_70_cast_fp16)[name = string("tile_35_cast_fp16")]; + tensor concat_88 = const()[name = string("concat_88"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_70_cast_fp16 = reshape(shape = concat_88, x = tile_35_cast_fp16)[name = string("reshape_70_cast_fp16")]; + tensor transpose_71_perm_0 = const()[name = string("transpose_71_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_89 = const()[name = string("concat_89"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_71_cast_fp16 = transpose(perm = transpose_71_perm_0, x = reshape_70_cast_fp16)[name = string("transpose_374")]; + tensor reshape_71_cast_fp16 = reshape(shape = concat_89, x = transpose_71_cast_fp16)[name = string("reshape_71_cast_fp16")]; + fp16 var_4384_to_fp16 = const()[name = string("op_4384_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_4385_cast_fp16 = mul(x = q_107_cast_fp16, y = var_4384_to_fp16)[name = string("op_4385_cast_fp16")]; + tensor transpose_385_perm_0 = const()[name = string("transpose_385_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_73_transpose_x_1 = const()[name = string("w_73_transpose_x_1"), val = bool(true)]; + bool w_73_transpose_y_1 = const()[name = string("w_73_transpose_y_1"), val = bool(false)]; + tensor transpose_385_cast_fp16 = transpose(perm = transpose_385_perm_0, x = reshape_69_cast_fp16)[name = string("transpose_373")]; + tensor w_73_cast_fp16 = matmul(transpose_x = w_73_transpose_x_1, transpose_y = w_73_transpose_y_1, x = var_4385_cast_fp16, y = transpose_385_cast_fp16)[name = string("w_73_cast_fp16")]; + tensor pad_35_to_fp16 = const()[name = string("pad_35_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635328)))]; + tensor var_4388_cast_fp16 = add(x = w_73_cast_fp16, y = pad_35_to_fp16)[name = string("op_4388_cast_fp16")]; + tensor w_75_cast_fp16 = softmax(axis = var_4263, x = var_4388_cast_fp16)[name = string("w_75_cast_fp16")]; + tensor transpose_386_perm_0 = const()[name = string("transpose_386_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_35_transpose_x_1 = const()[name = string("attn_35_transpose_x_1"), val = bool(false)]; + bool attn_35_transpose_y_1 = const()[name = string("attn_35_transpose_y_1"), val = bool(true)]; + tensor transpose_386_cast_fp16 = transpose(perm = transpose_386_perm_0, x = reshape_71_cast_fp16)[name = string("transpose_372")]; + tensor attn_35_cast_fp16 = matmul(transpose_x = attn_35_transpose_x_1, transpose_y = attn_35_transpose_y_1, x = transpose_386_cast_fp16, y = w_75_cast_fp16)[name = string("attn_35_cast_fp16")]; + tensor var_4392 = const()[name = string("op_4392"), val = tensor([1, 2048, 1, 1])]; + tensor input_181_cast_fp16 = reshape(shape = var_4392, x = attn_35_cast_fp16)[name = string("input_181_cast_fp16")]; + string attn_output_35_pad_type_0 = const()[name = string("attn_output_35_pad_type_0"), val = string("valid")]; + tensor attn_output_35_strides_0 = const()[name = string("attn_output_35_strides_0"), val = tensor([1, 1])]; + tensor attn_output_35_pad_0 = const()[name = string("attn_output_35_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_35_dilations_0 = const()[name = string("attn_output_35_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_35_groups_0 = const()[name = string("attn_output_35_groups_0"), val = int32(1)]; + tensor attn_output_35_cast_fp16 = conv(dilations = attn_output_35_dilations_0, groups = attn_output_35_groups_0, pad = attn_output_35_pad_0, pad_type = attn_output_35_pad_type_0, strides = attn_output_35_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_181_cast_fp16)[name = string("attn_output_35_cast_fp16")]; + tensor x_135_cast_fp16 = add(x = x_129_cast_fp16, y = attn_output_35_cast_fp16)[name = string("x_135_cast_fp16")]; + tensor var_4406_cast_fp16 = mul(x = x_135_cast_fp16, y = x_135_cast_fp16)[name = string("op_4406_cast_fp16")]; + tensor variance_147_axes_0 = const()[name = string("variance_147_axes_0"), val = tensor([1])]; + bool variance_147_keep_dims_0 = const()[name = string("variance_147_keep_dims_0"), val = bool(true)]; + tensor variance_147_cast_fp16 = reduce_mean(axes = variance_147_axes_0, keep_dims = variance_147_keep_dims_0, x = var_4406_cast_fp16)[name = string("variance_147_cast_fp16")]; + fp16 var_4409_to_fp16 = const()[name = string("op_4409_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4410_cast_fp16 = add(x = variance_147_cast_fp16, y = var_4409_to_fp16)[name = string("op_4410_cast_fp16")]; + fp32 var_4411_epsilon_0 = const()[name = string("op_4411_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4411_cast_fp16 = rsqrt(epsilon = var_4411_epsilon_0, x = var_4410_cast_fp16)[name = string("op_4411_cast_fp16")]; + tensor var_4412_cast_fp16 = mul(x = x_135_cast_fp16, y = var_4411_cast_fp16)[name = string("op_4412_cast_fp16")]; + tensor input_183_cast_fp16 = mul(x = var_4412_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_183_cast_fp16")]; + string input_185_pad_type_0 = const()[name = string("input_185_pad_type_0"), val = string("valid")]; + tensor input_185_strides_0 = const()[name = string("input_185_strides_0"), val = tensor([1, 1])]; + tensor input_185_pad_0 = const()[name = string("input_185_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_185_dilations_0 = const()[name = string("input_185_dilations_0"), val = tensor([1, 1])]; + int32 input_185_groups_0 = const()[name = string("input_185_groups_0"), val = int32(1)]; + tensor input_185_cast_fp16 = conv(dilations = input_185_dilations_0, groups = input_185_groups_0, pad = input_185_pad_0, pad_type = input_185_pad_type_0, strides = input_185_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_183_cast_fp16)[name = string("input_185_cast_fp16")]; + tensor var_4420_cast_fp16 = silu(x = input_185_cast_fp16)[name = string("op_4420_cast_fp16")]; + string var_4426_pad_type_0 = const()[name = string("op_4426_pad_type_0"), val = string("valid")]; + tensor var_4426_strides_0 = const()[name = string("op_4426_strides_0"), val = tensor([1, 1])]; + tensor var_4426_pad_0 = const()[name = string("op_4426_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4426_dilations_0 = const()[name = string("op_4426_dilations_0"), val = tensor([1, 1])]; + int32 var_4426_groups_0 = const()[name = string("op_4426_groups_0"), val = int32(1)]; + tensor var_4426_cast_fp16 = conv(dilations = var_4426_dilations_0, groups = var_4426_groups_0, pad = var_4426_pad_0, pad_type = var_4426_pad_type_0, strides = var_4426_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_183_cast_fp16)[name = string("op_4426_cast_fp16")]; + tensor input_187_cast_fp16 = mul(x = var_4420_cast_fp16, y = var_4426_cast_fp16)[name = string("input_187_cast_fp16")]; + string h_35_pad_type_0 = const()[name = string("h_35_pad_type_0"), val = string("valid")]; + tensor h_35_strides_0 = const()[name = string("h_35_strides_0"), val = tensor([1, 1])]; + tensor h_35_pad_0 = const()[name = string("h_35_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_35_dilations_0 = const()[name = string("h_35_dilations_0"), val = tensor([1, 1])]; + int32 h_35_groups_0 = const()[name = string("h_35_groups_0"), val = int32(1)]; + tensor h_35_cast_fp16 = conv(dilations = h_35_dilations_0, groups = h_35_groups_0, pad = h_35_pad_0, pad_type = h_35_pad_type_0, strides = h_35_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_187_cast_fp16)[name = string("h_35_cast_fp16")]; + tensor x_137_cast_fp16 = add(x = x_135_cast_fp16, y = h_35_cast_fp16)[name = string("x_137_cast_fp16")]; + tensor key_cache_37_begin_0 = const()[name = string("key_cache_37_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_37_end_0 = const()[name = string("key_cache_37_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_37_end_mask_0 = const()[name = string("key_cache_37_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_37_cast_fp16 = slice_by_index(begin = key_cache_37_begin_0, end = key_cache_37_end_0, end_mask = key_cache_37_end_mask_0, x = layer_key_caches_7_cast_fp16)[name = string("key_cache_37_cast_fp16")]; + tensor value_cache_37_begin_0 = const()[name = string("value_cache_37_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_37_end_0 = const()[name = string("value_cache_37_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_37_end_mask_0 = const()[name = string("value_cache_37_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_37_cast_fp16 = slice_by_index(begin = value_cache_37_begin_0, end = value_cache_37_end_0, end_mask = value_cache_37_end_mask_0, x = layer_value_caches_7_cast_fp16)[name = string("value_cache_37_cast_fp16")]; + int32 var_4479 = const()[name = string("op_4479"), val = int32(2)]; + int32 var_4483 = const()[name = string("op_4483"), val = int32(3)]; + tensor var_4498_cast_fp16 = mul(x = x_137_cast_fp16, y = x_137_cast_fp16)[name = string("op_4498_cast_fp16")]; + tensor variance_149_axes_0 = const()[name = string("variance_149_axes_0"), val = tensor([1])]; + bool variance_149_keep_dims_0 = const()[name = string("variance_149_keep_dims_0"), val = bool(true)]; + tensor variance_149_cast_fp16 = reduce_mean(axes = variance_149_axes_0, keep_dims = variance_149_keep_dims_0, x = var_4498_cast_fp16)[name = string("variance_149_cast_fp16")]; + fp16 var_4501_to_fp16 = const()[name = string("op_4501_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4502_cast_fp16 = add(x = variance_149_cast_fp16, y = var_4501_to_fp16)[name = string("op_4502_cast_fp16")]; + fp32 var_4503_epsilon_0 = const()[name = string("op_4503_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4503_cast_fp16 = rsqrt(epsilon = var_4503_epsilon_0, x = var_4502_cast_fp16)[name = string("op_4503_cast_fp16")]; + tensor var_4504_cast_fp16 = mul(x = x_137_cast_fp16, y = var_4503_cast_fp16)[name = string("op_4504_cast_fp16")]; + tensor input_189_cast_fp16 = mul(x = var_4504_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_189_cast_fp16")]; + string q_109_pad_type_0 = const()[name = string("q_109_pad_type_0"), val = string("valid")]; + tensor q_109_strides_0 = const()[name = string("q_109_strides_0"), val = tensor([1, 1])]; + tensor q_109_pad_0 = const()[name = string("q_109_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_109_dilations_0 = const()[name = string("q_109_dilations_0"), val = tensor([1, 1])]; + int32 q_109_groups_0 = const()[name = string("q_109_groups_0"), val = int32(1)]; + tensor q_109_cast_fp16 = conv(dilations = q_109_dilations_0, groups = q_109_groups_0, pad = q_109_pad_0, pad_type = q_109_pad_type_0, strides = q_109_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_189_cast_fp16)[name = string("q_109_cast_fp16")]; + string k_109_pad_type_0 = const()[name = string("k_109_pad_type_0"), val = string("valid")]; + tensor k_109_strides_0 = const()[name = string("k_109_strides_0"), val = tensor([1, 1])]; + tensor k_109_pad_0 = const()[name = string("k_109_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_109_dilations_0 = const()[name = string("k_109_dilations_0"), val = tensor([1, 1])]; + int32 k_109_groups_0 = const()[name = string("k_109_groups_0"), val = int32(1)]; + tensor k_109_cast_fp16 = conv(dilations = k_109_dilations_0, groups = k_109_groups_0, pad = k_109_pad_0, pad_type = k_109_pad_type_0, strides = k_109_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_189_cast_fp16)[name = string("k_109_cast_fp16")]; + string v_37_pad_type_0 = const()[name = string("v_37_pad_type_0"), val = string("valid")]; + tensor v_37_strides_0 = const()[name = string("v_37_strides_0"), val = tensor([1, 1])]; + tensor v_37_pad_0 = const()[name = string("v_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_37_dilations_0 = const()[name = string("v_37_dilations_0"), val = tensor([1, 1])]; + int32 v_37_groups_0 = const()[name = string("v_37_groups_0"), val = int32(1)]; + tensor v_37_cast_fp16 = conv(dilations = v_37_dilations_0, groups = v_37_groups_0, pad = v_37_pad_0, pad_type = v_37_pad_type_0, strides = v_37_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_189_cast_fp16)[name = string("v_37_cast_fp16")]; + tensor var_4538 = const()[name = string("op_4538"), val = tensor([16, 128, 1, 1])]; + tensor x_139_cast_fp16 = reshape(shape = var_4538, x = q_109_cast_fp16)[name = string("x_139_cast_fp16")]; + tensor var_4541_cast_fp16 = mul(x = x_139_cast_fp16, y = x_139_cast_fp16)[name = string("op_4541_cast_fp16")]; + tensor variance_151_axes_0 = const()[name = string("variance_151_axes_0"), val = tensor([1])]; + bool variance_151_keep_dims_0 = const()[name = string("variance_151_keep_dims_0"), val = bool(true)]; + tensor variance_151_cast_fp16 = reduce_mean(axes = variance_151_axes_0, keep_dims = variance_151_keep_dims_0, x = var_4541_cast_fp16)[name = string("variance_151_cast_fp16")]; + fp16 var_4544_to_fp16 = const()[name = string("op_4544_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4545_cast_fp16 = add(x = variance_151_cast_fp16, y = var_4544_to_fp16)[name = string("op_4545_cast_fp16")]; + fp32 var_4546_epsilon_0 = const()[name = string("op_4546_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4546_cast_fp16 = rsqrt(epsilon = var_4546_epsilon_0, x = var_4545_cast_fp16)[name = string("op_4546_cast_fp16")]; + tensor var_4547_cast_fp16 = mul(x = x_139_cast_fp16, y = var_4546_cast_fp16)[name = string("op_4547_cast_fp16")]; + tensor q_111_cast_fp16 = mul(x = var_4547_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_111_cast_fp16")]; + tensor var_4549 = const()[name = string("op_4549"), val = tensor([8, 128, 1, 1])]; + tensor x_141_cast_fp16 = reshape(shape = var_4549, x = k_109_cast_fp16)[name = string("x_141_cast_fp16")]; + tensor var_4552_cast_fp16 = mul(x = x_141_cast_fp16, y = x_141_cast_fp16)[name = string("op_4552_cast_fp16")]; + tensor variance_153_axes_0 = const()[name = string("variance_153_axes_0"), val = tensor([1])]; + bool variance_153_keep_dims_0 = const()[name = string("variance_153_keep_dims_0"), val = bool(true)]; + tensor variance_153_cast_fp16 = reduce_mean(axes = variance_153_axes_0, keep_dims = variance_153_keep_dims_0, x = var_4552_cast_fp16)[name = string("variance_153_cast_fp16")]; + fp16 var_4555_to_fp16 = const()[name = string("op_4555_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4556_cast_fp16 = add(x = variance_153_cast_fp16, y = var_4555_to_fp16)[name = string("op_4556_cast_fp16")]; + fp32 var_4557_epsilon_0 = const()[name = string("op_4557_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4557_cast_fp16 = rsqrt(epsilon = var_4557_epsilon_0, x = var_4556_cast_fp16)[name = string("op_4557_cast_fp16")]; + tensor var_4558_cast_fp16 = mul(x = x_141_cast_fp16, y = var_4557_cast_fp16)[name = string("op_4558_cast_fp16")]; + tensor k_111_cast_fp16 = mul(x = var_4558_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_111_cast_fp16")]; + tensor var_4560 = const()[name = string("op_4560"), val = tensor([1, 16, 128, 1])]; + tensor z_73_cast_fp16 = reshape(shape = var_4560, x = q_111_cast_fp16)[name = string("z_73_cast_fp16")]; + tensor var_4562 = const()[name = string("op_4562"), val = tensor([1, 8, 128, 1])]; + tensor z_75_cast_fp16 = reshape(shape = var_4562, x = k_111_cast_fp16)[name = string("z_75_cast_fp16")]; + tensor z1_73_begin_0 = const()[name = string("z1_73_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_73_end_0 = const()[name = string("z1_73_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_73_end_mask_0 = const()[name = string("z1_73_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_73_cast_fp16 = slice_by_index(begin = z1_73_begin_0, end = z1_73_end_0, end_mask = z1_73_end_mask_0, x = z_73_cast_fp16)[name = string("z1_73_cast_fp16")]; + tensor z2_73_begin_0 = const()[name = string("z2_73_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_73_end_0 = const()[name = string("z2_73_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_73_end_mask_0 = const()[name = string("z2_73_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_73_cast_fp16 = slice_by_index(begin = z2_73_begin_0, end = z2_73_end_0, end_mask = z2_73_end_mask_0, x = z_73_cast_fp16)[name = string("z2_73_cast_fp16")]; + tensor var_4570_cast_fp16 = mul(x = z_73_cast_fp16, y = cos_31_to_fp16)[name = string("op_4570_cast_fp16")]; + fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4571_cast_fp16 = mul(x = z2_73_cast_fp16, y = const_40_promoted_to_fp16)[name = string("op_4571_cast_fp16")]; + bool var_4573_interleave_0 = const()[name = string("op_4573_interleave_0"), val = bool(false)]; + tensor var_4573_cast_fp16 = concat(axis = var_4479, interleave = var_4573_interleave_0, values = (var_4571_cast_fp16, z1_73_cast_fp16))[name = string("op_4573_cast_fp16")]; + tensor var_4574_cast_fp16 = mul(x = var_4573_cast_fp16, y = sin_31_to_fp16)[name = string("op_4574_cast_fp16")]; + tensor q_113_cast_fp16 = add(x = var_4570_cast_fp16, y = var_4574_cast_fp16)[name = string("q_113_cast_fp16")]; + tensor z1_75_begin_0 = const()[name = string("z1_75_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_75_end_0 = const()[name = string("z1_75_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_75_end_mask_0 = const()[name = string("z1_75_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_75_cast_fp16 = slice_by_index(begin = z1_75_begin_0, end = z1_75_end_0, end_mask = z1_75_end_mask_0, x = z_75_cast_fp16)[name = string("z1_75_cast_fp16")]; + tensor z2_75_begin_0 = const()[name = string("z2_75_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_75_end_0 = const()[name = string("z2_75_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_75_end_mask_0 = const()[name = string("z2_75_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_75_cast_fp16 = slice_by_index(begin = z2_75_begin_0, end = z2_75_end_0, end_mask = z2_75_end_mask_0, x = z_75_cast_fp16)[name = string("z2_75_cast_fp16")]; + tensor var_4582_cast_fp16 = mul(x = z_75_cast_fp16, y = cos_31_to_fp16)[name = string("op_4582_cast_fp16")]; + fp16 const_41_promoted_to_fp16 = const()[name = string("const_41_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4583_cast_fp16 = mul(x = z2_75_cast_fp16, y = const_41_promoted_to_fp16)[name = string("op_4583_cast_fp16")]; + bool var_4585_interleave_0 = const()[name = string("op_4585_interleave_0"), val = bool(false)]; + tensor var_4585_cast_fp16 = concat(axis = var_4479, interleave = var_4585_interleave_0, values = (var_4583_cast_fp16, z1_75_cast_fp16))[name = string("op_4585_cast_fp16")]; + tensor var_4586_cast_fp16 = mul(x = var_4585_cast_fp16, y = sin_31_to_fp16)[name = string("op_4586_cast_fp16")]; + tensor k_113_cast_fp16 = add(x = var_4582_cast_fp16, y = var_4586_cast_fp16)[name = string("k_113_cast_fp16")]; + tensor var_4588 = const()[name = string("op_4588"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_37_cast_fp16 = reshape(shape = var_4588, x = k_113_cast_fp16)[name = string("cur_key_37_cast_fp16")]; + tensor var_4590_to_fp16 = const()[name = string("op_4590_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635072)))]; + tensor var_4591_cast_fp16 = mul(x = key_cache_37_cast_fp16, y = var_4590_to_fp16)[name = string("op_4591_cast_fp16")]; + tensor upd_37_to_fp16 = const()[name = string("upd_37_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635200)))]; + tensor var_4592_cast_fp16 = mul(x = cur_key_37_cast_fp16, y = upd_37_to_fp16)[name = string("op_4592_cast_fp16")]; + tensor key_37_cast_fp16 = add(x = var_4591_cast_fp16, y = var_4592_cast_fp16)[name = string("key_37_cast_fp16")]; + tensor var_4594_to_fp16 = const()[name = string("op_4594_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635072)))]; + tensor var_4595_cast_fp16 = mul(x = value_cache_37_cast_fp16, y = var_4594_to_fp16)[name = string("op_4595_cast_fp16")]; + tensor var_4596_cast_fp16 = mul(x = v_37_cast_fp16, y = upd_37_to_fp16)[name = string("op_4596_cast_fp16")]; + tensor value_37_cast_fp16 = add(x = var_4595_cast_fp16, y = var_4596_cast_fp16)[name = string("value_37_cast_fp16")]; + tensor var_4598 = const()[name = string("op_4598"), val = tensor([1, 8, 128, 16])]; + tensor kh_73_cast_fp16 = reshape(shape = var_4598, x = key_37_cast_fp16)[name = string("kh_73_cast_fp16")]; + tensor var_4600 = const()[name = string("op_4600"), val = tensor([1, 8, 128, 16])]; + tensor vh_73_cast_fp16 = reshape(shape = var_4600, x = value_37_cast_fp16)[name = string("vh_73_cast_fp16")]; + tensor transpose_72_perm_0 = const()[name = string("transpose_72_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_36_reps_0 = const()[name = string("tile_36_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_72_cast_fp16 = transpose(perm = transpose_72_perm_0, x = kh_73_cast_fp16)[name = string("transpose_371")]; + tensor tile_36_cast_fp16 = tile(reps = tile_36_reps_0, x = transpose_72_cast_fp16)[name = string("tile_36_cast_fp16")]; + tensor concat_90 = const()[name = string("concat_90"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_72_cast_fp16 = reshape(shape = concat_90, x = tile_36_cast_fp16)[name = string("reshape_72_cast_fp16")]; + tensor transpose_73_perm_0 = const()[name = string("transpose_73_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_91 = const()[name = string("concat_91"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_73_cast_fp16 = transpose(perm = transpose_73_perm_0, x = reshape_72_cast_fp16)[name = string("transpose_370")]; + tensor reshape_73_cast_fp16 = reshape(shape = concat_91, x = transpose_73_cast_fp16)[name = string("reshape_73_cast_fp16")]; + tensor transpose_74_perm_0 = const()[name = string("transpose_74_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_37_reps_0 = const()[name = string("tile_37_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_74_cast_fp16 = transpose(perm = transpose_74_perm_0, x = vh_73_cast_fp16)[name = string("transpose_369")]; + tensor tile_37_cast_fp16 = tile(reps = tile_37_reps_0, x = transpose_74_cast_fp16)[name = string("tile_37_cast_fp16")]; + tensor concat_92 = const()[name = string("concat_92"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_74_cast_fp16 = reshape(shape = concat_92, x = tile_37_cast_fp16)[name = string("reshape_74_cast_fp16")]; + tensor transpose_75_perm_0 = const()[name = string("transpose_75_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_93 = const()[name = string("concat_93"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_75_cast_fp16 = transpose(perm = transpose_75_perm_0, x = reshape_74_cast_fp16)[name = string("transpose_368")]; + tensor reshape_75_cast_fp16 = reshape(shape = concat_93, x = transpose_75_cast_fp16)[name = string("reshape_75_cast_fp16")]; + fp16 var_4604_to_fp16 = const()[name = string("op_4604_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_4605_cast_fp16 = mul(x = q_113_cast_fp16, y = var_4604_to_fp16)[name = string("op_4605_cast_fp16")]; + tensor transpose_389_perm_0 = const()[name = string("transpose_389_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_77_transpose_x_1 = const()[name = string("w_77_transpose_x_1"), val = bool(true)]; + bool w_77_transpose_y_1 = const()[name = string("w_77_transpose_y_1"), val = bool(false)]; + tensor transpose_389_cast_fp16 = transpose(perm = transpose_389_perm_0, x = reshape_73_cast_fp16)[name = string("transpose_367")]; + tensor w_77_cast_fp16 = matmul(transpose_x = w_77_transpose_x_1, transpose_y = w_77_transpose_y_1, x = var_4605_cast_fp16, y = transpose_389_cast_fp16)[name = string("w_77_cast_fp16")]; + tensor pad_37_to_fp16 = const()[name = string("pad_37_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635328)))]; + tensor var_4608_cast_fp16 = add(x = w_77_cast_fp16, y = pad_37_to_fp16)[name = string("op_4608_cast_fp16")]; + tensor w_79_cast_fp16 = softmax(axis = var_4483, x = var_4608_cast_fp16)[name = string("w_79_cast_fp16")]; + tensor transpose_390_perm_0 = const()[name = string("transpose_390_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_37_transpose_x_1 = const()[name = string("attn_37_transpose_x_1"), val = bool(false)]; + bool attn_37_transpose_y_1 = const()[name = string("attn_37_transpose_y_1"), val = bool(true)]; + tensor transpose_390_cast_fp16 = transpose(perm = transpose_390_perm_0, x = reshape_75_cast_fp16)[name = string("transpose_366")]; + tensor attn_37_cast_fp16 = matmul(transpose_x = attn_37_transpose_x_1, transpose_y = attn_37_transpose_y_1, x = transpose_390_cast_fp16, y = w_79_cast_fp16)[name = string("attn_37_cast_fp16")]; + tensor var_4612 = const()[name = string("op_4612"), val = tensor([1, 2048, 1, 1])]; + tensor input_191_cast_fp16 = reshape(shape = var_4612, x = attn_37_cast_fp16)[name = string("input_191_cast_fp16")]; + string attn_output_37_pad_type_0 = const()[name = string("attn_output_37_pad_type_0"), val = string("valid")]; + tensor attn_output_37_strides_0 = const()[name = string("attn_output_37_strides_0"), val = tensor([1, 1])]; + tensor attn_output_37_pad_0 = const()[name = string("attn_output_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_37_dilations_0 = const()[name = string("attn_output_37_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_37_groups_0 = const()[name = string("attn_output_37_groups_0"), val = int32(1)]; + tensor attn_output_37_cast_fp16 = conv(dilations = attn_output_37_dilations_0, groups = attn_output_37_groups_0, pad = attn_output_37_pad_0, pad_type = attn_output_37_pad_type_0, strides = attn_output_37_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_191_cast_fp16)[name = string("attn_output_37_cast_fp16")]; + tensor x_143_cast_fp16 = add(x = x_137_cast_fp16, y = attn_output_37_cast_fp16)[name = string("x_143_cast_fp16")]; + tensor var_4626_cast_fp16 = mul(x = x_143_cast_fp16, y = x_143_cast_fp16)[name = string("op_4626_cast_fp16")]; + tensor variance_155_axes_0 = const()[name = string("variance_155_axes_0"), val = tensor([1])]; + bool variance_155_keep_dims_0 = const()[name = string("variance_155_keep_dims_0"), val = bool(true)]; + tensor variance_155_cast_fp16 = reduce_mean(axes = variance_155_axes_0, keep_dims = variance_155_keep_dims_0, x = var_4626_cast_fp16)[name = string("variance_155_cast_fp16")]; + fp16 var_4629_to_fp16 = const()[name = string("op_4629_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4630_cast_fp16 = add(x = variance_155_cast_fp16, y = var_4629_to_fp16)[name = string("op_4630_cast_fp16")]; + fp32 var_4631_epsilon_0 = const()[name = string("op_4631_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4631_cast_fp16 = rsqrt(epsilon = var_4631_epsilon_0, x = var_4630_cast_fp16)[name = string("op_4631_cast_fp16")]; + tensor var_4632_cast_fp16 = mul(x = x_143_cast_fp16, y = var_4631_cast_fp16)[name = string("op_4632_cast_fp16")]; + tensor input_193_cast_fp16 = mul(x = var_4632_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_193_cast_fp16")]; + string input_195_pad_type_0 = const()[name = string("input_195_pad_type_0"), val = string("valid")]; + tensor input_195_strides_0 = const()[name = string("input_195_strides_0"), val = tensor([1, 1])]; + tensor input_195_pad_0 = const()[name = string("input_195_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_195_dilations_0 = const()[name = string("input_195_dilations_0"), val = tensor([1, 1])]; + int32 input_195_groups_0 = const()[name = string("input_195_groups_0"), val = int32(1)]; + tensor input_195_cast_fp16 = conv(dilations = input_195_dilations_0, groups = input_195_groups_0, pad = input_195_pad_0, pad_type = input_195_pad_type_0, strides = input_195_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_193_cast_fp16)[name = string("input_195_cast_fp16")]; + tensor var_4640_cast_fp16 = silu(x = input_195_cast_fp16)[name = string("op_4640_cast_fp16")]; + string var_4646_pad_type_0 = const()[name = string("op_4646_pad_type_0"), val = string("valid")]; + tensor var_4646_strides_0 = const()[name = string("op_4646_strides_0"), val = tensor([1, 1])]; + tensor var_4646_pad_0 = const()[name = string("op_4646_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4646_dilations_0 = const()[name = string("op_4646_dilations_0"), val = tensor([1, 1])]; + int32 var_4646_groups_0 = const()[name = string("op_4646_groups_0"), val = int32(1)]; + tensor var_4646_cast_fp16 = conv(dilations = var_4646_dilations_0, groups = var_4646_groups_0, pad = var_4646_pad_0, pad_type = var_4646_pad_type_0, strides = var_4646_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_193_cast_fp16)[name = string("op_4646_cast_fp16")]; + tensor input_197_cast_fp16 = mul(x = var_4640_cast_fp16, y = var_4646_cast_fp16)[name = string("input_197_cast_fp16")]; + string h_37_pad_type_0 = const()[name = string("h_37_pad_type_0"), val = string("valid")]; + tensor h_37_strides_0 = const()[name = string("h_37_strides_0"), val = tensor([1, 1])]; + tensor h_37_pad_0 = const()[name = string("h_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_37_dilations_0 = const()[name = string("h_37_dilations_0"), val = tensor([1, 1])]; + int32 h_37_groups_0 = const()[name = string("h_37_groups_0"), val = int32(1)]; + tensor h_37_cast_fp16 = conv(dilations = h_37_dilations_0, groups = h_37_groups_0, pad = h_37_pad_0, pad_type = h_37_pad_type_0, strides = h_37_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_197_cast_fp16)[name = string("h_37_cast_fp16")]; + tensor x_145_cast_fp16 = add(x = x_143_cast_fp16, y = h_37_cast_fp16)[name = string("x_145_cast_fp16")]; + tensor key_cache_39_begin_0 = const()[name = string("key_cache_39_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_39_end_0 = const()[name = string("key_cache_39_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_39_end_mask_0 = const()[name = string("key_cache_39_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_39_cast_fp16 = slice_by_index(begin = key_cache_39_begin_0, end = key_cache_39_end_0, end_mask = key_cache_39_end_mask_0, x = layer_key_caches_7_cast_fp16)[name = string("key_cache_39_cast_fp16")]; + tensor value_cache_39_begin_0 = const()[name = string("value_cache_39_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_39_end_0 = const()[name = string("value_cache_39_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_39_end_mask_0 = const()[name = string("value_cache_39_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_39_cast_fp16 = slice_by_index(begin = value_cache_39_begin_0, end = value_cache_39_end_0, end_mask = value_cache_39_end_mask_0, x = layer_value_caches_7_cast_fp16)[name = string("value_cache_39_cast_fp16")]; + int32 var_4699 = const()[name = string("op_4699"), val = int32(2)]; + int32 var_4703 = const()[name = string("op_4703"), val = int32(3)]; + tensor var_4718_cast_fp16 = mul(x = x_145_cast_fp16, y = x_145_cast_fp16)[name = string("op_4718_cast_fp16")]; + tensor variance_157_axes_0 = const()[name = string("variance_157_axes_0"), val = tensor([1])]; + bool variance_157_keep_dims_0 = const()[name = string("variance_157_keep_dims_0"), val = bool(true)]; + tensor variance_157_cast_fp16 = reduce_mean(axes = variance_157_axes_0, keep_dims = variance_157_keep_dims_0, x = var_4718_cast_fp16)[name = string("variance_157_cast_fp16")]; + fp16 var_4721_to_fp16 = const()[name = string("op_4721_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4722_cast_fp16 = add(x = variance_157_cast_fp16, y = var_4721_to_fp16)[name = string("op_4722_cast_fp16")]; + fp32 var_4723_epsilon_0 = const()[name = string("op_4723_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4723_cast_fp16 = rsqrt(epsilon = var_4723_epsilon_0, x = var_4722_cast_fp16)[name = string("op_4723_cast_fp16")]; + tensor var_4724_cast_fp16 = mul(x = x_145_cast_fp16, y = var_4723_cast_fp16)[name = string("op_4724_cast_fp16")]; + tensor input_199_cast_fp16 = mul(x = var_4724_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_199_cast_fp16")]; + string q_115_pad_type_0 = const()[name = string("q_115_pad_type_0"), val = string("valid")]; + tensor q_115_strides_0 = const()[name = string("q_115_strides_0"), val = tensor([1, 1])]; + tensor q_115_pad_0 = const()[name = string("q_115_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_115_dilations_0 = const()[name = string("q_115_dilations_0"), val = tensor([1, 1])]; + int32 q_115_groups_0 = const()[name = string("q_115_groups_0"), val = int32(1)]; + tensor q_115_cast_fp16 = conv(dilations = q_115_dilations_0, groups = q_115_groups_0, pad = q_115_pad_0, pad_type = q_115_pad_type_0, strides = q_115_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_199_cast_fp16)[name = string("q_115_cast_fp16")]; + string k_115_pad_type_0 = const()[name = string("k_115_pad_type_0"), val = string("valid")]; + tensor k_115_strides_0 = const()[name = string("k_115_strides_0"), val = tensor([1, 1])]; + tensor k_115_pad_0 = const()[name = string("k_115_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_115_dilations_0 = const()[name = string("k_115_dilations_0"), val = tensor([1, 1])]; + int32 k_115_groups_0 = const()[name = string("k_115_groups_0"), val = int32(1)]; + tensor k_115_cast_fp16 = conv(dilations = k_115_dilations_0, groups = k_115_groups_0, pad = k_115_pad_0, pad_type = k_115_pad_type_0, strides = k_115_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_199_cast_fp16)[name = string("k_115_cast_fp16")]; + string v_39_pad_type_0 = const()[name = string("v_39_pad_type_0"), val = string("valid")]; + tensor v_39_strides_0 = const()[name = string("v_39_strides_0"), val = tensor([1, 1])]; + tensor v_39_pad_0 = const()[name = string("v_39_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_39_dilations_0 = const()[name = string("v_39_dilations_0"), val = tensor([1, 1])]; + int32 v_39_groups_0 = const()[name = string("v_39_groups_0"), val = int32(1)]; + tensor v_39_cast_fp16 = conv(dilations = v_39_dilations_0, groups = v_39_groups_0, pad = v_39_pad_0, pad_type = v_39_pad_type_0, strides = v_39_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_199_cast_fp16)[name = string("v_39_cast_fp16")]; + tensor var_4758 = const()[name = string("op_4758"), val = tensor([16, 128, 1, 1])]; + tensor x_147_cast_fp16 = reshape(shape = var_4758, x = q_115_cast_fp16)[name = string("x_147_cast_fp16")]; + tensor var_4761_cast_fp16 = mul(x = x_147_cast_fp16, y = x_147_cast_fp16)[name = string("op_4761_cast_fp16")]; + tensor variance_159_axes_0 = const()[name = string("variance_159_axes_0"), val = tensor([1])]; + bool variance_159_keep_dims_0 = const()[name = string("variance_159_keep_dims_0"), val = bool(true)]; + tensor variance_159_cast_fp16 = reduce_mean(axes = variance_159_axes_0, keep_dims = variance_159_keep_dims_0, x = var_4761_cast_fp16)[name = string("variance_159_cast_fp16")]; + fp16 var_4764_to_fp16 = const()[name = string("op_4764_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4765_cast_fp16 = add(x = variance_159_cast_fp16, y = var_4764_to_fp16)[name = string("op_4765_cast_fp16")]; + fp32 var_4766_epsilon_0 = const()[name = string("op_4766_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4766_cast_fp16 = rsqrt(epsilon = var_4766_epsilon_0, x = var_4765_cast_fp16)[name = string("op_4766_cast_fp16")]; + tensor var_4767_cast_fp16 = mul(x = x_147_cast_fp16, y = var_4766_cast_fp16)[name = string("op_4767_cast_fp16")]; + tensor q_117_cast_fp16 = mul(x = var_4767_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_117_cast_fp16")]; + tensor var_4769 = const()[name = string("op_4769"), val = tensor([8, 128, 1, 1])]; + tensor x_149_cast_fp16 = reshape(shape = var_4769, x = k_115_cast_fp16)[name = string("x_149_cast_fp16")]; + tensor var_4772_cast_fp16 = mul(x = x_149_cast_fp16, y = x_149_cast_fp16)[name = string("op_4772_cast_fp16")]; + tensor variance_161_axes_0 = const()[name = string("variance_161_axes_0"), val = tensor([1])]; + bool variance_161_keep_dims_0 = const()[name = string("variance_161_keep_dims_0"), val = bool(true)]; + tensor variance_161_cast_fp16 = reduce_mean(axes = variance_161_axes_0, keep_dims = variance_161_keep_dims_0, x = var_4772_cast_fp16)[name = string("variance_161_cast_fp16")]; + fp16 var_4775_to_fp16 = const()[name = string("op_4775_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4776_cast_fp16 = add(x = variance_161_cast_fp16, y = var_4775_to_fp16)[name = string("op_4776_cast_fp16")]; + fp32 var_4777_epsilon_0 = const()[name = string("op_4777_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4777_cast_fp16 = rsqrt(epsilon = var_4777_epsilon_0, x = var_4776_cast_fp16)[name = string("op_4777_cast_fp16")]; + tensor var_4778_cast_fp16 = mul(x = x_149_cast_fp16, y = var_4777_cast_fp16)[name = string("op_4778_cast_fp16")]; + tensor k_117_cast_fp16 = mul(x = var_4778_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_117_cast_fp16")]; + tensor var_4780 = const()[name = string("op_4780"), val = tensor([1, 16, 128, 1])]; + tensor z_77_cast_fp16 = reshape(shape = var_4780, x = q_117_cast_fp16)[name = string("z_77_cast_fp16")]; + tensor var_4782 = const()[name = string("op_4782"), val = tensor([1, 8, 128, 1])]; + tensor z_79_cast_fp16 = reshape(shape = var_4782, x = k_117_cast_fp16)[name = string("z_79_cast_fp16")]; + tensor z1_77_begin_0 = const()[name = string("z1_77_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_77_end_0 = const()[name = string("z1_77_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_77_end_mask_0 = const()[name = string("z1_77_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_77_cast_fp16 = slice_by_index(begin = z1_77_begin_0, end = z1_77_end_0, end_mask = z1_77_end_mask_0, x = z_77_cast_fp16)[name = string("z1_77_cast_fp16")]; + tensor z2_77_begin_0 = const()[name = string("z2_77_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_77_end_0 = const()[name = string("z2_77_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_77_end_mask_0 = const()[name = string("z2_77_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_77_cast_fp16 = slice_by_index(begin = z2_77_begin_0, end = z2_77_end_0, end_mask = z2_77_end_mask_0, x = z_77_cast_fp16)[name = string("z2_77_cast_fp16")]; + tensor var_4790_cast_fp16 = mul(x = z_77_cast_fp16, y = cos_31_to_fp16)[name = string("op_4790_cast_fp16")]; + fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4791_cast_fp16 = mul(x = z2_77_cast_fp16, y = const_42_promoted_to_fp16)[name = string("op_4791_cast_fp16")]; + bool var_4793_interleave_0 = const()[name = string("op_4793_interleave_0"), val = bool(false)]; + tensor var_4793_cast_fp16 = concat(axis = var_4699, interleave = var_4793_interleave_0, values = (var_4791_cast_fp16, z1_77_cast_fp16))[name = string("op_4793_cast_fp16")]; + tensor var_4794_cast_fp16 = mul(x = var_4793_cast_fp16, y = sin_31_to_fp16)[name = string("op_4794_cast_fp16")]; + tensor q_119_cast_fp16 = add(x = var_4790_cast_fp16, y = var_4794_cast_fp16)[name = string("q_119_cast_fp16")]; + tensor z1_79_begin_0 = const()[name = string("z1_79_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_79_end_0 = const()[name = string("z1_79_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_79_end_mask_0 = const()[name = string("z1_79_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_79_cast_fp16 = slice_by_index(begin = z1_79_begin_0, end = z1_79_end_0, end_mask = z1_79_end_mask_0, x = z_79_cast_fp16)[name = string("z1_79_cast_fp16")]; + tensor z2_79_begin_0 = const()[name = string("z2_79_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_79_end_0 = const()[name = string("z2_79_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_79_end_mask_0 = const()[name = string("z2_79_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_79_cast_fp16 = slice_by_index(begin = z2_79_begin_0, end = z2_79_end_0, end_mask = z2_79_end_mask_0, x = z_79_cast_fp16)[name = string("z2_79_cast_fp16")]; + tensor var_4802_cast_fp16 = mul(x = z_79_cast_fp16, y = cos_31_to_fp16)[name = string("op_4802_cast_fp16")]; + fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4803_cast_fp16 = mul(x = z2_79_cast_fp16, y = const_43_promoted_to_fp16)[name = string("op_4803_cast_fp16")]; + bool var_4805_interleave_0 = const()[name = string("op_4805_interleave_0"), val = bool(false)]; + tensor var_4805_cast_fp16 = concat(axis = var_4699, interleave = var_4805_interleave_0, values = (var_4803_cast_fp16, z1_79_cast_fp16))[name = string("op_4805_cast_fp16")]; + tensor var_4806_cast_fp16 = mul(x = var_4805_cast_fp16, y = sin_31_to_fp16)[name = string("op_4806_cast_fp16")]; + tensor k_119_cast_fp16 = add(x = var_4802_cast_fp16, y = var_4806_cast_fp16)[name = string("k_119_cast_fp16")]; + tensor var_4808 = const()[name = string("op_4808"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_39_cast_fp16 = reshape(shape = var_4808, x = k_119_cast_fp16)[name = string("cur_key_39_cast_fp16")]; + tensor var_4810_to_fp16 = const()[name = string("op_4810_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635072)))]; + tensor var_4811_cast_fp16 = mul(x = key_cache_39_cast_fp16, y = var_4810_to_fp16)[name = string("op_4811_cast_fp16")]; + tensor upd_39_to_fp16 = const()[name = string("upd_39_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635200)))]; + tensor var_4812_cast_fp16 = mul(x = cur_key_39_cast_fp16, y = upd_39_to_fp16)[name = string("op_4812_cast_fp16")]; + tensor key_39_cast_fp16 = add(x = var_4811_cast_fp16, y = var_4812_cast_fp16)[name = string("key_39_cast_fp16")]; + tensor var_4814_to_fp16 = const()[name = string("op_4814_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635072)))]; + tensor var_4815_cast_fp16 = mul(x = value_cache_39_cast_fp16, y = var_4814_to_fp16)[name = string("op_4815_cast_fp16")]; + tensor var_4816_cast_fp16 = mul(x = v_39_cast_fp16, y = upd_39_to_fp16)[name = string("op_4816_cast_fp16")]; + tensor value_39_cast_fp16 = add(x = var_4815_cast_fp16, y = var_4816_cast_fp16)[name = string("value_39_cast_fp16")]; + tensor var_4818 = const()[name = string("op_4818"), val = tensor([1, 8, 128, 16])]; + tensor kh_77_cast_fp16 = reshape(shape = var_4818, x = key_39_cast_fp16)[name = string("kh_77_cast_fp16")]; + tensor var_4820 = const()[name = string("op_4820"), val = tensor([1, 8, 128, 16])]; + tensor vh_77_cast_fp16 = reshape(shape = var_4820, x = value_39_cast_fp16)[name = string("vh_77_cast_fp16")]; + tensor transpose_76_perm_0 = const()[name = string("transpose_76_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_38_reps_0 = const()[name = string("tile_38_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_76_cast_fp16 = transpose(perm = transpose_76_perm_0, x = kh_77_cast_fp16)[name = string("transpose_365")]; + tensor tile_38_cast_fp16 = tile(reps = tile_38_reps_0, x = transpose_76_cast_fp16)[name = string("tile_38_cast_fp16")]; + tensor concat_94 = const()[name = string("concat_94"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_76_cast_fp16 = reshape(shape = concat_94, x = tile_38_cast_fp16)[name = string("reshape_76_cast_fp16")]; + tensor transpose_77_perm_0 = const()[name = string("transpose_77_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_95 = const()[name = string("concat_95"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_77_cast_fp16 = transpose(perm = transpose_77_perm_0, x = reshape_76_cast_fp16)[name = string("transpose_364")]; + tensor reshape_77_cast_fp16 = reshape(shape = concat_95, x = transpose_77_cast_fp16)[name = string("reshape_77_cast_fp16")]; + tensor transpose_78_perm_0 = const()[name = string("transpose_78_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_39_reps_0 = const()[name = string("tile_39_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_78_cast_fp16 = transpose(perm = transpose_78_perm_0, x = vh_77_cast_fp16)[name = string("transpose_363")]; + tensor tile_39_cast_fp16 = tile(reps = tile_39_reps_0, x = transpose_78_cast_fp16)[name = string("tile_39_cast_fp16")]; + tensor concat_96 = const()[name = string("concat_96"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_78_cast_fp16 = reshape(shape = concat_96, x = tile_39_cast_fp16)[name = string("reshape_78_cast_fp16")]; + tensor transpose_79_perm_0 = const()[name = string("transpose_79_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_97 = const()[name = string("concat_97"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_79_cast_fp16 = transpose(perm = transpose_79_perm_0, x = reshape_78_cast_fp16)[name = string("transpose_362")]; + tensor reshape_79_cast_fp16 = reshape(shape = concat_97, x = transpose_79_cast_fp16)[name = string("reshape_79_cast_fp16")]; + fp16 var_4824_to_fp16 = const()[name = string("op_4824_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_4825_cast_fp16 = mul(x = q_119_cast_fp16, y = var_4824_to_fp16)[name = string("op_4825_cast_fp16")]; + tensor transpose_393_perm_0 = const()[name = string("transpose_393_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_81_transpose_x_1 = const()[name = string("w_81_transpose_x_1"), val = bool(true)]; + bool w_81_transpose_y_1 = const()[name = string("w_81_transpose_y_1"), val = bool(false)]; + tensor transpose_393_cast_fp16 = transpose(perm = transpose_393_perm_0, x = reshape_77_cast_fp16)[name = string("transpose_361")]; + tensor w_81_cast_fp16 = matmul(transpose_x = w_81_transpose_x_1, transpose_y = w_81_transpose_y_1, x = var_4825_cast_fp16, y = transpose_393_cast_fp16)[name = string("w_81_cast_fp16")]; + tensor pad_39_to_fp16 = const()[name = string("pad_39_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635328)))]; + tensor var_4828_cast_fp16 = add(x = w_81_cast_fp16, y = pad_39_to_fp16)[name = string("op_4828_cast_fp16")]; + tensor w_83_cast_fp16 = softmax(axis = var_4703, x = var_4828_cast_fp16)[name = string("w_83_cast_fp16")]; + tensor transpose_394_perm_0 = const()[name = string("transpose_394_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_39_transpose_x_1 = const()[name = string("attn_39_transpose_x_1"), val = bool(false)]; + bool attn_39_transpose_y_1 = const()[name = string("attn_39_transpose_y_1"), val = bool(true)]; + tensor transpose_394_cast_fp16 = transpose(perm = transpose_394_perm_0, x = reshape_79_cast_fp16)[name = string("transpose_360")]; + tensor attn_39_cast_fp16 = matmul(transpose_x = attn_39_transpose_x_1, transpose_y = attn_39_transpose_y_1, x = transpose_394_cast_fp16, y = w_83_cast_fp16)[name = string("attn_39_cast_fp16")]; + tensor var_4832 = const()[name = string("op_4832"), val = tensor([1, 2048, 1, 1])]; + tensor input_201_cast_fp16 = reshape(shape = var_4832, x = attn_39_cast_fp16)[name = string("input_201_cast_fp16")]; + string attn_output_39_pad_type_0 = const()[name = string("attn_output_39_pad_type_0"), val = string("valid")]; + tensor attn_output_39_strides_0 = const()[name = string("attn_output_39_strides_0"), val = tensor([1, 1])]; + tensor attn_output_39_pad_0 = const()[name = string("attn_output_39_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_39_dilations_0 = const()[name = string("attn_output_39_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_39_groups_0 = const()[name = string("attn_output_39_groups_0"), val = int32(1)]; + tensor attn_output_39_cast_fp16 = conv(dilations = attn_output_39_dilations_0, groups = attn_output_39_groups_0, pad = attn_output_39_pad_0, pad_type = attn_output_39_pad_type_0, strides = attn_output_39_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_201_cast_fp16)[name = string("attn_output_39_cast_fp16")]; + tensor x_151_cast_fp16 = add(x = x_145_cast_fp16, y = attn_output_39_cast_fp16)[name = string("x_151_cast_fp16")]; + tensor var_4846_cast_fp16 = mul(x = x_151_cast_fp16, y = x_151_cast_fp16)[name = string("op_4846_cast_fp16")]; + tensor variance_163_axes_0 = const()[name = string("variance_163_axes_0"), val = tensor([1])]; + bool variance_163_keep_dims_0 = const()[name = string("variance_163_keep_dims_0"), val = bool(true)]; + tensor variance_163_cast_fp16 = reduce_mean(axes = variance_163_axes_0, keep_dims = variance_163_keep_dims_0, x = var_4846_cast_fp16)[name = string("variance_163_cast_fp16")]; + fp16 var_4849_to_fp16 = const()[name = string("op_4849_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4850_cast_fp16 = add(x = variance_163_cast_fp16, y = var_4849_to_fp16)[name = string("op_4850_cast_fp16")]; + fp32 var_4851_epsilon_0 = const()[name = string("op_4851_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4851_cast_fp16 = rsqrt(epsilon = var_4851_epsilon_0, x = var_4850_cast_fp16)[name = string("op_4851_cast_fp16")]; + tensor var_4852_cast_fp16 = mul(x = x_151_cast_fp16, y = var_4851_cast_fp16)[name = string("op_4852_cast_fp16")]; + tensor input_203_cast_fp16 = mul(x = var_4852_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_203_cast_fp16")]; + string input_205_pad_type_0 = const()[name = string("input_205_pad_type_0"), val = string("valid")]; + tensor input_205_strides_0 = const()[name = string("input_205_strides_0"), val = tensor([1, 1])]; + tensor input_205_pad_0 = const()[name = string("input_205_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_205_dilations_0 = const()[name = string("input_205_dilations_0"), val = tensor([1, 1])]; + int32 input_205_groups_0 = const()[name = string("input_205_groups_0"), val = int32(1)]; + tensor input_205_cast_fp16 = conv(dilations = input_205_dilations_0, groups = input_205_groups_0, pad = input_205_pad_0, pad_type = input_205_pad_type_0, strides = input_205_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_203_cast_fp16)[name = string("input_205_cast_fp16")]; + tensor var_4860_cast_fp16 = silu(x = input_205_cast_fp16)[name = string("op_4860_cast_fp16")]; + string var_4866_pad_type_0 = const()[name = string("op_4866_pad_type_0"), val = string("valid")]; + tensor var_4866_strides_0 = const()[name = string("op_4866_strides_0"), val = tensor([1, 1])]; + tensor var_4866_pad_0 = const()[name = string("op_4866_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4866_dilations_0 = const()[name = string("op_4866_dilations_0"), val = tensor([1, 1])]; + int32 var_4866_groups_0 = const()[name = string("op_4866_groups_0"), val = int32(1)]; + tensor var_4866_cast_fp16 = conv(dilations = var_4866_dilations_0, groups = var_4866_groups_0, pad = var_4866_pad_0, pad_type = var_4866_pad_type_0, strides = var_4866_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_203_cast_fp16)[name = string("op_4866_cast_fp16")]; + tensor input_207_cast_fp16 = mul(x = var_4860_cast_fp16, y = var_4866_cast_fp16)[name = string("input_207_cast_fp16")]; + string h_39_pad_type_0 = const()[name = string("h_39_pad_type_0"), val = string("valid")]; + tensor h_39_strides_0 = const()[name = string("h_39_strides_0"), val = tensor([1, 1])]; + tensor h_39_pad_0 = const()[name = string("h_39_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_39_dilations_0 = const()[name = string("h_39_dilations_0"), val = tensor([1, 1])]; + int32 h_39_groups_0 = const()[name = string("h_39_groups_0"), val = int32(1)]; + tensor h_39_cast_fp16 = conv(dilations = h_39_dilations_0, groups = h_39_groups_0, pad = h_39_pad_0, pad_type = h_39_pad_type_0, strides = h_39_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_207_cast_fp16)[name = string("h_39_cast_fp16")]; + tensor inputs_5_cast_fp16 = add(x = x_151_cast_fp16, y = h_39_cast_fp16)[name = string("inputs_5_cast_fp16")]; + int32 var_4894 = const()[name = string("op_4894"), val = int32(1)]; + bool layer_key_caches_9_interleave_0 = const()[name = string("layer_key_caches_9_interleave_0"), val = bool(false)]; + tensor layer_key_caches_9_cast_fp16 = concat(axis = var_4894, interleave = layer_key_caches_9_interleave_0, values = (key_31_cast_fp16, key_33_cast_fp16, key_35_cast_fp16, key_37_cast_fp16, key_39_cast_fp16))[name = string("layer_key_caches_9_cast_fp16")]; + int32 var_4897 = const()[name = string("op_4897"), val = int32(1)]; + bool layer_value_caches_9_interleave_0 = const()[name = string("layer_value_caches_9_interleave_0"), val = bool(false)]; + tensor layer_value_caches_9_cast_fp16 = concat(axis = var_4897, interleave = layer_value_caches_9_interleave_0, values = (value_31_cast_fp16, value_33_cast_fp16, value_35_cast_fp16, value_37_cast_fp16, value_39_cast_fp16))[name = string("layer_value_caches_9_cast_fp16")]; + tensor inputs_sq_5_cast_fp16 = mul(x = inputs_5_cast_fp16, y = inputs_5_cast_fp16)[name = string("inputs_sq_5_cast_fp16")]; + tensor variance_165_axes_0 = const()[name = string("variance_165_axes_0"), val = tensor([1])]; + bool variance_165_keep_dims_0 = const()[name = string("variance_165_keep_dims_0"), val = bool(true)]; + tensor variance_165_cast_fp16 = reduce_mean(axes = variance_165_axes_0, keep_dims = variance_165_keep_dims_0, x = inputs_sq_5_cast_fp16)[name = string("variance_165_cast_fp16")]; + fp16 var_4907_to_fp16 = const()[name = string("op_4907_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4908_cast_fp16 = add(x = variance_165_cast_fp16, y = var_4907_to_fp16)[name = string("op_4908_cast_fp16")]; + fp32 var_4909_epsilon_0 = const()[name = string("op_4909_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4909_cast_fp16 = rsqrt(epsilon = var_4909_epsilon_0, x = var_4908_cast_fp16)[name = string("op_4909_cast_fp16")]; + tensor hidden_states_5_cast_fp16 = mul(x = inputs_5_cast_fp16, y = var_4909_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; + tensor input_209_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_5_cast_fp16)[name = string("input_209_cast_fp16")]; + string logits_9_pad_type_0 = const()[name = string("logits_9_pad_type_0"), val = string("valid")]; + tensor logits_9_strides_0 = const()[name = string("logits_9_strides_0"), val = tensor([1, 1])]; + tensor logits_9_pad_0 = const()[name = string("logits_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_9_dilations_0 = const()[name = string("logits_9_dilations_0"), val = tensor([1, 1])]; + int32 logits_9_groups_0 = const()[name = string("logits_9_groups_0"), val = int32(1)]; + tensor lm_heads_2_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82902272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84999488))))[name = string("lm_heads_2_weight_to_fp16_palettized")]; + tensor logits_9_cast_fp16 = conv(dilations = logits_9_dilations_0, groups = logits_9_groups_0, pad = logits_9_pad_0, pad_type = logits_9_pad_type_0, strides = logits_9_strides_0, weight = lm_heads_2_weight_to_fp16_palettized, x = input_209_cast_fp16)[name = string("logits_9_cast_fp16")]; + tensor var_4927 = const()[name = string("op_4927"), val = tensor([1, 2048])]; + tensor logits_11_cast_fp16 = reshape(shape = var_4927, x = logits_9_cast_fp16)[name = string("logits_11_cast_fp16")]; + tensor scaled_logits_5_cast_fp16 = real_div(x = logits_11_cast_fp16, y = temperature)[name = string("scaled_logits_5_cast_fp16")]; + int32 var_4937 = const()[name = string("op_4937"), val = int32(100)]; + int32 top_values_5_axis_0 = const()[name = string("top_values_5_axis_0"), val = int32(1)]; + bool top_values_5_ascending_0 = const()[name = string("top_values_5_ascending_0"), val = bool(false)]; + bool top_values_5_sort_0 = const()[name = string("top_values_5_sort_0"), val = bool(true)]; + bool top_values_5_return_indices_0 = const()[name = string("top_values_5_return_indices_0"), val = bool(true)]; + string top_values_5_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_5_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_5_cast_fp16_cast_uint16_0, tensor top_values_5_cast_fp16_cast_uint16_1 = topk(ascending = top_values_5_ascending_0, axis = top_values_5_axis_0, k = var_4937, output_indices_dtype = top_values_5_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_5_return_indices_0, sort = top_values_5_sort_0, x = scaled_logits_5_cast_fp16)[name = string("top_values_5_cast_fp16_cast_uint16")]; + tensor var_4943_cast_fp16 = mul(x = top_values_5_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_4943_cast_fp16")]; + tensor var_4947_cast_fp16 = add(x = var_4943_cast_fp16, y = var_2438_cast_fp16)[name = string("op_4947_cast_fp16")]; + tensor reduce_min_2_axes_0 = const()[name = string("reduce_min_2_axes_0"), val = tensor([1])]; + bool reduce_min_2_keep_dims_0 = const()[name = string("reduce_min_2_keep_dims_0"), val = bool(true)]; + tensor reduce_min_2_cast_fp16 = reduce_min(axes = reduce_min_2_axes_0, keep_dims = reduce_min_2_keep_dims_0, x = var_4947_cast_fp16)[name = string("reduce_min_2_cast_fp16")]; + tensor var_4950_cast_fp16 = greater_equal(x = scaled_logits_5_cast_fp16, y = reduce_min_2_cast_fp16)[name = string("op_4950_cast_fp16")]; + fp16 var_4951_value_0_to_fp16 = const()[name = string("op_4951_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_4951_cast_fp16 = fill_like(ref_tensor = scaled_logits_5_cast_fp16, value = var_4951_value_0_to_fp16)[name = string("op_4951_cast_fp16")]; + tensor masked_logits_5_cast_fp16 = select(a = scaled_logits_5_cast_fp16, b = var_4951_cast_fp16, cond = var_4950_cast_fp16)[name = string("masked_logits_5_cast_fp16")]; + tensor var_4955_begin_0 = const()[name = string("op_4955_begin_0"), val = tensor([2, 0])]; + tensor var_4955_end_0 = const()[name = string("op_4955_end_0"), val = tensor([3, 2048])]; + tensor var_4955_end_mask_0 = const()[name = string("op_4955_end_mask_0"), val = tensor([false, true])]; + tensor var_4955_squeeze_mask_0 = const()[name = string("op_4955_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_4955_cast_fp16 = slice_by_index(begin = var_4955_begin_0, end = var_4955_end_0, end_mask = var_4955_end_mask_0, squeeze_mask = var_4955_squeeze_mask_0, x = gumbel)[name = string("op_4955_cast_fp16")]; + tensor var_4958 = const()[name = string("op_4958"), val = tensor([1, 2048])]; + tensor var_4959_cast_fp16 = reshape(shape = var_4958, x = var_4955_cast_fp16)[name = string("op_4959_cast_fp16")]; + tensor noisy_logits_5_cast_fp16 = add(x = masked_logits_5_cast_fp16, y = var_4959_cast_fp16)[name = string("noisy_logits_5_cast_fp16")]; + int32 code_5_axis_0 = const()[name = string("code_5_axis_0"), val = int32(1)]; + bool code_5_keep_dims_0 = const()[name = string("code_5_keep_dims_0"), val = bool(false)]; + string code_5_output_dtype_0 = const()[name = string("code_5_output_dtype_0"), val = string("int32")]; + tensor code_5_cast_fp16 = reduce_argmax(axis = code_5_axis_0, keep_dims = code_5_keep_dims_0, output_dtype = code_5_output_dtype_0, x = noisy_logits_5_cast_fp16)[name = string("code_5_cast_fp16")]; + int32 var_4970 = const()[name = string("op_4970"), val = int32(4096)]; + tensor input_211 = add(x = code_5_cast_fp16, y = var_4970)[name = string("input_211")]; + int32 code_embed_9_axis_0 = const()[name = string("code_embed_9_axis_0"), val = int32(0)]; + int32 code_embed_9_batch_dims_0 = const()[name = string("code_embed_9_batch_dims_0"), val = int32(0)]; + bool code_embed_9_validate_indices_0 = const()[name = string("code_embed_9_validate_indices_0"), val = bool(false)]; + string input_211_to_uint16_dtype_0 = const()[name = string("input_211_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_211_to_uint16 = cast(dtype = input_211_to_uint16_dtype_0, x = input_211)[name = string("cast_12")]; + tensor code_embed_9_cast_fp16_cast_uint16 = gather(axis = code_embed_9_axis_0, batch_dims = code_embed_9_batch_dims_0, indices = input_211_to_uint16, validate_indices = code_embed_9_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_9_cast_fp16_cast_uint16")]; + tensor var_4974 = const()[name = string("op_4974"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_11_cast_fp16 = reshape(shape = var_4974, x = code_embed_9_cast_fp16_cast_uint16)[name = string("code_embed_11_cast_fp16")]; + tensor embed_sum_7_cast_fp16 = add(x = embed_sum_5_cast_fp16, y = code_embed_11_cast_fp16)[name = string("embed_sum_7_cast_fp16")]; + tensor key_cache_41_begin_0 = const()[name = string("key_cache_41_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_41_end_0 = const()[name = string("key_cache_41_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_41_end_mask_0 = const()[name = string("key_cache_41_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_41_cast_fp16 = slice_by_index(begin = key_cache_41_begin_0, end = key_cache_41_end_0, end_mask = key_cache_41_end_mask_0, x = layer_key_caches_9_cast_fp16)[name = string("key_cache_41_cast_fp16")]; + tensor value_cache_41_begin_0 = const()[name = string("value_cache_41_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_41_end_0 = const()[name = string("value_cache_41_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_41_end_mask_0 = const()[name = string("value_cache_41_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_41_cast_fp16 = slice_by_index(begin = value_cache_41_begin_0, end = value_cache_41_end_0, end_mask = value_cache_41_end_mask_0, x = layer_value_caches_9_cast_fp16)[name = string("value_cache_41_cast_fp16")]; + int32 var_5073 = const()[name = string("op_5073"), val = int32(2)]; + int32 var_5077 = const()[name = string("op_5077"), val = int32(3)]; + tensor var_5092_cast_fp16 = mul(x = code_embed_11_cast_fp16, y = code_embed_11_cast_fp16)[name = string("op_5092_cast_fp16")]; + tensor variance_167_axes_0 = const()[name = string("variance_167_axes_0"), val = tensor([1])]; + bool variance_167_keep_dims_0 = const()[name = string("variance_167_keep_dims_0"), val = bool(true)]; + tensor variance_167_cast_fp16 = reduce_mean(axes = variance_167_axes_0, keep_dims = variance_167_keep_dims_0, x = var_5092_cast_fp16)[name = string("variance_167_cast_fp16")]; + fp16 var_5095_to_fp16 = const()[name = string("op_5095_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5096_cast_fp16 = add(x = variance_167_cast_fp16, y = var_5095_to_fp16)[name = string("op_5096_cast_fp16")]; + fp32 var_5097_epsilon_0 = const()[name = string("op_5097_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5097_cast_fp16 = rsqrt(epsilon = var_5097_epsilon_0, x = var_5096_cast_fp16)[name = string("op_5097_cast_fp16")]; + tensor var_5098_cast_fp16 = mul(x = code_embed_11_cast_fp16, y = var_5097_cast_fp16)[name = string("op_5098_cast_fp16")]; + tensor input_213_cast_fp16 = mul(x = var_5098_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_213_cast_fp16")]; + string q_121_pad_type_0 = const()[name = string("q_121_pad_type_0"), val = string("valid")]; + tensor q_121_strides_0 = const()[name = string("q_121_strides_0"), val = tensor([1, 1])]; + tensor q_121_pad_0 = const()[name = string("q_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_121_dilations_0 = const()[name = string("q_121_dilations_0"), val = tensor([1, 1])]; + int32 q_121_groups_0 = const()[name = string("q_121_groups_0"), val = int32(1)]; + tensor q_121_cast_fp16 = conv(dilations = q_121_dilations_0, groups = q_121_groups_0, pad = q_121_pad_0, pad_type = q_121_pad_type_0, strides = q_121_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_213_cast_fp16)[name = string("q_121_cast_fp16")]; + string k_121_pad_type_0 = const()[name = string("k_121_pad_type_0"), val = string("valid")]; + tensor k_121_strides_0 = const()[name = string("k_121_strides_0"), val = tensor([1, 1])]; + tensor k_121_pad_0 = const()[name = string("k_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_121_dilations_0 = const()[name = string("k_121_dilations_0"), val = tensor([1, 1])]; + int32 k_121_groups_0 = const()[name = string("k_121_groups_0"), val = int32(1)]; + tensor k_121_cast_fp16 = conv(dilations = k_121_dilations_0, groups = k_121_groups_0, pad = k_121_pad_0, pad_type = k_121_pad_type_0, strides = k_121_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_213_cast_fp16)[name = string("k_121_cast_fp16")]; + string v_41_pad_type_0 = const()[name = string("v_41_pad_type_0"), val = string("valid")]; + tensor v_41_strides_0 = const()[name = string("v_41_strides_0"), val = tensor([1, 1])]; + tensor v_41_pad_0 = const()[name = string("v_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_41_dilations_0 = const()[name = string("v_41_dilations_0"), val = tensor([1, 1])]; + int32 v_41_groups_0 = const()[name = string("v_41_groups_0"), val = int32(1)]; + tensor v_41_cast_fp16 = conv(dilations = v_41_dilations_0, groups = v_41_groups_0, pad = v_41_pad_0, pad_type = v_41_pad_type_0, strides = v_41_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_213_cast_fp16)[name = string("v_41_cast_fp16")]; + tensor var_5132 = const()[name = string("op_5132"), val = tensor([16, 128, 1, 1])]; + tensor x_153_cast_fp16 = reshape(shape = var_5132, x = q_121_cast_fp16)[name = string("x_153_cast_fp16")]; + tensor var_5135_cast_fp16 = mul(x = x_153_cast_fp16, y = x_153_cast_fp16)[name = string("op_5135_cast_fp16")]; + tensor variance_169_axes_0 = const()[name = string("variance_169_axes_0"), val = tensor([1])]; + bool variance_169_keep_dims_0 = const()[name = string("variance_169_keep_dims_0"), val = bool(true)]; + tensor variance_169_cast_fp16 = reduce_mean(axes = variance_169_axes_0, keep_dims = variance_169_keep_dims_0, x = var_5135_cast_fp16)[name = string("variance_169_cast_fp16")]; + fp16 var_5138_to_fp16 = const()[name = string("op_5138_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5139_cast_fp16 = add(x = variance_169_cast_fp16, y = var_5138_to_fp16)[name = string("op_5139_cast_fp16")]; + fp32 var_5140_epsilon_0 = const()[name = string("op_5140_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5140_cast_fp16 = rsqrt(epsilon = var_5140_epsilon_0, x = var_5139_cast_fp16)[name = string("op_5140_cast_fp16")]; + tensor var_5141_cast_fp16 = mul(x = x_153_cast_fp16, y = var_5140_cast_fp16)[name = string("op_5141_cast_fp16")]; + tensor q_123_cast_fp16 = mul(x = var_5141_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_123_cast_fp16")]; + tensor var_5143 = const()[name = string("op_5143"), val = tensor([8, 128, 1, 1])]; + tensor x_155_cast_fp16 = reshape(shape = var_5143, x = k_121_cast_fp16)[name = string("x_155_cast_fp16")]; + tensor var_5146_cast_fp16 = mul(x = x_155_cast_fp16, y = x_155_cast_fp16)[name = string("op_5146_cast_fp16")]; + tensor variance_171_axes_0 = const()[name = string("variance_171_axes_0"), val = tensor([1])]; + bool variance_171_keep_dims_0 = const()[name = string("variance_171_keep_dims_0"), val = bool(true)]; + tensor variance_171_cast_fp16 = reduce_mean(axes = variance_171_axes_0, keep_dims = variance_171_keep_dims_0, x = var_5146_cast_fp16)[name = string("variance_171_cast_fp16")]; + fp16 var_5149_to_fp16 = const()[name = string("op_5149_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5150_cast_fp16 = add(x = variance_171_cast_fp16, y = var_5149_to_fp16)[name = string("op_5150_cast_fp16")]; + fp32 var_5151_epsilon_0 = const()[name = string("op_5151_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5151_cast_fp16 = rsqrt(epsilon = var_5151_epsilon_0, x = var_5150_cast_fp16)[name = string("op_5151_cast_fp16")]; + tensor var_5152_cast_fp16 = mul(x = x_155_cast_fp16, y = var_5151_cast_fp16)[name = string("op_5152_cast_fp16")]; + tensor k_123_cast_fp16 = mul(x = var_5152_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_123_cast_fp16")]; + tensor var_5154 = const()[name = string("op_5154"), val = tensor([1, 16, 128, 1])]; + tensor z_81_cast_fp16 = reshape(shape = var_5154, x = q_123_cast_fp16)[name = string("z_81_cast_fp16")]; + tensor var_5156 = const()[name = string("op_5156"), val = tensor([1, 8, 128, 1])]; + tensor z_83_cast_fp16 = reshape(shape = var_5156, x = k_123_cast_fp16)[name = string("z_83_cast_fp16")]; + tensor z1_81_begin_0 = const()[name = string("z1_81_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_81_end_0 = const()[name = string("z1_81_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_81_end_mask_0 = const()[name = string("z1_81_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_81_cast_fp16 = slice_by_index(begin = z1_81_begin_0, end = z1_81_end_0, end_mask = z1_81_end_mask_0, x = z_81_cast_fp16)[name = string("z1_81_cast_fp16")]; + tensor z2_81_begin_0 = const()[name = string("z2_81_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_81_end_0 = const()[name = string("z2_81_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_81_end_mask_0 = const()[name = string("z2_81_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_81_cast_fp16 = slice_by_index(begin = z2_81_begin_0, end = z2_81_end_0, end_mask = z2_81_end_mask_0, x = z_81_cast_fp16)[name = string("z2_81_cast_fp16")]; + tensor cos_41_to_fp16 = const()[name = string("cos_41_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635456)))]; + tensor var_5164_cast_fp16 = mul(x = z_81_cast_fp16, y = cos_41_to_fp16)[name = string("op_5164_cast_fp16")]; + fp16 const_45_promoted_to_fp16 = const()[name = string("const_45_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5165_cast_fp16 = mul(x = z2_81_cast_fp16, y = const_45_promoted_to_fp16)[name = string("op_5165_cast_fp16")]; + bool var_5167_interleave_0 = const()[name = string("op_5167_interleave_0"), val = bool(false)]; + tensor var_5167_cast_fp16 = concat(axis = var_5073, interleave = var_5167_interleave_0, values = (var_5165_cast_fp16, z1_81_cast_fp16))[name = string("op_5167_cast_fp16")]; + tensor sin_41_to_fp16 = const()[name = string("sin_41_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141635776)))]; + tensor var_5168_cast_fp16 = mul(x = var_5167_cast_fp16, y = sin_41_to_fp16)[name = string("op_5168_cast_fp16")]; + tensor q_125_cast_fp16 = add(x = var_5164_cast_fp16, y = var_5168_cast_fp16)[name = string("q_125_cast_fp16")]; + tensor z1_83_begin_0 = const()[name = string("z1_83_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_83_end_0 = const()[name = string("z1_83_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_83_end_mask_0 = const()[name = string("z1_83_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_83_cast_fp16 = slice_by_index(begin = z1_83_begin_0, end = z1_83_end_0, end_mask = z1_83_end_mask_0, x = z_83_cast_fp16)[name = string("z1_83_cast_fp16")]; + tensor z2_83_begin_0 = const()[name = string("z2_83_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_83_end_0 = const()[name = string("z2_83_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_83_end_mask_0 = const()[name = string("z2_83_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_83_cast_fp16 = slice_by_index(begin = z2_83_begin_0, end = z2_83_end_0, end_mask = z2_83_end_mask_0, x = z_83_cast_fp16)[name = string("z2_83_cast_fp16")]; + tensor var_5176_cast_fp16 = mul(x = z_83_cast_fp16, y = cos_41_to_fp16)[name = string("op_5176_cast_fp16")]; + fp16 const_46_promoted_to_fp16 = const()[name = string("const_46_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5177_cast_fp16 = mul(x = z2_83_cast_fp16, y = const_46_promoted_to_fp16)[name = string("op_5177_cast_fp16")]; + bool var_5179_interleave_0 = const()[name = string("op_5179_interleave_0"), val = bool(false)]; + tensor var_5179_cast_fp16 = concat(axis = var_5073, interleave = var_5179_interleave_0, values = (var_5177_cast_fp16, z1_83_cast_fp16))[name = string("op_5179_cast_fp16")]; + tensor var_5180_cast_fp16 = mul(x = var_5179_cast_fp16, y = sin_41_to_fp16)[name = string("op_5180_cast_fp16")]; + tensor k_125_cast_fp16 = add(x = var_5176_cast_fp16, y = var_5180_cast_fp16)[name = string("k_125_cast_fp16")]; + tensor var_5182 = const()[name = string("op_5182"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_41_cast_fp16 = reshape(shape = var_5182, x = k_125_cast_fp16)[name = string("cur_key_41_cast_fp16")]; + tensor var_5184_to_fp16 = const()[name = string("op_5184_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636096)))]; + tensor var_5185_cast_fp16 = mul(x = key_cache_41_cast_fp16, y = var_5184_to_fp16)[name = string("op_5185_cast_fp16")]; + tensor upd_41_to_fp16 = const()[name = string("upd_41_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636224)))]; + tensor var_5186_cast_fp16 = mul(x = cur_key_41_cast_fp16, y = upd_41_to_fp16)[name = string("op_5186_cast_fp16")]; + tensor key_41_cast_fp16 = add(x = var_5185_cast_fp16, y = var_5186_cast_fp16)[name = string("key_41_cast_fp16")]; + tensor var_5188_to_fp16 = const()[name = string("op_5188_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636096)))]; + tensor var_5189_cast_fp16 = mul(x = value_cache_41_cast_fp16, y = var_5188_to_fp16)[name = string("op_5189_cast_fp16")]; + tensor var_5190_cast_fp16 = mul(x = v_41_cast_fp16, y = upd_41_to_fp16)[name = string("op_5190_cast_fp16")]; + tensor value_41_cast_fp16 = add(x = var_5189_cast_fp16, y = var_5190_cast_fp16)[name = string("value_41_cast_fp16")]; + tensor var_5192 = const()[name = string("op_5192"), val = tensor([1, 8, 128, 16])]; + tensor kh_81_cast_fp16 = reshape(shape = var_5192, x = key_41_cast_fp16)[name = string("kh_81_cast_fp16")]; + tensor var_5194 = const()[name = string("op_5194"), val = tensor([1, 8, 128, 16])]; + tensor vh_81_cast_fp16 = reshape(shape = var_5194, x = value_41_cast_fp16)[name = string("vh_81_cast_fp16")]; + tensor transpose_80_perm_0 = const()[name = string("transpose_80_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_40_reps_0 = const()[name = string("tile_40_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_80_cast_fp16 = transpose(perm = transpose_80_perm_0, x = kh_81_cast_fp16)[name = string("transpose_359")]; + tensor tile_40_cast_fp16 = tile(reps = tile_40_reps_0, x = transpose_80_cast_fp16)[name = string("tile_40_cast_fp16")]; + tensor concat_103 = const()[name = string("concat_103"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_80_cast_fp16 = reshape(shape = concat_103, x = tile_40_cast_fp16)[name = string("reshape_80_cast_fp16")]; + tensor transpose_81_perm_0 = const()[name = string("transpose_81_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_104 = const()[name = string("concat_104"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_81_cast_fp16 = transpose(perm = transpose_81_perm_0, x = reshape_80_cast_fp16)[name = string("transpose_358")]; + tensor reshape_81_cast_fp16 = reshape(shape = concat_104, x = transpose_81_cast_fp16)[name = string("reshape_81_cast_fp16")]; + tensor transpose_82_perm_0 = const()[name = string("transpose_82_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_41_reps_0 = const()[name = string("tile_41_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_82_cast_fp16 = transpose(perm = transpose_82_perm_0, x = vh_81_cast_fp16)[name = string("transpose_357")]; + tensor tile_41_cast_fp16 = tile(reps = tile_41_reps_0, x = transpose_82_cast_fp16)[name = string("tile_41_cast_fp16")]; + tensor concat_105 = const()[name = string("concat_105"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_82_cast_fp16 = reshape(shape = concat_105, x = tile_41_cast_fp16)[name = string("reshape_82_cast_fp16")]; + tensor transpose_83_perm_0 = const()[name = string("transpose_83_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_106 = const()[name = string("concat_106"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_83_cast_fp16 = transpose(perm = transpose_83_perm_0, x = reshape_82_cast_fp16)[name = string("transpose_356")]; + tensor reshape_83_cast_fp16 = reshape(shape = concat_106, x = transpose_83_cast_fp16)[name = string("reshape_83_cast_fp16")]; + fp16 var_5198_to_fp16 = const()[name = string("op_5198_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_5199_cast_fp16 = mul(x = q_125_cast_fp16, y = var_5198_to_fp16)[name = string("op_5199_cast_fp16")]; + tensor transpose_397_perm_0 = const()[name = string("transpose_397_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_87_transpose_x_1 = const()[name = string("w_87_transpose_x_1"), val = bool(true)]; + bool w_87_transpose_y_1 = const()[name = string("w_87_transpose_y_1"), val = bool(false)]; + tensor transpose_397_cast_fp16 = transpose(perm = transpose_397_perm_0, x = reshape_81_cast_fp16)[name = string("transpose_355")]; + tensor w_87_cast_fp16 = matmul(transpose_x = w_87_transpose_x_1, transpose_y = w_87_transpose_y_1, x = var_5199_cast_fp16, y = transpose_397_cast_fp16)[name = string("w_87_cast_fp16")]; + tensor pad_41_to_fp16 = const()[name = string("pad_41_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636352)))]; + tensor var_5202_cast_fp16 = add(x = w_87_cast_fp16, y = pad_41_to_fp16)[name = string("op_5202_cast_fp16")]; + tensor w_89_cast_fp16 = softmax(axis = var_5077, x = var_5202_cast_fp16)[name = string("w_89_cast_fp16")]; + tensor transpose_398_perm_0 = const()[name = string("transpose_398_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_41_transpose_x_1 = const()[name = string("attn_41_transpose_x_1"), val = bool(false)]; + bool attn_41_transpose_y_1 = const()[name = string("attn_41_transpose_y_1"), val = bool(true)]; + tensor transpose_398_cast_fp16 = transpose(perm = transpose_398_perm_0, x = reshape_83_cast_fp16)[name = string("transpose_354")]; + tensor attn_41_cast_fp16 = matmul(transpose_x = attn_41_transpose_x_1, transpose_y = attn_41_transpose_y_1, x = transpose_398_cast_fp16, y = w_89_cast_fp16)[name = string("attn_41_cast_fp16")]; + tensor var_5206 = const()[name = string("op_5206"), val = tensor([1, 2048, 1, 1])]; + tensor input_215_cast_fp16 = reshape(shape = var_5206, x = attn_41_cast_fp16)[name = string("input_215_cast_fp16")]; + string attn_output_41_pad_type_0 = const()[name = string("attn_output_41_pad_type_0"), val = string("valid")]; + tensor attn_output_41_strides_0 = const()[name = string("attn_output_41_strides_0"), val = tensor([1, 1])]; + tensor attn_output_41_pad_0 = const()[name = string("attn_output_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_41_dilations_0 = const()[name = string("attn_output_41_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_41_groups_0 = const()[name = string("attn_output_41_groups_0"), val = int32(1)]; + tensor attn_output_41_cast_fp16 = conv(dilations = attn_output_41_dilations_0, groups = attn_output_41_groups_0, pad = attn_output_41_pad_0, pad_type = attn_output_41_pad_type_0, strides = attn_output_41_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_215_cast_fp16)[name = string("attn_output_41_cast_fp16")]; + tensor x_157_cast_fp16 = add(x = code_embed_11_cast_fp16, y = attn_output_41_cast_fp16)[name = string("x_157_cast_fp16")]; + tensor var_5220_cast_fp16 = mul(x = x_157_cast_fp16, y = x_157_cast_fp16)[name = string("op_5220_cast_fp16")]; + tensor variance_173_axes_0 = const()[name = string("variance_173_axes_0"), val = tensor([1])]; + bool variance_173_keep_dims_0 = const()[name = string("variance_173_keep_dims_0"), val = bool(true)]; + tensor variance_173_cast_fp16 = reduce_mean(axes = variance_173_axes_0, keep_dims = variance_173_keep_dims_0, x = var_5220_cast_fp16)[name = string("variance_173_cast_fp16")]; + fp16 var_5223_to_fp16 = const()[name = string("op_5223_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5224_cast_fp16 = add(x = variance_173_cast_fp16, y = var_5223_to_fp16)[name = string("op_5224_cast_fp16")]; + fp32 var_5225_epsilon_0 = const()[name = string("op_5225_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5225_cast_fp16 = rsqrt(epsilon = var_5225_epsilon_0, x = var_5224_cast_fp16)[name = string("op_5225_cast_fp16")]; + tensor var_5226_cast_fp16 = mul(x = x_157_cast_fp16, y = var_5225_cast_fp16)[name = string("op_5226_cast_fp16")]; + tensor input_217_cast_fp16 = mul(x = var_5226_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_217_cast_fp16")]; + string input_219_pad_type_0 = const()[name = string("input_219_pad_type_0"), val = string("valid")]; + tensor input_219_strides_0 = const()[name = string("input_219_strides_0"), val = tensor([1, 1])]; + tensor input_219_pad_0 = const()[name = string("input_219_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_219_dilations_0 = const()[name = string("input_219_dilations_0"), val = tensor([1, 1])]; + int32 input_219_groups_0 = const()[name = string("input_219_groups_0"), val = int32(1)]; + tensor input_219_cast_fp16 = conv(dilations = input_219_dilations_0, groups = input_219_groups_0, pad = input_219_pad_0, pad_type = input_219_pad_type_0, strides = input_219_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_217_cast_fp16)[name = string("input_219_cast_fp16")]; + tensor var_5234_cast_fp16 = silu(x = input_219_cast_fp16)[name = string("op_5234_cast_fp16")]; + string var_5240_pad_type_0 = const()[name = string("op_5240_pad_type_0"), val = string("valid")]; + tensor var_5240_strides_0 = const()[name = string("op_5240_strides_0"), val = tensor([1, 1])]; + tensor var_5240_pad_0 = const()[name = string("op_5240_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5240_dilations_0 = const()[name = string("op_5240_dilations_0"), val = tensor([1, 1])]; + int32 var_5240_groups_0 = const()[name = string("op_5240_groups_0"), val = int32(1)]; + tensor var_5240_cast_fp16 = conv(dilations = var_5240_dilations_0, groups = var_5240_groups_0, pad = var_5240_pad_0, pad_type = var_5240_pad_type_0, strides = var_5240_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_217_cast_fp16)[name = string("op_5240_cast_fp16")]; + tensor input_221_cast_fp16 = mul(x = var_5234_cast_fp16, y = var_5240_cast_fp16)[name = string("input_221_cast_fp16")]; + string h_41_pad_type_0 = const()[name = string("h_41_pad_type_0"), val = string("valid")]; + tensor h_41_strides_0 = const()[name = string("h_41_strides_0"), val = tensor([1, 1])]; + tensor h_41_pad_0 = const()[name = string("h_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_41_dilations_0 = const()[name = string("h_41_dilations_0"), val = tensor([1, 1])]; + int32 h_41_groups_0 = const()[name = string("h_41_groups_0"), val = int32(1)]; + tensor h_41_cast_fp16 = conv(dilations = h_41_dilations_0, groups = h_41_groups_0, pad = h_41_pad_0, pad_type = h_41_pad_type_0, strides = h_41_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_221_cast_fp16)[name = string("h_41_cast_fp16")]; + tensor x_159_cast_fp16 = add(x = x_157_cast_fp16, y = h_41_cast_fp16)[name = string("x_159_cast_fp16")]; + tensor key_cache_43_begin_0 = const()[name = string("key_cache_43_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_43_end_0 = const()[name = string("key_cache_43_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_43_end_mask_0 = const()[name = string("key_cache_43_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_43_cast_fp16 = slice_by_index(begin = key_cache_43_begin_0, end = key_cache_43_end_0, end_mask = key_cache_43_end_mask_0, x = layer_key_caches_9_cast_fp16)[name = string("key_cache_43_cast_fp16")]; + tensor value_cache_43_begin_0 = const()[name = string("value_cache_43_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_43_end_0 = const()[name = string("value_cache_43_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_43_end_mask_0 = const()[name = string("value_cache_43_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_43_cast_fp16 = slice_by_index(begin = value_cache_43_begin_0, end = value_cache_43_end_0, end_mask = value_cache_43_end_mask_0, x = layer_value_caches_9_cast_fp16)[name = string("value_cache_43_cast_fp16")]; + int32 var_5293 = const()[name = string("op_5293"), val = int32(2)]; + int32 var_5297 = const()[name = string("op_5297"), val = int32(3)]; + tensor var_5312_cast_fp16 = mul(x = x_159_cast_fp16, y = x_159_cast_fp16)[name = string("op_5312_cast_fp16")]; + tensor variance_175_axes_0 = const()[name = string("variance_175_axes_0"), val = tensor([1])]; + bool variance_175_keep_dims_0 = const()[name = string("variance_175_keep_dims_0"), val = bool(true)]; + tensor variance_175_cast_fp16 = reduce_mean(axes = variance_175_axes_0, keep_dims = variance_175_keep_dims_0, x = var_5312_cast_fp16)[name = string("variance_175_cast_fp16")]; + fp16 var_5315_to_fp16 = const()[name = string("op_5315_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5316_cast_fp16 = add(x = variance_175_cast_fp16, y = var_5315_to_fp16)[name = string("op_5316_cast_fp16")]; + fp32 var_5317_epsilon_0 = const()[name = string("op_5317_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5317_cast_fp16 = rsqrt(epsilon = var_5317_epsilon_0, x = var_5316_cast_fp16)[name = string("op_5317_cast_fp16")]; + tensor var_5318_cast_fp16 = mul(x = x_159_cast_fp16, y = var_5317_cast_fp16)[name = string("op_5318_cast_fp16")]; + tensor input_223_cast_fp16 = mul(x = var_5318_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_223_cast_fp16")]; + string q_127_pad_type_0 = const()[name = string("q_127_pad_type_0"), val = string("valid")]; + tensor q_127_strides_0 = const()[name = string("q_127_strides_0"), val = tensor([1, 1])]; + tensor q_127_pad_0 = const()[name = string("q_127_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_127_dilations_0 = const()[name = string("q_127_dilations_0"), val = tensor([1, 1])]; + int32 q_127_groups_0 = const()[name = string("q_127_groups_0"), val = int32(1)]; + tensor q_127_cast_fp16 = conv(dilations = q_127_dilations_0, groups = q_127_groups_0, pad = q_127_pad_0, pad_type = q_127_pad_type_0, strides = q_127_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_223_cast_fp16)[name = string("q_127_cast_fp16")]; + string k_127_pad_type_0 = const()[name = string("k_127_pad_type_0"), val = string("valid")]; + tensor k_127_strides_0 = const()[name = string("k_127_strides_0"), val = tensor([1, 1])]; + tensor k_127_pad_0 = const()[name = string("k_127_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_127_dilations_0 = const()[name = string("k_127_dilations_0"), val = tensor([1, 1])]; + int32 k_127_groups_0 = const()[name = string("k_127_groups_0"), val = int32(1)]; + tensor k_127_cast_fp16 = conv(dilations = k_127_dilations_0, groups = k_127_groups_0, pad = k_127_pad_0, pad_type = k_127_pad_type_0, strides = k_127_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_223_cast_fp16)[name = string("k_127_cast_fp16")]; + string v_43_pad_type_0 = const()[name = string("v_43_pad_type_0"), val = string("valid")]; + tensor v_43_strides_0 = const()[name = string("v_43_strides_0"), val = tensor([1, 1])]; + tensor v_43_pad_0 = const()[name = string("v_43_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_43_dilations_0 = const()[name = string("v_43_dilations_0"), val = tensor([1, 1])]; + int32 v_43_groups_0 = const()[name = string("v_43_groups_0"), val = int32(1)]; + tensor v_43_cast_fp16 = conv(dilations = v_43_dilations_0, groups = v_43_groups_0, pad = v_43_pad_0, pad_type = v_43_pad_type_0, strides = v_43_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_223_cast_fp16)[name = string("v_43_cast_fp16")]; + tensor var_5352 = const()[name = string("op_5352"), val = tensor([16, 128, 1, 1])]; + tensor x_161_cast_fp16 = reshape(shape = var_5352, x = q_127_cast_fp16)[name = string("x_161_cast_fp16")]; + tensor var_5355_cast_fp16 = mul(x = x_161_cast_fp16, y = x_161_cast_fp16)[name = string("op_5355_cast_fp16")]; + tensor variance_177_axes_0 = const()[name = string("variance_177_axes_0"), val = tensor([1])]; + bool variance_177_keep_dims_0 = const()[name = string("variance_177_keep_dims_0"), val = bool(true)]; + tensor variance_177_cast_fp16 = reduce_mean(axes = variance_177_axes_0, keep_dims = variance_177_keep_dims_0, x = var_5355_cast_fp16)[name = string("variance_177_cast_fp16")]; + fp16 var_5358_to_fp16 = const()[name = string("op_5358_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5359_cast_fp16 = add(x = variance_177_cast_fp16, y = var_5358_to_fp16)[name = string("op_5359_cast_fp16")]; + fp32 var_5360_epsilon_0 = const()[name = string("op_5360_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5360_cast_fp16 = rsqrt(epsilon = var_5360_epsilon_0, x = var_5359_cast_fp16)[name = string("op_5360_cast_fp16")]; + tensor var_5361_cast_fp16 = mul(x = x_161_cast_fp16, y = var_5360_cast_fp16)[name = string("op_5361_cast_fp16")]; + tensor q_129_cast_fp16 = mul(x = var_5361_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_129_cast_fp16")]; + tensor var_5363 = const()[name = string("op_5363"), val = tensor([8, 128, 1, 1])]; + tensor x_163_cast_fp16 = reshape(shape = var_5363, x = k_127_cast_fp16)[name = string("x_163_cast_fp16")]; + tensor var_5366_cast_fp16 = mul(x = x_163_cast_fp16, y = x_163_cast_fp16)[name = string("op_5366_cast_fp16")]; + tensor variance_179_axes_0 = const()[name = string("variance_179_axes_0"), val = tensor([1])]; + bool variance_179_keep_dims_0 = const()[name = string("variance_179_keep_dims_0"), val = bool(true)]; + tensor variance_179_cast_fp16 = reduce_mean(axes = variance_179_axes_0, keep_dims = variance_179_keep_dims_0, x = var_5366_cast_fp16)[name = string("variance_179_cast_fp16")]; + fp16 var_5369_to_fp16 = const()[name = string("op_5369_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5370_cast_fp16 = add(x = variance_179_cast_fp16, y = var_5369_to_fp16)[name = string("op_5370_cast_fp16")]; + fp32 var_5371_epsilon_0 = const()[name = string("op_5371_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5371_cast_fp16 = rsqrt(epsilon = var_5371_epsilon_0, x = var_5370_cast_fp16)[name = string("op_5371_cast_fp16")]; + tensor var_5372_cast_fp16 = mul(x = x_163_cast_fp16, y = var_5371_cast_fp16)[name = string("op_5372_cast_fp16")]; + tensor k_129_cast_fp16 = mul(x = var_5372_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_129_cast_fp16")]; + tensor var_5374 = const()[name = string("op_5374"), val = tensor([1, 16, 128, 1])]; + tensor z_85_cast_fp16 = reshape(shape = var_5374, x = q_129_cast_fp16)[name = string("z_85_cast_fp16")]; + tensor var_5376 = const()[name = string("op_5376"), val = tensor([1, 8, 128, 1])]; + tensor z_87_cast_fp16 = reshape(shape = var_5376, x = k_129_cast_fp16)[name = string("z_87_cast_fp16")]; + tensor z1_85_begin_0 = const()[name = string("z1_85_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_85_end_0 = const()[name = string("z1_85_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_85_end_mask_0 = const()[name = string("z1_85_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_85_cast_fp16 = slice_by_index(begin = z1_85_begin_0, end = z1_85_end_0, end_mask = z1_85_end_mask_0, x = z_85_cast_fp16)[name = string("z1_85_cast_fp16")]; + tensor z2_85_begin_0 = const()[name = string("z2_85_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_85_end_0 = const()[name = string("z2_85_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_85_end_mask_0 = const()[name = string("z2_85_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_85_cast_fp16 = slice_by_index(begin = z2_85_begin_0, end = z2_85_end_0, end_mask = z2_85_end_mask_0, x = z_85_cast_fp16)[name = string("z2_85_cast_fp16")]; + tensor var_5384_cast_fp16 = mul(x = z_85_cast_fp16, y = cos_41_to_fp16)[name = string("op_5384_cast_fp16")]; + fp16 const_47_promoted_to_fp16 = const()[name = string("const_47_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5385_cast_fp16 = mul(x = z2_85_cast_fp16, y = const_47_promoted_to_fp16)[name = string("op_5385_cast_fp16")]; + bool var_5387_interleave_0 = const()[name = string("op_5387_interleave_0"), val = bool(false)]; + tensor var_5387_cast_fp16 = concat(axis = var_5293, interleave = var_5387_interleave_0, values = (var_5385_cast_fp16, z1_85_cast_fp16))[name = string("op_5387_cast_fp16")]; + tensor var_5388_cast_fp16 = mul(x = var_5387_cast_fp16, y = sin_41_to_fp16)[name = string("op_5388_cast_fp16")]; + tensor q_131_cast_fp16 = add(x = var_5384_cast_fp16, y = var_5388_cast_fp16)[name = string("q_131_cast_fp16")]; + tensor z1_87_begin_0 = const()[name = string("z1_87_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_87_end_0 = const()[name = string("z1_87_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_87_end_mask_0 = const()[name = string("z1_87_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_87_cast_fp16 = slice_by_index(begin = z1_87_begin_0, end = z1_87_end_0, end_mask = z1_87_end_mask_0, x = z_87_cast_fp16)[name = string("z1_87_cast_fp16")]; + tensor z2_87_begin_0 = const()[name = string("z2_87_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_87_end_0 = const()[name = string("z2_87_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_87_end_mask_0 = const()[name = string("z2_87_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_87_cast_fp16 = slice_by_index(begin = z2_87_begin_0, end = z2_87_end_0, end_mask = z2_87_end_mask_0, x = z_87_cast_fp16)[name = string("z2_87_cast_fp16")]; + tensor var_5396_cast_fp16 = mul(x = z_87_cast_fp16, y = cos_41_to_fp16)[name = string("op_5396_cast_fp16")]; + fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5397_cast_fp16 = mul(x = z2_87_cast_fp16, y = const_48_promoted_to_fp16)[name = string("op_5397_cast_fp16")]; + bool var_5399_interleave_0 = const()[name = string("op_5399_interleave_0"), val = bool(false)]; + tensor var_5399_cast_fp16 = concat(axis = var_5293, interleave = var_5399_interleave_0, values = (var_5397_cast_fp16, z1_87_cast_fp16))[name = string("op_5399_cast_fp16")]; + tensor var_5400_cast_fp16 = mul(x = var_5399_cast_fp16, y = sin_41_to_fp16)[name = string("op_5400_cast_fp16")]; + tensor k_131_cast_fp16 = add(x = var_5396_cast_fp16, y = var_5400_cast_fp16)[name = string("k_131_cast_fp16")]; + tensor var_5402 = const()[name = string("op_5402"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_43_cast_fp16 = reshape(shape = var_5402, x = k_131_cast_fp16)[name = string("cur_key_43_cast_fp16")]; + tensor var_5404_to_fp16 = const()[name = string("op_5404_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636096)))]; + tensor var_5405_cast_fp16 = mul(x = key_cache_43_cast_fp16, y = var_5404_to_fp16)[name = string("op_5405_cast_fp16")]; + tensor upd_43_to_fp16 = const()[name = string("upd_43_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636224)))]; + tensor var_5406_cast_fp16 = mul(x = cur_key_43_cast_fp16, y = upd_43_to_fp16)[name = string("op_5406_cast_fp16")]; + tensor key_43_cast_fp16 = add(x = var_5405_cast_fp16, y = var_5406_cast_fp16)[name = string("key_43_cast_fp16")]; + tensor var_5408_to_fp16 = const()[name = string("op_5408_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636096)))]; + tensor var_5409_cast_fp16 = mul(x = value_cache_43_cast_fp16, y = var_5408_to_fp16)[name = string("op_5409_cast_fp16")]; + tensor var_5410_cast_fp16 = mul(x = v_43_cast_fp16, y = upd_43_to_fp16)[name = string("op_5410_cast_fp16")]; + tensor value_43_cast_fp16 = add(x = var_5409_cast_fp16, y = var_5410_cast_fp16)[name = string("value_43_cast_fp16")]; + tensor var_5412 = const()[name = string("op_5412"), val = tensor([1, 8, 128, 16])]; + tensor kh_85_cast_fp16 = reshape(shape = var_5412, x = key_43_cast_fp16)[name = string("kh_85_cast_fp16")]; + tensor var_5414 = const()[name = string("op_5414"), val = tensor([1, 8, 128, 16])]; + tensor vh_85_cast_fp16 = reshape(shape = var_5414, x = value_43_cast_fp16)[name = string("vh_85_cast_fp16")]; + tensor transpose_84_perm_0 = const()[name = string("transpose_84_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_42_reps_0 = const()[name = string("tile_42_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_84_cast_fp16 = transpose(perm = transpose_84_perm_0, x = kh_85_cast_fp16)[name = string("transpose_353")]; + tensor tile_42_cast_fp16 = tile(reps = tile_42_reps_0, x = transpose_84_cast_fp16)[name = string("tile_42_cast_fp16")]; + tensor concat_107 = const()[name = string("concat_107"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_84_cast_fp16 = reshape(shape = concat_107, x = tile_42_cast_fp16)[name = string("reshape_84_cast_fp16")]; + tensor transpose_85_perm_0 = const()[name = string("transpose_85_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_108 = const()[name = string("concat_108"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_85_cast_fp16 = transpose(perm = transpose_85_perm_0, x = reshape_84_cast_fp16)[name = string("transpose_352")]; + tensor reshape_85_cast_fp16 = reshape(shape = concat_108, x = transpose_85_cast_fp16)[name = string("reshape_85_cast_fp16")]; + tensor transpose_86_perm_0 = const()[name = string("transpose_86_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_43_reps_0 = const()[name = string("tile_43_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_86_cast_fp16 = transpose(perm = transpose_86_perm_0, x = vh_85_cast_fp16)[name = string("transpose_351")]; + tensor tile_43_cast_fp16 = tile(reps = tile_43_reps_0, x = transpose_86_cast_fp16)[name = string("tile_43_cast_fp16")]; + tensor concat_109 = const()[name = string("concat_109"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_86_cast_fp16 = reshape(shape = concat_109, x = tile_43_cast_fp16)[name = string("reshape_86_cast_fp16")]; + tensor transpose_87_perm_0 = const()[name = string("transpose_87_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_110 = const()[name = string("concat_110"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_87_cast_fp16 = transpose(perm = transpose_87_perm_0, x = reshape_86_cast_fp16)[name = string("transpose_350")]; + tensor reshape_87_cast_fp16 = reshape(shape = concat_110, x = transpose_87_cast_fp16)[name = string("reshape_87_cast_fp16")]; + fp16 var_5418_to_fp16 = const()[name = string("op_5418_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_5419_cast_fp16 = mul(x = q_131_cast_fp16, y = var_5418_to_fp16)[name = string("op_5419_cast_fp16")]; + tensor transpose_401_perm_0 = const()[name = string("transpose_401_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_91_transpose_x_1 = const()[name = string("w_91_transpose_x_1"), val = bool(true)]; + bool w_91_transpose_y_1 = const()[name = string("w_91_transpose_y_1"), val = bool(false)]; + tensor transpose_401_cast_fp16 = transpose(perm = transpose_401_perm_0, x = reshape_85_cast_fp16)[name = string("transpose_349")]; + tensor w_91_cast_fp16 = matmul(transpose_x = w_91_transpose_x_1, transpose_y = w_91_transpose_y_1, x = var_5419_cast_fp16, y = transpose_401_cast_fp16)[name = string("w_91_cast_fp16")]; + tensor pad_43_to_fp16 = const()[name = string("pad_43_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636352)))]; + tensor var_5422_cast_fp16 = add(x = w_91_cast_fp16, y = pad_43_to_fp16)[name = string("op_5422_cast_fp16")]; + tensor w_93_cast_fp16 = softmax(axis = var_5297, x = var_5422_cast_fp16)[name = string("w_93_cast_fp16")]; + tensor transpose_402_perm_0 = const()[name = string("transpose_402_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_43_transpose_x_1 = const()[name = string("attn_43_transpose_x_1"), val = bool(false)]; + bool attn_43_transpose_y_1 = const()[name = string("attn_43_transpose_y_1"), val = bool(true)]; + tensor transpose_402_cast_fp16 = transpose(perm = transpose_402_perm_0, x = reshape_87_cast_fp16)[name = string("transpose_348")]; + tensor attn_43_cast_fp16 = matmul(transpose_x = attn_43_transpose_x_1, transpose_y = attn_43_transpose_y_1, x = transpose_402_cast_fp16, y = w_93_cast_fp16)[name = string("attn_43_cast_fp16")]; + tensor var_5426 = const()[name = string("op_5426"), val = tensor([1, 2048, 1, 1])]; + tensor input_225_cast_fp16 = reshape(shape = var_5426, x = attn_43_cast_fp16)[name = string("input_225_cast_fp16")]; + string attn_output_43_pad_type_0 = const()[name = string("attn_output_43_pad_type_0"), val = string("valid")]; + tensor attn_output_43_strides_0 = const()[name = string("attn_output_43_strides_0"), val = tensor([1, 1])]; + tensor attn_output_43_pad_0 = const()[name = string("attn_output_43_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_43_dilations_0 = const()[name = string("attn_output_43_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_43_groups_0 = const()[name = string("attn_output_43_groups_0"), val = int32(1)]; + tensor attn_output_43_cast_fp16 = conv(dilations = attn_output_43_dilations_0, groups = attn_output_43_groups_0, pad = attn_output_43_pad_0, pad_type = attn_output_43_pad_type_0, strides = attn_output_43_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_225_cast_fp16)[name = string("attn_output_43_cast_fp16")]; + tensor x_165_cast_fp16 = add(x = x_159_cast_fp16, y = attn_output_43_cast_fp16)[name = string("x_165_cast_fp16")]; + tensor var_5440_cast_fp16 = mul(x = x_165_cast_fp16, y = x_165_cast_fp16)[name = string("op_5440_cast_fp16")]; + tensor variance_181_axes_0 = const()[name = string("variance_181_axes_0"), val = tensor([1])]; + bool variance_181_keep_dims_0 = const()[name = string("variance_181_keep_dims_0"), val = bool(true)]; + tensor variance_181_cast_fp16 = reduce_mean(axes = variance_181_axes_0, keep_dims = variance_181_keep_dims_0, x = var_5440_cast_fp16)[name = string("variance_181_cast_fp16")]; + fp16 var_5443_to_fp16 = const()[name = string("op_5443_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5444_cast_fp16 = add(x = variance_181_cast_fp16, y = var_5443_to_fp16)[name = string("op_5444_cast_fp16")]; + fp32 var_5445_epsilon_0 = const()[name = string("op_5445_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5445_cast_fp16 = rsqrt(epsilon = var_5445_epsilon_0, x = var_5444_cast_fp16)[name = string("op_5445_cast_fp16")]; + tensor var_5446_cast_fp16 = mul(x = x_165_cast_fp16, y = var_5445_cast_fp16)[name = string("op_5446_cast_fp16")]; + tensor input_227_cast_fp16 = mul(x = var_5446_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_227_cast_fp16")]; + string input_229_pad_type_0 = const()[name = string("input_229_pad_type_0"), val = string("valid")]; + tensor input_229_strides_0 = const()[name = string("input_229_strides_0"), val = tensor([1, 1])]; + tensor input_229_pad_0 = const()[name = string("input_229_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_229_dilations_0 = const()[name = string("input_229_dilations_0"), val = tensor([1, 1])]; + int32 input_229_groups_0 = const()[name = string("input_229_groups_0"), val = int32(1)]; + tensor input_229_cast_fp16 = conv(dilations = input_229_dilations_0, groups = input_229_groups_0, pad = input_229_pad_0, pad_type = input_229_pad_type_0, strides = input_229_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_227_cast_fp16)[name = string("input_229_cast_fp16")]; + tensor var_5454_cast_fp16 = silu(x = input_229_cast_fp16)[name = string("op_5454_cast_fp16")]; + string var_5460_pad_type_0 = const()[name = string("op_5460_pad_type_0"), val = string("valid")]; + tensor var_5460_strides_0 = const()[name = string("op_5460_strides_0"), val = tensor([1, 1])]; + tensor var_5460_pad_0 = const()[name = string("op_5460_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5460_dilations_0 = const()[name = string("op_5460_dilations_0"), val = tensor([1, 1])]; + int32 var_5460_groups_0 = const()[name = string("op_5460_groups_0"), val = int32(1)]; + tensor var_5460_cast_fp16 = conv(dilations = var_5460_dilations_0, groups = var_5460_groups_0, pad = var_5460_pad_0, pad_type = var_5460_pad_type_0, strides = var_5460_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_227_cast_fp16)[name = string("op_5460_cast_fp16")]; + tensor input_231_cast_fp16 = mul(x = var_5454_cast_fp16, y = var_5460_cast_fp16)[name = string("input_231_cast_fp16")]; + string h_43_pad_type_0 = const()[name = string("h_43_pad_type_0"), val = string("valid")]; + tensor h_43_strides_0 = const()[name = string("h_43_strides_0"), val = tensor([1, 1])]; + tensor h_43_pad_0 = const()[name = string("h_43_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_43_dilations_0 = const()[name = string("h_43_dilations_0"), val = tensor([1, 1])]; + int32 h_43_groups_0 = const()[name = string("h_43_groups_0"), val = int32(1)]; + tensor h_43_cast_fp16 = conv(dilations = h_43_dilations_0, groups = h_43_groups_0, pad = h_43_pad_0, pad_type = h_43_pad_type_0, strides = h_43_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_231_cast_fp16)[name = string("h_43_cast_fp16")]; + tensor x_167_cast_fp16 = add(x = x_165_cast_fp16, y = h_43_cast_fp16)[name = string("x_167_cast_fp16")]; + tensor key_cache_45_begin_0 = const()[name = string("key_cache_45_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_45_end_0 = const()[name = string("key_cache_45_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_45_end_mask_0 = const()[name = string("key_cache_45_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_45_cast_fp16 = slice_by_index(begin = key_cache_45_begin_0, end = key_cache_45_end_0, end_mask = key_cache_45_end_mask_0, x = layer_key_caches_9_cast_fp16)[name = string("key_cache_45_cast_fp16")]; + tensor value_cache_45_begin_0 = const()[name = string("value_cache_45_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_45_end_0 = const()[name = string("value_cache_45_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_45_end_mask_0 = const()[name = string("value_cache_45_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_45_cast_fp16 = slice_by_index(begin = value_cache_45_begin_0, end = value_cache_45_end_0, end_mask = value_cache_45_end_mask_0, x = layer_value_caches_9_cast_fp16)[name = string("value_cache_45_cast_fp16")]; + int32 var_5513 = const()[name = string("op_5513"), val = int32(2)]; + int32 var_5517 = const()[name = string("op_5517"), val = int32(3)]; + tensor var_5532_cast_fp16 = mul(x = x_167_cast_fp16, y = x_167_cast_fp16)[name = string("op_5532_cast_fp16")]; + tensor variance_183_axes_0 = const()[name = string("variance_183_axes_0"), val = tensor([1])]; + bool variance_183_keep_dims_0 = const()[name = string("variance_183_keep_dims_0"), val = bool(true)]; + tensor variance_183_cast_fp16 = reduce_mean(axes = variance_183_axes_0, keep_dims = variance_183_keep_dims_0, x = var_5532_cast_fp16)[name = string("variance_183_cast_fp16")]; + fp16 var_5535_to_fp16 = const()[name = string("op_5535_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5536_cast_fp16 = add(x = variance_183_cast_fp16, y = var_5535_to_fp16)[name = string("op_5536_cast_fp16")]; + fp32 var_5537_epsilon_0 = const()[name = string("op_5537_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5537_cast_fp16 = rsqrt(epsilon = var_5537_epsilon_0, x = var_5536_cast_fp16)[name = string("op_5537_cast_fp16")]; + tensor var_5538_cast_fp16 = mul(x = x_167_cast_fp16, y = var_5537_cast_fp16)[name = string("op_5538_cast_fp16")]; + tensor input_233_cast_fp16 = mul(x = var_5538_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_233_cast_fp16")]; + string q_133_pad_type_0 = const()[name = string("q_133_pad_type_0"), val = string("valid")]; + tensor q_133_strides_0 = const()[name = string("q_133_strides_0"), val = tensor([1, 1])]; + tensor q_133_pad_0 = const()[name = string("q_133_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_133_dilations_0 = const()[name = string("q_133_dilations_0"), val = tensor([1, 1])]; + int32 q_133_groups_0 = const()[name = string("q_133_groups_0"), val = int32(1)]; + tensor q_133_cast_fp16 = conv(dilations = q_133_dilations_0, groups = q_133_groups_0, pad = q_133_pad_0, pad_type = q_133_pad_type_0, strides = q_133_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_233_cast_fp16)[name = string("q_133_cast_fp16")]; + string k_133_pad_type_0 = const()[name = string("k_133_pad_type_0"), val = string("valid")]; + tensor k_133_strides_0 = const()[name = string("k_133_strides_0"), val = tensor([1, 1])]; + tensor k_133_pad_0 = const()[name = string("k_133_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_133_dilations_0 = const()[name = string("k_133_dilations_0"), val = tensor([1, 1])]; + int32 k_133_groups_0 = const()[name = string("k_133_groups_0"), val = int32(1)]; + tensor k_133_cast_fp16 = conv(dilations = k_133_dilations_0, groups = k_133_groups_0, pad = k_133_pad_0, pad_type = k_133_pad_type_0, strides = k_133_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_233_cast_fp16)[name = string("k_133_cast_fp16")]; + string v_45_pad_type_0 = const()[name = string("v_45_pad_type_0"), val = string("valid")]; + tensor v_45_strides_0 = const()[name = string("v_45_strides_0"), val = tensor([1, 1])]; + tensor v_45_pad_0 = const()[name = string("v_45_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_45_dilations_0 = const()[name = string("v_45_dilations_0"), val = tensor([1, 1])]; + int32 v_45_groups_0 = const()[name = string("v_45_groups_0"), val = int32(1)]; + tensor v_45_cast_fp16 = conv(dilations = v_45_dilations_0, groups = v_45_groups_0, pad = v_45_pad_0, pad_type = v_45_pad_type_0, strides = v_45_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_233_cast_fp16)[name = string("v_45_cast_fp16")]; + tensor var_5572 = const()[name = string("op_5572"), val = tensor([16, 128, 1, 1])]; + tensor x_169_cast_fp16 = reshape(shape = var_5572, x = q_133_cast_fp16)[name = string("x_169_cast_fp16")]; + tensor var_5575_cast_fp16 = mul(x = x_169_cast_fp16, y = x_169_cast_fp16)[name = string("op_5575_cast_fp16")]; + tensor variance_185_axes_0 = const()[name = string("variance_185_axes_0"), val = tensor([1])]; + bool variance_185_keep_dims_0 = const()[name = string("variance_185_keep_dims_0"), val = bool(true)]; + tensor variance_185_cast_fp16 = reduce_mean(axes = variance_185_axes_0, keep_dims = variance_185_keep_dims_0, x = var_5575_cast_fp16)[name = string("variance_185_cast_fp16")]; + fp16 var_5578_to_fp16 = const()[name = string("op_5578_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5579_cast_fp16 = add(x = variance_185_cast_fp16, y = var_5578_to_fp16)[name = string("op_5579_cast_fp16")]; + fp32 var_5580_epsilon_0 = const()[name = string("op_5580_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5580_cast_fp16 = rsqrt(epsilon = var_5580_epsilon_0, x = var_5579_cast_fp16)[name = string("op_5580_cast_fp16")]; + tensor var_5581_cast_fp16 = mul(x = x_169_cast_fp16, y = var_5580_cast_fp16)[name = string("op_5581_cast_fp16")]; + tensor q_135_cast_fp16 = mul(x = var_5581_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_135_cast_fp16")]; + tensor var_5583 = const()[name = string("op_5583"), val = tensor([8, 128, 1, 1])]; + tensor x_171_cast_fp16 = reshape(shape = var_5583, x = k_133_cast_fp16)[name = string("x_171_cast_fp16")]; + tensor var_5586_cast_fp16 = mul(x = x_171_cast_fp16, y = x_171_cast_fp16)[name = string("op_5586_cast_fp16")]; + tensor variance_187_axes_0 = const()[name = string("variance_187_axes_0"), val = tensor([1])]; + bool variance_187_keep_dims_0 = const()[name = string("variance_187_keep_dims_0"), val = bool(true)]; + tensor variance_187_cast_fp16 = reduce_mean(axes = variance_187_axes_0, keep_dims = variance_187_keep_dims_0, x = var_5586_cast_fp16)[name = string("variance_187_cast_fp16")]; + fp16 var_5589_to_fp16 = const()[name = string("op_5589_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5590_cast_fp16 = add(x = variance_187_cast_fp16, y = var_5589_to_fp16)[name = string("op_5590_cast_fp16")]; + fp32 var_5591_epsilon_0 = const()[name = string("op_5591_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5591_cast_fp16 = rsqrt(epsilon = var_5591_epsilon_0, x = var_5590_cast_fp16)[name = string("op_5591_cast_fp16")]; + tensor var_5592_cast_fp16 = mul(x = x_171_cast_fp16, y = var_5591_cast_fp16)[name = string("op_5592_cast_fp16")]; + tensor k_135_cast_fp16 = mul(x = var_5592_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_135_cast_fp16")]; + tensor var_5594 = const()[name = string("op_5594"), val = tensor([1, 16, 128, 1])]; + tensor z_89_cast_fp16 = reshape(shape = var_5594, x = q_135_cast_fp16)[name = string("z_89_cast_fp16")]; + tensor var_5596 = const()[name = string("op_5596"), val = tensor([1, 8, 128, 1])]; + tensor z_91_cast_fp16 = reshape(shape = var_5596, x = k_135_cast_fp16)[name = string("z_91_cast_fp16")]; + tensor z1_89_begin_0 = const()[name = string("z1_89_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_89_end_0 = const()[name = string("z1_89_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_89_end_mask_0 = const()[name = string("z1_89_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_89_cast_fp16 = slice_by_index(begin = z1_89_begin_0, end = z1_89_end_0, end_mask = z1_89_end_mask_0, x = z_89_cast_fp16)[name = string("z1_89_cast_fp16")]; + tensor z2_89_begin_0 = const()[name = string("z2_89_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_89_end_0 = const()[name = string("z2_89_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_89_end_mask_0 = const()[name = string("z2_89_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_89_cast_fp16 = slice_by_index(begin = z2_89_begin_0, end = z2_89_end_0, end_mask = z2_89_end_mask_0, x = z_89_cast_fp16)[name = string("z2_89_cast_fp16")]; + tensor var_5604_cast_fp16 = mul(x = z_89_cast_fp16, y = cos_41_to_fp16)[name = string("op_5604_cast_fp16")]; + fp16 const_49_promoted_to_fp16 = const()[name = string("const_49_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5605_cast_fp16 = mul(x = z2_89_cast_fp16, y = const_49_promoted_to_fp16)[name = string("op_5605_cast_fp16")]; + bool var_5607_interleave_0 = const()[name = string("op_5607_interleave_0"), val = bool(false)]; + tensor var_5607_cast_fp16 = concat(axis = var_5513, interleave = var_5607_interleave_0, values = (var_5605_cast_fp16, z1_89_cast_fp16))[name = string("op_5607_cast_fp16")]; + tensor var_5608_cast_fp16 = mul(x = var_5607_cast_fp16, y = sin_41_to_fp16)[name = string("op_5608_cast_fp16")]; + tensor q_137_cast_fp16 = add(x = var_5604_cast_fp16, y = var_5608_cast_fp16)[name = string("q_137_cast_fp16")]; + tensor z1_91_begin_0 = const()[name = string("z1_91_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_91_end_0 = const()[name = string("z1_91_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_91_end_mask_0 = const()[name = string("z1_91_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_91_cast_fp16 = slice_by_index(begin = z1_91_begin_0, end = z1_91_end_0, end_mask = z1_91_end_mask_0, x = z_91_cast_fp16)[name = string("z1_91_cast_fp16")]; + tensor z2_91_begin_0 = const()[name = string("z2_91_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_91_end_0 = const()[name = string("z2_91_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_91_end_mask_0 = const()[name = string("z2_91_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_91_cast_fp16 = slice_by_index(begin = z2_91_begin_0, end = z2_91_end_0, end_mask = z2_91_end_mask_0, x = z_91_cast_fp16)[name = string("z2_91_cast_fp16")]; + tensor var_5616_cast_fp16 = mul(x = z_91_cast_fp16, y = cos_41_to_fp16)[name = string("op_5616_cast_fp16")]; + fp16 const_50_promoted_to_fp16 = const()[name = string("const_50_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5617_cast_fp16 = mul(x = z2_91_cast_fp16, y = const_50_promoted_to_fp16)[name = string("op_5617_cast_fp16")]; + bool var_5619_interleave_0 = const()[name = string("op_5619_interleave_0"), val = bool(false)]; + tensor var_5619_cast_fp16 = concat(axis = var_5513, interleave = var_5619_interleave_0, values = (var_5617_cast_fp16, z1_91_cast_fp16))[name = string("op_5619_cast_fp16")]; + tensor var_5620_cast_fp16 = mul(x = var_5619_cast_fp16, y = sin_41_to_fp16)[name = string("op_5620_cast_fp16")]; + tensor k_137_cast_fp16 = add(x = var_5616_cast_fp16, y = var_5620_cast_fp16)[name = string("k_137_cast_fp16")]; + tensor var_5622 = const()[name = string("op_5622"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_45_cast_fp16 = reshape(shape = var_5622, x = k_137_cast_fp16)[name = string("cur_key_45_cast_fp16")]; + tensor var_5624_to_fp16 = const()[name = string("op_5624_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636096)))]; + tensor var_5625_cast_fp16 = mul(x = key_cache_45_cast_fp16, y = var_5624_to_fp16)[name = string("op_5625_cast_fp16")]; + tensor upd_45_to_fp16 = const()[name = string("upd_45_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636224)))]; + tensor var_5626_cast_fp16 = mul(x = cur_key_45_cast_fp16, y = upd_45_to_fp16)[name = string("op_5626_cast_fp16")]; + tensor key_45_cast_fp16 = add(x = var_5625_cast_fp16, y = var_5626_cast_fp16)[name = string("key_45_cast_fp16")]; + tensor var_5628_to_fp16 = const()[name = string("op_5628_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636096)))]; + tensor var_5629_cast_fp16 = mul(x = value_cache_45_cast_fp16, y = var_5628_to_fp16)[name = string("op_5629_cast_fp16")]; + tensor var_5630_cast_fp16 = mul(x = v_45_cast_fp16, y = upd_45_to_fp16)[name = string("op_5630_cast_fp16")]; + tensor value_45_cast_fp16 = add(x = var_5629_cast_fp16, y = var_5630_cast_fp16)[name = string("value_45_cast_fp16")]; + tensor var_5632 = const()[name = string("op_5632"), val = tensor([1, 8, 128, 16])]; + tensor kh_89_cast_fp16 = reshape(shape = var_5632, x = key_45_cast_fp16)[name = string("kh_89_cast_fp16")]; + tensor var_5634 = const()[name = string("op_5634"), val = tensor([1, 8, 128, 16])]; + tensor vh_89_cast_fp16 = reshape(shape = var_5634, x = value_45_cast_fp16)[name = string("vh_89_cast_fp16")]; + tensor transpose_88_perm_0 = const()[name = string("transpose_88_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_44_reps_0 = const()[name = string("tile_44_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_88_cast_fp16 = transpose(perm = transpose_88_perm_0, x = kh_89_cast_fp16)[name = string("transpose_347")]; + tensor tile_44_cast_fp16 = tile(reps = tile_44_reps_0, x = transpose_88_cast_fp16)[name = string("tile_44_cast_fp16")]; + tensor concat_111 = const()[name = string("concat_111"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_88_cast_fp16 = reshape(shape = concat_111, x = tile_44_cast_fp16)[name = string("reshape_88_cast_fp16")]; + tensor transpose_89_perm_0 = const()[name = string("transpose_89_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_112 = const()[name = string("concat_112"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_89_cast_fp16 = transpose(perm = transpose_89_perm_0, x = reshape_88_cast_fp16)[name = string("transpose_346")]; + tensor reshape_89_cast_fp16 = reshape(shape = concat_112, x = transpose_89_cast_fp16)[name = string("reshape_89_cast_fp16")]; + tensor transpose_90_perm_0 = const()[name = string("transpose_90_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_45_reps_0 = const()[name = string("tile_45_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_90_cast_fp16 = transpose(perm = transpose_90_perm_0, x = vh_89_cast_fp16)[name = string("transpose_345")]; + tensor tile_45_cast_fp16 = tile(reps = tile_45_reps_0, x = transpose_90_cast_fp16)[name = string("tile_45_cast_fp16")]; + tensor concat_113 = const()[name = string("concat_113"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_90_cast_fp16 = reshape(shape = concat_113, x = tile_45_cast_fp16)[name = string("reshape_90_cast_fp16")]; + tensor transpose_91_perm_0 = const()[name = string("transpose_91_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_114 = const()[name = string("concat_114"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_91_cast_fp16 = transpose(perm = transpose_91_perm_0, x = reshape_90_cast_fp16)[name = string("transpose_344")]; + tensor reshape_91_cast_fp16 = reshape(shape = concat_114, x = transpose_91_cast_fp16)[name = string("reshape_91_cast_fp16")]; + fp16 var_5638_to_fp16 = const()[name = string("op_5638_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_5639_cast_fp16 = mul(x = q_137_cast_fp16, y = var_5638_to_fp16)[name = string("op_5639_cast_fp16")]; + tensor transpose_405_perm_0 = const()[name = string("transpose_405_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_95_transpose_x_1 = const()[name = string("w_95_transpose_x_1"), val = bool(true)]; + bool w_95_transpose_y_1 = const()[name = string("w_95_transpose_y_1"), val = bool(false)]; + tensor transpose_405_cast_fp16 = transpose(perm = transpose_405_perm_0, x = reshape_89_cast_fp16)[name = string("transpose_343")]; + tensor w_95_cast_fp16 = matmul(transpose_x = w_95_transpose_x_1, transpose_y = w_95_transpose_y_1, x = var_5639_cast_fp16, y = transpose_405_cast_fp16)[name = string("w_95_cast_fp16")]; + tensor pad_45_to_fp16 = const()[name = string("pad_45_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636352)))]; + tensor var_5642_cast_fp16 = add(x = w_95_cast_fp16, y = pad_45_to_fp16)[name = string("op_5642_cast_fp16")]; + tensor w_97_cast_fp16 = softmax(axis = var_5517, x = var_5642_cast_fp16)[name = string("w_97_cast_fp16")]; + tensor transpose_406_perm_0 = const()[name = string("transpose_406_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_45_transpose_x_1 = const()[name = string("attn_45_transpose_x_1"), val = bool(false)]; + bool attn_45_transpose_y_1 = const()[name = string("attn_45_transpose_y_1"), val = bool(true)]; + tensor transpose_406_cast_fp16 = transpose(perm = transpose_406_perm_0, x = reshape_91_cast_fp16)[name = string("transpose_342")]; + tensor attn_45_cast_fp16 = matmul(transpose_x = attn_45_transpose_x_1, transpose_y = attn_45_transpose_y_1, x = transpose_406_cast_fp16, y = w_97_cast_fp16)[name = string("attn_45_cast_fp16")]; + tensor var_5646 = const()[name = string("op_5646"), val = tensor([1, 2048, 1, 1])]; + tensor input_235_cast_fp16 = reshape(shape = var_5646, x = attn_45_cast_fp16)[name = string("input_235_cast_fp16")]; + string attn_output_45_pad_type_0 = const()[name = string("attn_output_45_pad_type_0"), val = string("valid")]; + tensor attn_output_45_strides_0 = const()[name = string("attn_output_45_strides_0"), val = tensor([1, 1])]; + tensor attn_output_45_pad_0 = const()[name = string("attn_output_45_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_45_dilations_0 = const()[name = string("attn_output_45_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_45_groups_0 = const()[name = string("attn_output_45_groups_0"), val = int32(1)]; + tensor attn_output_45_cast_fp16 = conv(dilations = attn_output_45_dilations_0, groups = attn_output_45_groups_0, pad = attn_output_45_pad_0, pad_type = attn_output_45_pad_type_0, strides = attn_output_45_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_235_cast_fp16)[name = string("attn_output_45_cast_fp16")]; + tensor x_173_cast_fp16 = add(x = x_167_cast_fp16, y = attn_output_45_cast_fp16)[name = string("x_173_cast_fp16")]; + tensor var_5660_cast_fp16 = mul(x = x_173_cast_fp16, y = x_173_cast_fp16)[name = string("op_5660_cast_fp16")]; + tensor variance_189_axes_0 = const()[name = string("variance_189_axes_0"), val = tensor([1])]; + bool variance_189_keep_dims_0 = const()[name = string("variance_189_keep_dims_0"), val = bool(true)]; + tensor variance_189_cast_fp16 = reduce_mean(axes = variance_189_axes_0, keep_dims = variance_189_keep_dims_0, x = var_5660_cast_fp16)[name = string("variance_189_cast_fp16")]; + fp16 var_5663_to_fp16 = const()[name = string("op_5663_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5664_cast_fp16 = add(x = variance_189_cast_fp16, y = var_5663_to_fp16)[name = string("op_5664_cast_fp16")]; + fp32 var_5665_epsilon_0 = const()[name = string("op_5665_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5665_cast_fp16 = rsqrt(epsilon = var_5665_epsilon_0, x = var_5664_cast_fp16)[name = string("op_5665_cast_fp16")]; + tensor var_5666_cast_fp16 = mul(x = x_173_cast_fp16, y = var_5665_cast_fp16)[name = string("op_5666_cast_fp16")]; + tensor input_237_cast_fp16 = mul(x = var_5666_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_237_cast_fp16")]; + string input_239_pad_type_0 = const()[name = string("input_239_pad_type_0"), val = string("valid")]; + tensor input_239_strides_0 = const()[name = string("input_239_strides_0"), val = tensor([1, 1])]; + tensor input_239_pad_0 = const()[name = string("input_239_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_239_dilations_0 = const()[name = string("input_239_dilations_0"), val = tensor([1, 1])]; + int32 input_239_groups_0 = const()[name = string("input_239_groups_0"), val = int32(1)]; + tensor input_239_cast_fp16 = conv(dilations = input_239_dilations_0, groups = input_239_groups_0, pad = input_239_pad_0, pad_type = input_239_pad_type_0, strides = input_239_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_237_cast_fp16)[name = string("input_239_cast_fp16")]; + tensor var_5674_cast_fp16 = silu(x = input_239_cast_fp16)[name = string("op_5674_cast_fp16")]; + string var_5680_pad_type_0 = const()[name = string("op_5680_pad_type_0"), val = string("valid")]; + tensor var_5680_strides_0 = const()[name = string("op_5680_strides_0"), val = tensor([1, 1])]; + tensor var_5680_pad_0 = const()[name = string("op_5680_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5680_dilations_0 = const()[name = string("op_5680_dilations_0"), val = tensor([1, 1])]; + int32 var_5680_groups_0 = const()[name = string("op_5680_groups_0"), val = int32(1)]; + tensor var_5680_cast_fp16 = conv(dilations = var_5680_dilations_0, groups = var_5680_groups_0, pad = var_5680_pad_0, pad_type = var_5680_pad_type_0, strides = var_5680_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_237_cast_fp16)[name = string("op_5680_cast_fp16")]; + tensor input_241_cast_fp16 = mul(x = var_5674_cast_fp16, y = var_5680_cast_fp16)[name = string("input_241_cast_fp16")]; + string h_45_pad_type_0 = const()[name = string("h_45_pad_type_0"), val = string("valid")]; + tensor h_45_strides_0 = const()[name = string("h_45_strides_0"), val = tensor([1, 1])]; + tensor h_45_pad_0 = const()[name = string("h_45_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_45_dilations_0 = const()[name = string("h_45_dilations_0"), val = tensor([1, 1])]; + int32 h_45_groups_0 = const()[name = string("h_45_groups_0"), val = int32(1)]; + tensor h_45_cast_fp16 = conv(dilations = h_45_dilations_0, groups = h_45_groups_0, pad = h_45_pad_0, pad_type = h_45_pad_type_0, strides = h_45_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_241_cast_fp16)[name = string("h_45_cast_fp16")]; + tensor x_175_cast_fp16 = add(x = x_173_cast_fp16, y = h_45_cast_fp16)[name = string("x_175_cast_fp16")]; + tensor key_cache_47_begin_0 = const()[name = string("key_cache_47_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_47_end_0 = const()[name = string("key_cache_47_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_47_end_mask_0 = const()[name = string("key_cache_47_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_47_cast_fp16 = slice_by_index(begin = key_cache_47_begin_0, end = key_cache_47_end_0, end_mask = key_cache_47_end_mask_0, x = layer_key_caches_9_cast_fp16)[name = string("key_cache_47_cast_fp16")]; + tensor value_cache_47_begin_0 = const()[name = string("value_cache_47_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_47_end_0 = const()[name = string("value_cache_47_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_47_end_mask_0 = const()[name = string("value_cache_47_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_47_cast_fp16 = slice_by_index(begin = value_cache_47_begin_0, end = value_cache_47_end_0, end_mask = value_cache_47_end_mask_0, x = layer_value_caches_9_cast_fp16)[name = string("value_cache_47_cast_fp16")]; + int32 var_5733 = const()[name = string("op_5733"), val = int32(2)]; + int32 var_5737 = const()[name = string("op_5737"), val = int32(3)]; + tensor var_5752_cast_fp16 = mul(x = x_175_cast_fp16, y = x_175_cast_fp16)[name = string("op_5752_cast_fp16")]; + tensor variance_191_axes_0 = const()[name = string("variance_191_axes_0"), val = tensor([1])]; + bool variance_191_keep_dims_0 = const()[name = string("variance_191_keep_dims_0"), val = bool(true)]; + tensor variance_191_cast_fp16 = reduce_mean(axes = variance_191_axes_0, keep_dims = variance_191_keep_dims_0, x = var_5752_cast_fp16)[name = string("variance_191_cast_fp16")]; + fp16 var_5755_to_fp16 = const()[name = string("op_5755_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5756_cast_fp16 = add(x = variance_191_cast_fp16, y = var_5755_to_fp16)[name = string("op_5756_cast_fp16")]; + fp32 var_5757_epsilon_0 = const()[name = string("op_5757_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5757_cast_fp16 = rsqrt(epsilon = var_5757_epsilon_0, x = var_5756_cast_fp16)[name = string("op_5757_cast_fp16")]; + tensor var_5758_cast_fp16 = mul(x = x_175_cast_fp16, y = var_5757_cast_fp16)[name = string("op_5758_cast_fp16")]; + tensor input_243_cast_fp16 = mul(x = var_5758_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_243_cast_fp16")]; + string q_139_pad_type_0 = const()[name = string("q_139_pad_type_0"), val = string("valid")]; + tensor q_139_strides_0 = const()[name = string("q_139_strides_0"), val = tensor([1, 1])]; + tensor q_139_pad_0 = const()[name = string("q_139_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_139_dilations_0 = const()[name = string("q_139_dilations_0"), val = tensor([1, 1])]; + int32 q_139_groups_0 = const()[name = string("q_139_groups_0"), val = int32(1)]; + tensor q_139_cast_fp16 = conv(dilations = q_139_dilations_0, groups = q_139_groups_0, pad = q_139_pad_0, pad_type = q_139_pad_type_0, strides = q_139_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_243_cast_fp16)[name = string("q_139_cast_fp16")]; + string k_139_pad_type_0 = const()[name = string("k_139_pad_type_0"), val = string("valid")]; + tensor k_139_strides_0 = const()[name = string("k_139_strides_0"), val = tensor([1, 1])]; + tensor k_139_pad_0 = const()[name = string("k_139_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_139_dilations_0 = const()[name = string("k_139_dilations_0"), val = tensor([1, 1])]; + int32 k_139_groups_0 = const()[name = string("k_139_groups_0"), val = int32(1)]; + tensor k_139_cast_fp16 = conv(dilations = k_139_dilations_0, groups = k_139_groups_0, pad = k_139_pad_0, pad_type = k_139_pad_type_0, strides = k_139_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_243_cast_fp16)[name = string("k_139_cast_fp16")]; + string v_47_pad_type_0 = const()[name = string("v_47_pad_type_0"), val = string("valid")]; + tensor v_47_strides_0 = const()[name = string("v_47_strides_0"), val = tensor([1, 1])]; + tensor v_47_pad_0 = const()[name = string("v_47_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_47_dilations_0 = const()[name = string("v_47_dilations_0"), val = tensor([1, 1])]; + int32 v_47_groups_0 = const()[name = string("v_47_groups_0"), val = int32(1)]; + tensor v_47_cast_fp16 = conv(dilations = v_47_dilations_0, groups = v_47_groups_0, pad = v_47_pad_0, pad_type = v_47_pad_type_0, strides = v_47_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_243_cast_fp16)[name = string("v_47_cast_fp16")]; + tensor var_5792 = const()[name = string("op_5792"), val = tensor([16, 128, 1, 1])]; + tensor x_177_cast_fp16 = reshape(shape = var_5792, x = q_139_cast_fp16)[name = string("x_177_cast_fp16")]; + tensor var_5795_cast_fp16 = mul(x = x_177_cast_fp16, y = x_177_cast_fp16)[name = string("op_5795_cast_fp16")]; + tensor variance_193_axes_0 = const()[name = string("variance_193_axes_0"), val = tensor([1])]; + bool variance_193_keep_dims_0 = const()[name = string("variance_193_keep_dims_0"), val = bool(true)]; + tensor variance_193_cast_fp16 = reduce_mean(axes = variance_193_axes_0, keep_dims = variance_193_keep_dims_0, x = var_5795_cast_fp16)[name = string("variance_193_cast_fp16")]; + fp16 var_5798_to_fp16 = const()[name = string("op_5798_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5799_cast_fp16 = add(x = variance_193_cast_fp16, y = var_5798_to_fp16)[name = string("op_5799_cast_fp16")]; + fp32 var_5800_epsilon_0 = const()[name = string("op_5800_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5800_cast_fp16 = rsqrt(epsilon = var_5800_epsilon_0, x = var_5799_cast_fp16)[name = string("op_5800_cast_fp16")]; + tensor var_5801_cast_fp16 = mul(x = x_177_cast_fp16, y = var_5800_cast_fp16)[name = string("op_5801_cast_fp16")]; + tensor q_141_cast_fp16 = mul(x = var_5801_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_141_cast_fp16")]; + tensor var_5803 = const()[name = string("op_5803"), val = tensor([8, 128, 1, 1])]; + tensor x_179_cast_fp16 = reshape(shape = var_5803, x = k_139_cast_fp16)[name = string("x_179_cast_fp16")]; + tensor var_5806_cast_fp16 = mul(x = x_179_cast_fp16, y = x_179_cast_fp16)[name = string("op_5806_cast_fp16")]; + tensor variance_195_axes_0 = const()[name = string("variance_195_axes_0"), val = tensor([1])]; + bool variance_195_keep_dims_0 = const()[name = string("variance_195_keep_dims_0"), val = bool(true)]; + tensor variance_195_cast_fp16 = reduce_mean(axes = variance_195_axes_0, keep_dims = variance_195_keep_dims_0, x = var_5806_cast_fp16)[name = string("variance_195_cast_fp16")]; + fp16 var_5809_to_fp16 = const()[name = string("op_5809_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5810_cast_fp16 = add(x = variance_195_cast_fp16, y = var_5809_to_fp16)[name = string("op_5810_cast_fp16")]; + fp32 var_5811_epsilon_0 = const()[name = string("op_5811_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5811_cast_fp16 = rsqrt(epsilon = var_5811_epsilon_0, x = var_5810_cast_fp16)[name = string("op_5811_cast_fp16")]; + tensor var_5812_cast_fp16 = mul(x = x_179_cast_fp16, y = var_5811_cast_fp16)[name = string("op_5812_cast_fp16")]; + tensor k_141_cast_fp16 = mul(x = var_5812_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_141_cast_fp16")]; + tensor var_5814 = const()[name = string("op_5814"), val = tensor([1, 16, 128, 1])]; + tensor z_93_cast_fp16 = reshape(shape = var_5814, x = q_141_cast_fp16)[name = string("z_93_cast_fp16")]; + tensor var_5816 = const()[name = string("op_5816"), val = tensor([1, 8, 128, 1])]; + tensor z_95_cast_fp16 = reshape(shape = var_5816, x = k_141_cast_fp16)[name = string("z_95_cast_fp16")]; + tensor z1_93_begin_0 = const()[name = string("z1_93_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_93_end_0 = const()[name = string("z1_93_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_93_end_mask_0 = const()[name = string("z1_93_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_93_cast_fp16 = slice_by_index(begin = z1_93_begin_0, end = z1_93_end_0, end_mask = z1_93_end_mask_0, x = z_93_cast_fp16)[name = string("z1_93_cast_fp16")]; + tensor z2_93_begin_0 = const()[name = string("z2_93_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_93_end_0 = const()[name = string("z2_93_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_93_end_mask_0 = const()[name = string("z2_93_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_93_cast_fp16 = slice_by_index(begin = z2_93_begin_0, end = z2_93_end_0, end_mask = z2_93_end_mask_0, x = z_93_cast_fp16)[name = string("z2_93_cast_fp16")]; + tensor var_5824_cast_fp16 = mul(x = z_93_cast_fp16, y = cos_41_to_fp16)[name = string("op_5824_cast_fp16")]; + fp16 const_51_promoted_to_fp16 = const()[name = string("const_51_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5825_cast_fp16 = mul(x = z2_93_cast_fp16, y = const_51_promoted_to_fp16)[name = string("op_5825_cast_fp16")]; + bool var_5827_interleave_0 = const()[name = string("op_5827_interleave_0"), val = bool(false)]; + tensor var_5827_cast_fp16 = concat(axis = var_5733, interleave = var_5827_interleave_0, values = (var_5825_cast_fp16, z1_93_cast_fp16))[name = string("op_5827_cast_fp16")]; + tensor var_5828_cast_fp16 = mul(x = var_5827_cast_fp16, y = sin_41_to_fp16)[name = string("op_5828_cast_fp16")]; + tensor q_143_cast_fp16 = add(x = var_5824_cast_fp16, y = var_5828_cast_fp16)[name = string("q_143_cast_fp16")]; + tensor z1_95_begin_0 = const()[name = string("z1_95_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_95_end_0 = const()[name = string("z1_95_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_95_end_mask_0 = const()[name = string("z1_95_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_95_cast_fp16 = slice_by_index(begin = z1_95_begin_0, end = z1_95_end_0, end_mask = z1_95_end_mask_0, x = z_95_cast_fp16)[name = string("z1_95_cast_fp16")]; + tensor z2_95_begin_0 = const()[name = string("z2_95_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_95_end_0 = const()[name = string("z2_95_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_95_end_mask_0 = const()[name = string("z2_95_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_95_cast_fp16 = slice_by_index(begin = z2_95_begin_0, end = z2_95_end_0, end_mask = z2_95_end_mask_0, x = z_95_cast_fp16)[name = string("z2_95_cast_fp16")]; + tensor var_5836_cast_fp16 = mul(x = z_95_cast_fp16, y = cos_41_to_fp16)[name = string("op_5836_cast_fp16")]; + fp16 const_52_promoted_to_fp16 = const()[name = string("const_52_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5837_cast_fp16 = mul(x = z2_95_cast_fp16, y = const_52_promoted_to_fp16)[name = string("op_5837_cast_fp16")]; + bool var_5839_interleave_0 = const()[name = string("op_5839_interleave_0"), val = bool(false)]; + tensor var_5839_cast_fp16 = concat(axis = var_5733, interleave = var_5839_interleave_0, values = (var_5837_cast_fp16, z1_95_cast_fp16))[name = string("op_5839_cast_fp16")]; + tensor var_5840_cast_fp16 = mul(x = var_5839_cast_fp16, y = sin_41_to_fp16)[name = string("op_5840_cast_fp16")]; + tensor k_143_cast_fp16 = add(x = var_5836_cast_fp16, y = var_5840_cast_fp16)[name = string("k_143_cast_fp16")]; + tensor var_5842 = const()[name = string("op_5842"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_47_cast_fp16 = reshape(shape = var_5842, x = k_143_cast_fp16)[name = string("cur_key_47_cast_fp16")]; + tensor var_5844_to_fp16 = const()[name = string("op_5844_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636096)))]; + tensor var_5845_cast_fp16 = mul(x = key_cache_47_cast_fp16, y = var_5844_to_fp16)[name = string("op_5845_cast_fp16")]; + tensor upd_47_to_fp16 = const()[name = string("upd_47_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636224)))]; + tensor var_5846_cast_fp16 = mul(x = cur_key_47_cast_fp16, y = upd_47_to_fp16)[name = string("op_5846_cast_fp16")]; + tensor key_47_cast_fp16 = add(x = var_5845_cast_fp16, y = var_5846_cast_fp16)[name = string("key_47_cast_fp16")]; + tensor var_5848_to_fp16 = const()[name = string("op_5848_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636096)))]; + tensor var_5849_cast_fp16 = mul(x = value_cache_47_cast_fp16, y = var_5848_to_fp16)[name = string("op_5849_cast_fp16")]; + tensor var_5850_cast_fp16 = mul(x = v_47_cast_fp16, y = upd_47_to_fp16)[name = string("op_5850_cast_fp16")]; + tensor value_47_cast_fp16 = add(x = var_5849_cast_fp16, y = var_5850_cast_fp16)[name = string("value_47_cast_fp16")]; + tensor var_5852 = const()[name = string("op_5852"), val = tensor([1, 8, 128, 16])]; + tensor kh_93_cast_fp16 = reshape(shape = var_5852, x = key_47_cast_fp16)[name = string("kh_93_cast_fp16")]; + tensor var_5854 = const()[name = string("op_5854"), val = tensor([1, 8, 128, 16])]; + tensor vh_93_cast_fp16 = reshape(shape = var_5854, x = value_47_cast_fp16)[name = string("vh_93_cast_fp16")]; + tensor transpose_92_perm_0 = const()[name = string("transpose_92_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_46_reps_0 = const()[name = string("tile_46_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_92_cast_fp16 = transpose(perm = transpose_92_perm_0, x = kh_93_cast_fp16)[name = string("transpose_341")]; + tensor tile_46_cast_fp16 = tile(reps = tile_46_reps_0, x = transpose_92_cast_fp16)[name = string("tile_46_cast_fp16")]; + tensor concat_115 = const()[name = string("concat_115"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_92_cast_fp16 = reshape(shape = concat_115, x = tile_46_cast_fp16)[name = string("reshape_92_cast_fp16")]; + tensor transpose_93_perm_0 = const()[name = string("transpose_93_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_116 = const()[name = string("concat_116"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_93_cast_fp16 = transpose(perm = transpose_93_perm_0, x = reshape_92_cast_fp16)[name = string("transpose_340")]; + tensor reshape_93_cast_fp16 = reshape(shape = concat_116, x = transpose_93_cast_fp16)[name = string("reshape_93_cast_fp16")]; + tensor transpose_94_perm_0 = const()[name = string("transpose_94_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_47_reps_0 = const()[name = string("tile_47_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_94_cast_fp16 = transpose(perm = transpose_94_perm_0, x = vh_93_cast_fp16)[name = string("transpose_339")]; + tensor tile_47_cast_fp16 = tile(reps = tile_47_reps_0, x = transpose_94_cast_fp16)[name = string("tile_47_cast_fp16")]; + tensor concat_117 = const()[name = string("concat_117"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_94_cast_fp16 = reshape(shape = concat_117, x = tile_47_cast_fp16)[name = string("reshape_94_cast_fp16")]; + tensor transpose_95_perm_0 = const()[name = string("transpose_95_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_118 = const()[name = string("concat_118"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_95_cast_fp16 = transpose(perm = transpose_95_perm_0, x = reshape_94_cast_fp16)[name = string("transpose_338")]; + tensor reshape_95_cast_fp16 = reshape(shape = concat_118, x = transpose_95_cast_fp16)[name = string("reshape_95_cast_fp16")]; + fp16 var_5858_to_fp16 = const()[name = string("op_5858_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_5859_cast_fp16 = mul(x = q_143_cast_fp16, y = var_5858_to_fp16)[name = string("op_5859_cast_fp16")]; + tensor transpose_409_perm_0 = const()[name = string("transpose_409_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_99_transpose_x_1 = const()[name = string("w_99_transpose_x_1"), val = bool(true)]; + bool w_99_transpose_y_1 = const()[name = string("w_99_transpose_y_1"), val = bool(false)]; + tensor transpose_409_cast_fp16 = transpose(perm = transpose_409_perm_0, x = reshape_93_cast_fp16)[name = string("transpose_337")]; + tensor w_99_cast_fp16 = matmul(transpose_x = w_99_transpose_x_1, transpose_y = w_99_transpose_y_1, x = var_5859_cast_fp16, y = transpose_409_cast_fp16)[name = string("w_99_cast_fp16")]; + tensor pad_47_to_fp16 = const()[name = string("pad_47_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636352)))]; + tensor var_5862_cast_fp16 = add(x = w_99_cast_fp16, y = pad_47_to_fp16)[name = string("op_5862_cast_fp16")]; + tensor w_101_cast_fp16 = softmax(axis = var_5737, x = var_5862_cast_fp16)[name = string("w_101_cast_fp16")]; + tensor transpose_410_perm_0 = const()[name = string("transpose_410_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_47_transpose_x_1 = const()[name = string("attn_47_transpose_x_1"), val = bool(false)]; + bool attn_47_transpose_y_1 = const()[name = string("attn_47_transpose_y_1"), val = bool(true)]; + tensor transpose_410_cast_fp16 = transpose(perm = transpose_410_perm_0, x = reshape_95_cast_fp16)[name = string("transpose_336")]; + tensor attn_47_cast_fp16 = matmul(transpose_x = attn_47_transpose_x_1, transpose_y = attn_47_transpose_y_1, x = transpose_410_cast_fp16, y = w_101_cast_fp16)[name = string("attn_47_cast_fp16")]; + tensor var_5866 = const()[name = string("op_5866"), val = tensor([1, 2048, 1, 1])]; + tensor input_245_cast_fp16 = reshape(shape = var_5866, x = attn_47_cast_fp16)[name = string("input_245_cast_fp16")]; + string attn_output_47_pad_type_0 = const()[name = string("attn_output_47_pad_type_0"), val = string("valid")]; + tensor attn_output_47_strides_0 = const()[name = string("attn_output_47_strides_0"), val = tensor([1, 1])]; + tensor attn_output_47_pad_0 = const()[name = string("attn_output_47_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_47_dilations_0 = const()[name = string("attn_output_47_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_47_groups_0 = const()[name = string("attn_output_47_groups_0"), val = int32(1)]; + tensor attn_output_47_cast_fp16 = conv(dilations = attn_output_47_dilations_0, groups = attn_output_47_groups_0, pad = attn_output_47_pad_0, pad_type = attn_output_47_pad_type_0, strides = attn_output_47_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_245_cast_fp16)[name = string("attn_output_47_cast_fp16")]; + tensor x_181_cast_fp16 = add(x = x_175_cast_fp16, y = attn_output_47_cast_fp16)[name = string("x_181_cast_fp16")]; + tensor var_5880_cast_fp16 = mul(x = x_181_cast_fp16, y = x_181_cast_fp16)[name = string("op_5880_cast_fp16")]; + tensor variance_197_axes_0 = const()[name = string("variance_197_axes_0"), val = tensor([1])]; + bool variance_197_keep_dims_0 = const()[name = string("variance_197_keep_dims_0"), val = bool(true)]; + tensor variance_197_cast_fp16 = reduce_mean(axes = variance_197_axes_0, keep_dims = variance_197_keep_dims_0, x = var_5880_cast_fp16)[name = string("variance_197_cast_fp16")]; + fp16 var_5883_to_fp16 = const()[name = string("op_5883_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5884_cast_fp16 = add(x = variance_197_cast_fp16, y = var_5883_to_fp16)[name = string("op_5884_cast_fp16")]; + fp32 var_5885_epsilon_0 = const()[name = string("op_5885_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5885_cast_fp16 = rsqrt(epsilon = var_5885_epsilon_0, x = var_5884_cast_fp16)[name = string("op_5885_cast_fp16")]; + tensor var_5886_cast_fp16 = mul(x = x_181_cast_fp16, y = var_5885_cast_fp16)[name = string("op_5886_cast_fp16")]; + tensor input_247_cast_fp16 = mul(x = var_5886_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_247_cast_fp16")]; + string input_249_pad_type_0 = const()[name = string("input_249_pad_type_0"), val = string("valid")]; + tensor input_249_strides_0 = const()[name = string("input_249_strides_0"), val = tensor([1, 1])]; + tensor input_249_pad_0 = const()[name = string("input_249_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_249_dilations_0 = const()[name = string("input_249_dilations_0"), val = tensor([1, 1])]; + int32 input_249_groups_0 = const()[name = string("input_249_groups_0"), val = int32(1)]; + tensor input_249_cast_fp16 = conv(dilations = input_249_dilations_0, groups = input_249_groups_0, pad = input_249_pad_0, pad_type = input_249_pad_type_0, strides = input_249_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_247_cast_fp16)[name = string("input_249_cast_fp16")]; + tensor var_5894_cast_fp16 = silu(x = input_249_cast_fp16)[name = string("op_5894_cast_fp16")]; + string var_5900_pad_type_0 = const()[name = string("op_5900_pad_type_0"), val = string("valid")]; + tensor var_5900_strides_0 = const()[name = string("op_5900_strides_0"), val = tensor([1, 1])]; + tensor var_5900_pad_0 = const()[name = string("op_5900_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5900_dilations_0 = const()[name = string("op_5900_dilations_0"), val = tensor([1, 1])]; + int32 var_5900_groups_0 = const()[name = string("op_5900_groups_0"), val = int32(1)]; + tensor var_5900_cast_fp16 = conv(dilations = var_5900_dilations_0, groups = var_5900_groups_0, pad = var_5900_pad_0, pad_type = var_5900_pad_type_0, strides = var_5900_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_247_cast_fp16)[name = string("op_5900_cast_fp16")]; + tensor input_251_cast_fp16 = mul(x = var_5894_cast_fp16, y = var_5900_cast_fp16)[name = string("input_251_cast_fp16")]; + string h_47_pad_type_0 = const()[name = string("h_47_pad_type_0"), val = string("valid")]; + tensor h_47_strides_0 = const()[name = string("h_47_strides_0"), val = tensor([1, 1])]; + tensor h_47_pad_0 = const()[name = string("h_47_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_47_dilations_0 = const()[name = string("h_47_dilations_0"), val = tensor([1, 1])]; + int32 h_47_groups_0 = const()[name = string("h_47_groups_0"), val = int32(1)]; + tensor h_47_cast_fp16 = conv(dilations = h_47_dilations_0, groups = h_47_groups_0, pad = h_47_pad_0, pad_type = h_47_pad_type_0, strides = h_47_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_251_cast_fp16)[name = string("h_47_cast_fp16")]; + tensor x_183_cast_fp16 = add(x = x_181_cast_fp16, y = h_47_cast_fp16)[name = string("x_183_cast_fp16")]; + tensor key_cache_49_begin_0 = const()[name = string("key_cache_49_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_49_end_0 = const()[name = string("key_cache_49_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_49_end_mask_0 = const()[name = string("key_cache_49_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_49_cast_fp16 = slice_by_index(begin = key_cache_49_begin_0, end = key_cache_49_end_0, end_mask = key_cache_49_end_mask_0, x = layer_key_caches_9_cast_fp16)[name = string("key_cache_49_cast_fp16")]; + tensor value_cache_49_begin_0 = const()[name = string("value_cache_49_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_49_end_0 = const()[name = string("value_cache_49_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_49_end_mask_0 = const()[name = string("value_cache_49_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_49_cast_fp16 = slice_by_index(begin = value_cache_49_begin_0, end = value_cache_49_end_0, end_mask = value_cache_49_end_mask_0, x = layer_value_caches_9_cast_fp16)[name = string("value_cache_49_cast_fp16")]; + int32 var_5953 = const()[name = string("op_5953"), val = int32(2)]; + int32 var_5957 = const()[name = string("op_5957"), val = int32(3)]; + tensor var_5972_cast_fp16 = mul(x = x_183_cast_fp16, y = x_183_cast_fp16)[name = string("op_5972_cast_fp16")]; + tensor variance_199_axes_0 = const()[name = string("variance_199_axes_0"), val = tensor([1])]; + bool variance_199_keep_dims_0 = const()[name = string("variance_199_keep_dims_0"), val = bool(true)]; + tensor variance_199_cast_fp16 = reduce_mean(axes = variance_199_axes_0, keep_dims = variance_199_keep_dims_0, x = var_5972_cast_fp16)[name = string("variance_199_cast_fp16")]; + fp16 var_5975_to_fp16 = const()[name = string("op_5975_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5976_cast_fp16 = add(x = variance_199_cast_fp16, y = var_5975_to_fp16)[name = string("op_5976_cast_fp16")]; + fp32 var_5977_epsilon_0 = const()[name = string("op_5977_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5977_cast_fp16 = rsqrt(epsilon = var_5977_epsilon_0, x = var_5976_cast_fp16)[name = string("op_5977_cast_fp16")]; + tensor var_5978_cast_fp16 = mul(x = x_183_cast_fp16, y = var_5977_cast_fp16)[name = string("op_5978_cast_fp16")]; + tensor input_253_cast_fp16 = mul(x = var_5978_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_253_cast_fp16")]; + string q_145_pad_type_0 = const()[name = string("q_145_pad_type_0"), val = string("valid")]; + tensor q_145_strides_0 = const()[name = string("q_145_strides_0"), val = tensor([1, 1])]; + tensor q_145_pad_0 = const()[name = string("q_145_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_145_dilations_0 = const()[name = string("q_145_dilations_0"), val = tensor([1, 1])]; + int32 q_145_groups_0 = const()[name = string("q_145_groups_0"), val = int32(1)]; + tensor q_145_cast_fp16 = conv(dilations = q_145_dilations_0, groups = q_145_groups_0, pad = q_145_pad_0, pad_type = q_145_pad_type_0, strides = q_145_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_253_cast_fp16)[name = string("q_145_cast_fp16")]; + string k_145_pad_type_0 = const()[name = string("k_145_pad_type_0"), val = string("valid")]; + tensor k_145_strides_0 = const()[name = string("k_145_strides_0"), val = tensor([1, 1])]; + tensor k_145_pad_0 = const()[name = string("k_145_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_145_dilations_0 = const()[name = string("k_145_dilations_0"), val = tensor([1, 1])]; + int32 k_145_groups_0 = const()[name = string("k_145_groups_0"), val = int32(1)]; + tensor k_145_cast_fp16 = conv(dilations = k_145_dilations_0, groups = k_145_groups_0, pad = k_145_pad_0, pad_type = k_145_pad_type_0, strides = k_145_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_253_cast_fp16)[name = string("k_145_cast_fp16")]; + string v_49_pad_type_0 = const()[name = string("v_49_pad_type_0"), val = string("valid")]; + tensor v_49_strides_0 = const()[name = string("v_49_strides_0"), val = tensor([1, 1])]; + tensor v_49_pad_0 = const()[name = string("v_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_49_dilations_0 = const()[name = string("v_49_dilations_0"), val = tensor([1, 1])]; + int32 v_49_groups_0 = const()[name = string("v_49_groups_0"), val = int32(1)]; + tensor v_49_cast_fp16 = conv(dilations = v_49_dilations_0, groups = v_49_groups_0, pad = v_49_pad_0, pad_type = v_49_pad_type_0, strides = v_49_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_253_cast_fp16)[name = string("v_49_cast_fp16")]; + tensor var_6012 = const()[name = string("op_6012"), val = tensor([16, 128, 1, 1])]; + tensor x_185_cast_fp16 = reshape(shape = var_6012, x = q_145_cast_fp16)[name = string("x_185_cast_fp16")]; + tensor var_6015_cast_fp16 = mul(x = x_185_cast_fp16, y = x_185_cast_fp16)[name = string("op_6015_cast_fp16")]; + tensor variance_201_axes_0 = const()[name = string("variance_201_axes_0"), val = tensor([1])]; + bool variance_201_keep_dims_0 = const()[name = string("variance_201_keep_dims_0"), val = bool(true)]; + tensor variance_201_cast_fp16 = reduce_mean(axes = variance_201_axes_0, keep_dims = variance_201_keep_dims_0, x = var_6015_cast_fp16)[name = string("variance_201_cast_fp16")]; + fp16 var_6018_to_fp16 = const()[name = string("op_6018_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6019_cast_fp16 = add(x = variance_201_cast_fp16, y = var_6018_to_fp16)[name = string("op_6019_cast_fp16")]; + fp32 var_6020_epsilon_0 = const()[name = string("op_6020_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6020_cast_fp16 = rsqrt(epsilon = var_6020_epsilon_0, x = var_6019_cast_fp16)[name = string("op_6020_cast_fp16")]; + tensor var_6021_cast_fp16 = mul(x = x_185_cast_fp16, y = var_6020_cast_fp16)[name = string("op_6021_cast_fp16")]; + tensor q_147_cast_fp16 = mul(x = var_6021_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_147_cast_fp16")]; + tensor var_6023 = const()[name = string("op_6023"), val = tensor([8, 128, 1, 1])]; + tensor x_187_cast_fp16 = reshape(shape = var_6023, x = k_145_cast_fp16)[name = string("x_187_cast_fp16")]; + tensor var_6026_cast_fp16 = mul(x = x_187_cast_fp16, y = x_187_cast_fp16)[name = string("op_6026_cast_fp16")]; + tensor variance_203_axes_0 = const()[name = string("variance_203_axes_0"), val = tensor([1])]; + bool variance_203_keep_dims_0 = const()[name = string("variance_203_keep_dims_0"), val = bool(true)]; + tensor variance_203_cast_fp16 = reduce_mean(axes = variance_203_axes_0, keep_dims = variance_203_keep_dims_0, x = var_6026_cast_fp16)[name = string("variance_203_cast_fp16")]; + fp16 var_6029_to_fp16 = const()[name = string("op_6029_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6030_cast_fp16 = add(x = variance_203_cast_fp16, y = var_6029_to_fp16)[name = string("op_6030_cast_fp16")]; + fp32 var_6031_epsilon_0 = const()[name = string("op_6031_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6031_cast_fp16 = rsqrt(epsilon = var_6031_epsilon_0, x = var_6030_cast_fp16)[name = string("op_6031_cast_fp16")]; + tensor var_6032_cast_fp16 = mul(x = x_187_cast_fp16, y = var_6031_cast_fp16)[name = string("op_6032_cast_fp16")]; + tensor k_147_cast_fp16 = mul(x = var_6032_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_147_cast_fp16")]; + tensor var_6034 = const()[name = string("op_6034"), val = tensor([1, 16, 128, 1])]; + tensor z_97_cast_fp16 = reshape(shape = var_6034, x = q_147_cast_fp16)[name = string("z_97_cast_fp16")]; + tensor var_6036 = const()[name = string("op_6036"), val = tensor([1, 8, 128, 1])]; + tensor z_99_cast_fp16 = reshape(shape = var_6036, x = k_147_cast_fp16)[name = string("z_99_cast_fp16")]; + tensor z1_97_begin_0 = const()[name = string("z1_97_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_97_end_0 = const()[name = string("z1_97_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_97_end_mask_0 = const()[name = string("z1_97_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_97_cast_fp16 = slice_by_index(begin = z1_97_begin_0, end = z1_97_end_0, end_mask = z1_97_end_mask_0, x = z_97_cast_fp16)[name = string("z1_97_cast_fp16")]; + tensor z2_97_begin_0 = const()[name = string("z2_97_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_97_end_0 = const()[name = string("z2_97_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_97_end_mask_0 = const()[name = string("z2_97_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_97_cast_fp16 = slice_by_index(begin = z2_97_begin_0, end = z2_97_end_0, end_mask = z2_97_end_mask_0, x = z_97_cast_fp16)[name = string("z2_97_cast_fp16")]; + tensor var_6044_cast_fp16 = mul(x = z_97_cast_fp16, y = cos_41_to_fp16)[name = string("op_6044_cast_fp16")]; + fp16 const_53_promoted_to_fp16 = const()[name = string("const_53_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6045_cast_fp16 = mul(x = z2_97_cast_fp16, y = const_53_promoted_to_fp16)[name = string("op_6045_cast_fp16")]; + bool var_6047_interleave_0 = const()[name = string("op_6047_interleave_0"), val = bool(false)]; + tensor var_6047_cast_fp16 = concat(axis = var_5953, interleave = var_6047_interleave_0, values = (var_6045_cast_fp16, z1_97_cast_fp16))[name = string("op_6047_cast_fp16")]; + tensor var_6048_cast_fp16 = mul(x = var_6047_cast_fp16, y = sin_41_to_fp16)[name = string("op_6048_cast_fp16")]; + tensor q_149_cast_fp16 = add(x = var_6044_cast_fp16, y = var_6048_cast_fp16)[name = string("q_149_cast_fp16")]; + tensor z1_99_begin_0 = const()[name = string("z1_99_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_99_end_0 = const()[name = string("z1_99_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_99_end_mask_0 = const()[name = string("z1_99_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_99_cast_fp16 = slice_by_index(begin = z1_99_begin_0, end = z1_99_end_0, end_mask = z1_99_end_mask_0, x = z_99_cast_fp16)[name = string("z1_99_cast_fp16")]; + tensor z2_99_begin_0 = const()[name = string("z2_99_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_99_end_0 = const()[name = string("z2_99_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_99_end_mask_0 = const()[name = string("z2_99_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_99_cast_fp16 = slice_by_index(begin = z2_99_begin_0, end = z2_99_end_0, end_mask = z2_99_end_mask_0, x = z_99_cast_fp16)[name = string("z2_99_cast_fp16")]; + tensor var_6056_cast_fp16 = mul(x = z_99_cast_fp16, y = cos_41_to_fp16)[name = string("op_6056_cast_fp16")]; + fp16 const_54_promoted_to_fp16 = const()[name = string("const_54_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6057_cast_fp16 = mul(x = z2_99_cast_fp16, y = const_54_promoted_to_fp16)[name = string("op_6057_cast_fp16")]; + bool var_6059_interleave_0 = const()[name = string("op_6059_interleave_0"), val = bool(false)]; + tensor var_6059_cast_fp16 = concat(axis = var_5953, interleave = var_6059_interleave_0, values = (var_6057_cast_fp16, z1_99_cast_fp16))[name = string("op_6059_cast_fp16")]; + tensor var_6060_cast_fp16 = mul(x = var_6059_cast_fp16, y = sin_41_to_fp16)[name = string("op_6060_cast_fp16")]; + tensor k_149_cast_fp16 = add(x = var_6056_cast_fp16, y = var_6060_cast_fp16)[name = string("k_149_cast_fp16")]; + tensor var_6062 = const()[name = string("op_6062"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_49_cast_fp16 = reshape(shape = var_6062, x = k_149_cast_fp16)[name = string("cur_key_49_cast_fp16")]; + tensor var_6064_to_fp16 = const()[name = string("op_6064_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636096)))]; + tensor var_6065_cast_fp16 = mul(x = key_cache_49_cast_fp16, y = var_6064_to_fp16)[name = string("op_6065_cast_fp16")]; + tensor upd_49_to_fp16 = const()[name = string("upd_49_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636224)))]; + tensor var_6066_cast_fp16 = mul(x = cur_key_49_cast_fp16, y = upd_49_to_fp16)[name = string("op_6066_cast_fp16")]; + tensor key_49_cast_fp16 = add(x = var_6065_cast_fp16, y = var_6066_cast_fp16)[name = string("key_49_cast_fp16")]; + tensor var_6068_to_fp16 = const()[name = string("op_6068_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636096)))]; + tensor var_6069_cast_fp16 = mul(x = value_cache_49_cast_fp16, y = var_6068_to_fp16)[name = string("op_6069_cast_fp16")]; + tensor var_6070_cast_fp16 = mul(x = v_49_cast_fp16, y = upd_49_to_fp16)[name = string("op_6070_cast_fp16")]; + tensor value_49_cast_fp16 = add(x = var_6069_cast_fp16, y = var_6070_cast_fp16)[name = string("value_49_cast_fp16")]; + tensor var_6072 = const()[name = string("op_6072"), val = tensor([1, 8, 128, 16])]; + tensor kh_97_cast_fp16 = reshape(shape = var_6072, x = key_49_cast_fp16)[name = string("kh_97_cast_fp16")]; + tensor var_6074 = const()[name = string("op_6074"), val = tensor([1, 8, 128, 16])]; + tensor vh_97_cast_fp16 = reshape(shape = var_6074, x = value_49_cast_fp16)[name = string("vh_97_cast_fp16")]; + tensor transpose_96_perm_0 = const()[name = string("transpose_96_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_48_reps_0 = const()[name = string("tile_48_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_96_cast_fp16 = transpose(perm = transpose_96_perm_0, x = kh_97_cast_fp16)[name = string("transpose_335")]; + tensor tile_48_cast_fp16 = tile(reps = tile_48_reps_0, x = transpose_96_cast_fp16)[name = string("tile_48_cast_fp16")]; + tensor concat_119 = const()[name = string("concat_119"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_96_cast_fp16 = reshape(shape = concat_119, x = tile_48_cast_fp16)[name = string("reshape_96_cast_fp16")]; + tensor transpose_97_perm_0 = const()[name = string("transpose_97_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_120 = const()[name = string("concat_120"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_97_cast_fp16 = transpose(perm = transpose_97_perm_0, x = reshape_96_cast_fp16)[name = string("transpose_334")]; + tensor reshape_97_cast_fp16 = reshape(shape = concat_120, x = transpose_97_cast_fp16)[name = string("reshape_97_cast_fp16")]; + tensor transpose_98_perm_0 = const()[name = string("transpose_98_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_49_reps_0 = const()[name = string("tile_49_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_98_cast_fp16 = transpose(perm = transpose_98_perm_0, x = vh_97_cast_fp16)[name = string("transpose_333")]; + tensor tile_49_cast_fp16 = tile(reps = tile_49_reps_0, x = transpose_98_cast_fp16)[name = string("tile_49_cast_fp16")]; + tensor concat_121 = const()[name = string("concat_121"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_98_cast_fp16 = reshape(shape = concat_121, x = tile_49_cast_fp16)[name = string("reshape_98_cast_fp16")]; + tensor transpose_99_perm_0 = const()[name = string("transpose_99_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_122 = const()[name = string("concat_122"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_99_cast_fp16 = transpose(perm = transpose_99_perm_0, x = reshape_98_cast_fp16)[name = string("transpose_332")]; + tensor reshape_99_cast_fp16 = reshape(shape = concat_122, x = transpose_99_cast_fp16)[name = string("reshape_99_cast_fp16")]; + fp16 var_6078_to_fp16 = const()[name = string("op_6078_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_6079_cast_fp16 = mul(x = q_149_cast_fp16, y = var_6078_to_fp16)[name = string("op_6079_cast_fp16")]; + tensor transpose_413_perm_0 = const()[name = string("transpose_413_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_103_transpose_x_1 = const()[name = string("w_103_transpose_x_1"), val = bool(true)]; + bool w_103_transpose_y_1 = const()[name = string("w_103_transpose_y_1"), val = bool(false)]; + tensor transpose_413_cast_fp16 = transpose(perm = transpose_413_perm_0, x = reshape_97_cast_fp16)[name = string("transpose_331")]; + tensor w_103_cast_fp16 = matmul(transpose_x = w_103_transpose_x_1, transpose_y = w_103_transpose_y_1, x = var_6079_cast_fp16, y = transpose_413_cast_fp16)[name = string("w_103_cast_fp16")]; + tensor pad_49_to_fp16 = const()[name = string("pad_49_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636352)))]; + tensor var_6082_cast_fp16 = add(x = w_103_cast_fp16, y = pad_49_to_fp16)[name = string("op_6082_cast_fp16")]; + tensor w_105_cast_fp16 = softmax(axis = var_5957, x = var_6082_cast_fp16)[name = string("w_105_cast_fp16")]; + tensor transpose_414_perm_0 = const()[name = string("transpose_414_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_49_transpose_x_1 = const()[name = string("attn_49_transpose_x_1"), val = bool(false)]; + bool attn_49_transpose_y_1 = const()[name = string("attn_49_transpose_y_1"), val = bool(true)]; + tensor transpose_414_cast_fp16 = transpose(perm = transpose_414_perm_0, x = reshape_99_cast_fp16)[name = string("transpose_330")]; + tensor attn_49_cast_fp16 = matmul(transpose_x = attn_49_transpose_x_1, transpose_y = attn_49_transpose_y_1, x = transpose_414_cast_fp16, y = w_105_cast_fp16)[name = string("attn_49_cast_fp16")]; + tensor var_6086 = const()[name = string("op_6086"), val = tensor([1, 2048, 1, 1])]; + tensor input_255_cast_fp16 = reshape(shape = var_6086, x = attn_49_cast_fp16)[name = string("input_255_cast_fp16")]; + string attn_output_49_pad_type_0 = const()[name = string("attn_output_49_pad_type_0"), val = string("valid")]; + tensor attn_output_49_strides_0 = const()[name = string("attn_output_49_strides_0"), val = tensor([1, 1])]; + tensor attn_output_49_pad_0 = const()[name = string("attn_output_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_49_dilations_0 = const()[name = string("attn_output_49_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_49_groups_0 = const()[name = string("attn_output_49_groups_0"), val = int32(1)]; + tensor attn_output_49_cast_fp16 = conv(dilations = attn_output_49_dilations_0, groups = attn_output_49_groups_0, pad = attn_output_49_pad_0, pad_type = attn_output_49_pad_type_0, strides = attn_output_49_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_255_cast_fp16)[name = string("attn_output_49_cast_fp16")]; + tensor x_189_cast_fp16 = add(x = x_183_cast_fp16, y = attn_output_49_cast_fp16)[name = string("x_189_cast_fp16")]; + tensor var_6100_cast_fp16 = mul(x = x_189_cast_fp16, y = x_189_cast_fp16)[name = string("op_6100_cast_fp16")]; + tensor variance_205_axes_0 = const()[name = string("variance_205_axes_0"), val = tensor([1])]; + bool variance_205_keep_dims_0 = const()[name = string("variance_205_keep_dims_0"), val = bool(true)]; + tensor variance_205_cast_fp16 = reduce_mean(axes = variance_205_axes_0, keep_dims = variance_205_keep_dims_0, x = var_6100_cast_fp16)[name = string("variance_205_cast_fp16")]; + fp16 var_6103_to_fp16 = const()[name = string("op_6103_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6104_cast_fp16 = add(x = variance_205_cast_fp16, y = var_6103_to_fp16)[name = string("op_6104_cast_fp16")]; + fp32 var_6105_epsilon_0 = const()[name = string("op_6105_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6105_cast_fp16 = rsqrt(epsilon = var_6105_epsilon_0, x = var_6104_cast_fp16)[name = string("op_6105_cast_fp16")]; + tensor var_6106_cast_fp16 = mul(x = x_189_cast_fp16, y = var_6105_cast_fp16)[name = string("op_6106_cast_fp16")]; + tensor input_257_cast_fp16 = mul(x = var_6106_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_257_cast_fp16")]; + string input_259_pad_type_0 = const()[name = string("input_259_pad_type_0"), val = string("valid")]; + tensor input_259_strides_0 = const()[name = string("input_259_strides_0"), val = tensor([1, 1])]; + tensor input_259_pad_0 = const()[name = string("input_259_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_259_dilations_0 = const()[name = string("input_259_dilations_0"), val = tensor([1, 1])]; + int32 input_259_groups_0 = const()[name = string("input_259_groups_0"), val = int32(1)]; + tensor input_259_cast_fp16 = conv(dilations = input_259_dilations_0, groups = input_259_groups_0, pad = input_259_pad_0, pad_type = input_259_pad_type_0, strides = input_259_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_257_cast_fp16)[name = string("input_259_cast_fp16")]; + tensor var_6114_cast_fp16 = silu(x = input_259_cast_fp16)[name = string("op_6114_cast_fp16")]; + string var_6120_pad_type_0 = const()[name = string("op_6120_pad_type_0"), val = string("valid")]; + tensor var_6120_strides_0 = const()[name = string("op_6120_strides_0"), val = tensor([1, 1])]; + tensor var_6120_pad_0 = const()[name = string("op_6120_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6120_dilations_0 = const()[name = string("op_6120_dilations_0"), val = tensor([1, 1])]; + int32 var_6120_groups_0 = const()[name = string("op_6120_groups_0"), val = int32(1)]; + tensor var_6120_cast_fp16 = conv(dilations = var_6120_dilations_0, groups = var_6120_groups_0, pad = var_6120_pad_0, pad_type = var_6120_pad_type_0, strides = var_6120_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_257_cast_fp16)[name = string("op_6120_cast_fp16")]; + tensor input_261_cast_fp16 = mul(x = var_6114_cast_fp16, y = var_6120_cast_fp16)[name = string("input_261_cast_fp16")]; + string h_49_pad_type_0 = const()[name = string("h_49_pad_type_0"), val = string("valid")]; + tensor h_49_strides_0 = const()[name = string("h_49_strides_0"), val = tensor([1, 1])]; + tensor h_49_pad_0 = const()[name = string("h_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_49_dilations_0 = const()[name = string("h_49_dilations_0"), val = tensor([1, 1])]; + int32 h_49_groups_0 = const()[name = string("h_49_groups_0"), val = int32(1)]; + tensor h_49_cast_fp16 = conv(dilations = h_49_dilations_0, groups = h_49_groups_0, pad = h_49_pad_0, pad_type = h_49_pad_type_0, strides = h_49_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_261_cast_fp16)[name = string("h_49_cast_fp16")]; + tensor inputs_7_cast_fp16 = add(x = x_189_cast_fp16, y = h_49_cast_fp16)[name = string("inputs_7_cast_fp16")]; + int32 var_6148 = const()[name = string("op_6148"), val = int32(1)]; + bool layer_key_caches_11_interleave_0 = const()[name = string("layer_key_caches_11_interleave_0"), val = bool(false)]; + tensor layer_key_caches_11_cast_fp16 = concat(axis = var_6148, interleave = layer_key_caches_11_interleave_0, values = (key_41_cast_fp16, key_43_cast_fp16, key_45_cast_fp16, key_47_cast_fp16, key_49_cast_fp16))[name = string("layer_key_caches_11_cast_fp16")]; + int32 var_6151 = const()[name = string("op_6151"), val = int32(1)]; + bool layer_value_caches_11_interleave_0 = const()[name = string("layer_value_caches_11_interleave_0"), val = bool(false)]; + tensor layer_value_caches_11_cast_fp16 = concat(axis = var_6151, interleave = layer_value_caches_11_interleave_0, values = (value_41_cast_fp16, value_43_cast_fp16, value_45_cast_fp16, value_47_cast_fp16, value_49_cast_fp16))[name = string("layer_value_caches_11_cast_fp16")]; + tensor inputs_sq_7_cast_fp16 = mul(x = inputs_7_cast_fp16, y = inputs_7_cast_fp16)[name = string("inputs_sq_7_cast_fp16")]; + tensor variance_207_axes_0 = const()[name = string("variance_207_axes_0"), val = tensor([1])]; + bool variance_207_keep_dims_0 = const()[name = string("variance_207_keep_dims_0"), val = bool(true)]; + tensor variance_207_cast_fp16 = reduce_mean(axes = variance_207_axes_0, keep_dims = variance_207_keep_dims_0, x = inputs_sq_7_cast_fp16)[name = string("variance_207_cast_fp16")]; + fp16 var_6161_to_fp16 = const()[name = string("op_6161_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6162_cast_fp16 = add(x = variance_207_cast_fp16, y = var_6161_to_fp16)[name = string("op_6162_cast_fp16")]; + fp32 var_6163_epsilon_0 = const()[name = string("op_6163_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6163_cast_fp16 = rsqrt(epsilon = var_6163_epsilon_0, x = var_6162_cast_fp16)[name = string("op_6163_cast_fp16")]; + tensor hidden_states_7_cast_fp16 = mul(x = inputs_7_cast_fp16, y = var_6163_cast_fp16)[name = string("hidden_states_7_cast_fp16")]; + tensor input_263_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_7_cast_fp16)[name = string("input_263_cast_fp16")]; + string logits_13_pad_type_0 = const()[name = string("logits_13_pad_type_0"), val = string("valid")]; + tensor logits_13_strides_0 = const()[name = string("logits_13_strides_0"), val = tensor([1, 1])]; + tensor logits_13_pad_0 = const()[name = string("logits_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_13_dilations_0 = const()[name = string("logits_13_dilations_0"), val = tensor([1, 1])]; + int32 logits_13_groups_0 = const()[name = string("logits_13_groups_0"), val = int32(1)]; + tensor lm_heads_3_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85000064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87097280))))[name = string("lm_heads_3_weight_to_fp16_palettized")]; + tensor logits_13_cast_fp16 = conv(dilations = logits_13_dilations_0, groups = logits_13_groups_0, pad = logits_13_pad_0, pad_type = logits_13_pad_type_0, strides = logits_13_strides_0, weight = lm_heads_3_weight_to_fp16_palettized, x = input_263_cast_fp16)[name = string("logits_13_cast_fp16")]; + tensor var_6181 = const()[name = string("op_6181"), val = tensor([1, 2048])]; + tensor logits_15_cast_fp16 = reshape(shape = var_6181, x = logits_13_cast_fp16)[name = string("logits_15_cast_fp16")]; + tensor scaled_logits_7_cast_fp16 = real_div(x = logits_15_cast_fp16, y = temperature)[name = string("scaled_logits_7_cast_fp16")]; + int32 var_6191 = const()[name = string("op_6191"), val = int32(100)]; + int32 top_values_7_axis_0 = const()[name = string("top_values_7_axis_0"), val = int32(1)]; + bool top_values_7_ascending_0 = const()[name = string("top_values_7_ascending_0"), val = bool(false)]; + bool top_values_7_sort_0 = const()[name = string("top_values_7_sort_0"), val = bool(true)]; + bool top_values_7_return_indices_0 = const()[name = string("top_values_7_return_indices_0"), val = bool(true)]; + string top_values_7_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_7_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_7_cast_fp16_cast_uint16_0, tensor top_values_7_cast_fp16_cast_uint16_1 = topk(ascending = top_values_7_ascending_0, axis = top_values_7_axis_0, k = var_6191, output_indices_dtype = top_values_7_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_7_return_indices_0, sort = top_values_7_sort_0, x = scaled_logits_7_cast_fp16)[name = string("top_values_7_cast_fp16_cast_uint16")]; + tensor var_6197_cast_fp16 = mul(x = top_values_7_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_6197_cast_fp16")]; + tensor var_6201_cast_fp16 = add(x = var_6197_cast_fp16, y = var_2438_cast_fp16)[name = string("op_6201_cast_fp16")]; + tensor reduce_min_3_axes_0 = const()[name = string("reduce_min_3_axes_0"), val = tensor([1])]; + bool reduce_min_3_keep_dims_0 = const()[name = string("reduce_min_3_keep_dims_0"), val = bool(true)]; + tensor reduce_min_3_cast_fp16 = reduce_min(axes = reduce_min_3_axes_0, keep_dims = reduce_min_3_keep_dims_0, x = var_6201_cast_fp16)[name = string("reduce_min_3_cast_fp16")]; + tensor var_6204_cast_fp16 = greater_equal(x = scaled_logits_7_cast_fp16, y = reduce_min_3_cast_fp16)[name = string("op_6204_cast_fp16")]; + fp16 var_6205_value_0_to_fp16 = const()[name = string("op_6205_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_6205_cast_fp16 = fill_like(ref_tensor = scaled_logits_7_cast_fp16, value = var_6205_value_0_to_fp16)[name = string("op_6205_cast_fp16")]; + tensor masked_logits_7_cast_fp16 = select(a = scaled_logits_7_cast_fp16, b = var_6205_cast_fp16, cond = var_6204_cast_fp16)[name = string("masked_logits_7_cast_fp16")]; + tensor var_6209_begin_0 = const()[name = string("op_6209_begin_0"), val = tensor([3, 0])]; + tensor var_6209_end_0 = const()[name = string("op_6209_end_0"), val = tensor([4, 2048])]; + tensor var_6209_end_mask_0 = const()[name = string("op_6209_end_mask_0"), val = tensor([false, true])]; + tensor var_6209_squeeze_mask_0 = const()[name = string("op_6209_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_6209_cast_fp16 = slice_by_index(begin = var_6209_begin_0, end = var_6209_end_0, end_mask = var_6209_end_mask_0, squeeze_mask = var_6209_squeeze_mask_0, x = gumbel)[name = string("op_6209_cast_fp16")]; + tensor var_6212 = const()[name = string("op_6212"), val = tensor([1, 2048])]; + tensor var_6213_cast_fp16 = reshape(shape = var_6212, x = var_6209_cast_fp16)[name = string("op_6213_cast_fp16")]; + tensor noisy_logits_7_cast_fp16 = add(x = masked_logits_7_cast_fp16, y = var_6213_cast_fp16)[name = string("noisy_logits_7_cast_fp16")]; + int32 code_7_axis_0 = const()[name = string("code_7_axis_0"), val = int32(1)]; + bool code_7_keep_dims_0 = const()[name = string("code_7_keep_dims_0"), val = bool(false)]; + string code_7_output_dtype_0 = const()[name = string("code_7_output_dtype_0"), val = string("int32")]; + tensor code_7_cast_fp16 = reduce_argmax(axis = code_7_axis_0, keep_dims = code_7_keep_dims_0, output_dtype = code_7_output_dtype_0, x = noisy_logits_7_cast_fp16)[name = string("code_7_cast_fp16")]; + int32 var_6224 = const()[name = string("op_6224"), val = int32(6144)]; + tensor input_265 = add(x = code_7_cast_fp16, y = var_6224)[name = string("input_265")]; + int32 code_embed_13_axis_0 = const()[name = string("code_embed_13_axis_0"), val = int32(0)]; + int32 code_embed_13_batch_dims_0 = const()[name = string("code_embed_13_batch_dims_0"), val = int32(0)]; + bool code_embed_13_validate_indices_0 = const()[name = string("code_embed_13_validate_indices_0"), val = bool(false)]; + string input_265_to_uint16_dtype_0 = const()[name = string("input_265_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_265_to_uint16 = cast(dtype = input_265_to_uint16_dtype_0, x = input_265)[name = string("cast_11")]; + tensor code_embed_13_cast_fp16_cast_uint16 = gather(axis = code_embed_13_axis_0, batch_dims = code_embed_13_batch_dims_0, indices = input_265_to_uint16, validate_indices = code_embed_13_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_13_cast_fp16_cast_uint16")]; + tensor var_6228 = const()[name = string("op_6228"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_15_cast_fp16 = reshape(shape = var_6228, x = code_embed_13_cast_fp16_cast_uint16)[name = string("code_embed_15_cast_fp16")]; + tensor embed_sum_9_cast_fp16 = add(x = embed_sum_7_cast_fp16, y = code_embed_15_cast_fp16)[name = string("embed_sum_9_cast_fp16")]; + tensor key_cache_51_begin_0 = const()[name = string("key_cache_51_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_51_end_0 = const()[name = string("key_cache_51_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_51_end_mask_0 = const()[name = string("key_cache_51_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_51_cast_fp16 = slice_by_index(begin = key_cache_51_begin_0, end = key_cache_51_end_0, end_mask = key_cache_51_end_mask_0, x = layer_key_caches_11_cast_fp16)[name = string("key_cache_51_cast_fp16")]; + tensor value_cache_51_begin_0 = const()[name = string("value_cache_51_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_51_end_0 = const()[name = string("value_cache_51_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_51_end_mask_0 = const()[name = string("value_cache_51_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_51_cast_fp16 = slice_by_index(begin = value_cache_51_begin_0, end = value_cache_51_end_0, end_mask = value_cache_51_end_mask_0, x = layer_value_caches_11_cast_fp16)[name = string("value_cache_51_cast_fp16")]; + int32 var_6327 = const()[name = string("op_6327"), val = int32(2)]; + int32 var_6331 = const()[name = string("op_6331"), val = int32(3)]; + tensor var_6346_cast_fp16 = mul(x = code_embed_15_cast_fp16, y = code_embed_15_cast_fp16)[name = string("op_6346_cast_fp16")]; + tensor variance_209_axes_0 = const()[name = string("variance_209_axes_0"), val = tensor([1])]; + bool variance_209_keep_dims_0 = const()[name = string("variance_209_keep_dims_0"), val = bool(true)]; + tensor variance_209_cast_fp16 = reduce_mean(axes = variance_209_axes_0, keep_dims = variance_209_keep_dims_0, x = var_6346_cast_fp16)[name = string("variance_209_cast_fp16")]; + fp16 var_6349_to_fp16 = const()[name = string("op_6349_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6350_cast_fp16 = add(x = variance_209_cast_fp16, y = var_6349_to_fp16)[name = string("op_6350_cast_fp16")]; + fp32 var_6351_epsilon_0 = const()[name = string("op_6351_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6351_cast_fp16 = rsqrt(epsilon = var_6351_epsilon_0, x = var_6350_cast_fp16)[name = string("op_6351_cast_fp16")]; + tensor var_6352_cast_fp16 = mul(x = code_embed_15_cast_fp16, y = var_6351_cast_fp16)[name = string("op_6352_cast_fp16")]; + tensor input_267_cast_fp16 = mul(x = var_6352_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_267_cast_fp16")]; + string q_151_pad_type_0 = const()[name = string("q_151_pad_type_0"), val = string("valid")]; + tensor q_151_strides_0 = const()[name = string("q_151_strides_0"), val = tensor([1, 1])]; + tensor q_151_pad_0 = const()[name = string("q_151_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_151_dilations_0 = const()[name = string("q_151_dilations_0"), val = tensor([1, 1])]; + int32 q_151_groups_0 = const()[name = string("q_151_groups_0"), val = int32(1)]; + tensor q_151_cast_fp16 = conv(dilations = q_151_dilations_0, groups = q_151_groups_0, pad = q_151_pad_0, pad_type = q_151_pad_type_0, strides = q_151_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_267_cast_fp16)[name = string("q_151_cast_fp16")]; + string k_151_pad_type_0 = const()[name = string("k_151_pad_type_0"), val = string("valid")]; + tensor k_151_strides_0 = const()[name = string("k_151_strides_0"), val = tensor([1, 1])]; + tensor k_151_pad_0 = const()[name = string("k_151_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_151_dilations_0 = const()[name = string("k_151_dilations_0"), val = tensor([1, 1])]; + int32 k_151_groups_0 = const()[name = string("k_151_groups_0"), val = int32(1)]; + tensor k_151_cast_fp16 = conv(dilations = k_151_dilations_0, groups = k_151_groups_0, pad = k_151_pad_0, pad_type = k_151_pad_type_0, strides = k_151_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_267_cast_fp16)[name = string("k_151_cast_fp16")]; + string v_51_pad_type_0 = const()[name = string("v_51_pad_type_0"), val = string("valid")]; + tensor v_51_strides_0 = const()[name = string("v_51_strides_0"), val = tensor([1, 1])]; + tensor v_51_pad_0 = const()[name = string("v_51_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_51_dilations_0 = const()[name = string("v_51_dilations_0"), val = tensor([1, 1])]; + int32 v_51_groups_0 = const()[name = string("v_51_groups_0"), val = int32(1)]; + tensor v_51_cast_fp16 = conv(dilations = v_51_dilations_0, groups = v_51_groups_0, pad = v_51_pad_0, pad_type = v_51_pad_type_0, strides = v_51_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_267_cast_fp16)[name = string("v_51_cast_fp16")]; + tensor var_6386 = const()[name = string("op_6386"), val = tensor([16, 128, 1, 1])]; + tensor x_191_cast_fp16 = reshape(shape = var_6386, x = q_151_cast_fp16)[name = string("x_191_cast_fp16")]; + tensor var_6389_cast_fp16 = mul(x = x_191_cast_fp16, y = x_191_cast_fp16)[name = string("op_6389_cast_fp16")]; + tensor variance_211_axes_0 = const()[name = string("variance_211_axes_0"), val = tensor([1])]; + bool variance_211_keep_dims_0 = const()[name = string("variance_211_keep_dims_0"), val = bool(true)]; + tensor variance_211_cast_fp16 = reduce_mean(axes = variance_211_axes_0, keep_dims = variance_211_keep_dims_0, x = var_6389_cast_fp16)[name = string("variance_211_cast_fp16")]; + fp16 var_6392_to_fp16 = const()[name = string("op_6392_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6393_cast_fp16 = add(x = variance_211_cast_fp16, y = var_6392_to_fp16)[name = string("op_6393_cast_fp16")]; + fp32 var_6394_epsilon_0 = const()[name = string("op_6394_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6394_cast_fp16 = rsqrt(epsilon = var_6394_epsilon_0, x = var_6393_cast_fp16)[name = string("op_6394_cast_fp16")]; + tensor var_6395_cast_fp16 = mul(x = x_191_cast_fp16, y = var_6394_cast_fp16)[name = string("op_6395_cast_fp16")]; + tensor q_153_cast_fp16 = mul(x = var_6395_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_153_cast_fp16")]; + tensor var_6397 = const()[name = string("op_6397"), val = tensor([8, 128, 1, 1])]; + tensor x_193_cast_fp16 = reshape(shape = var_6397, x = k_151_cast_fp16)[name = string("x_193_cast_fp16")]; + tensor var_6400_cast_fp16 = mul(x = x_193_cast_fp16, y = x_193_cast_fp16)[name = string("op_6400_cast_fp16")]; + tensor variance_213_axes_0 = const()[name = string("variance_213_axes_0"), val = tensor([1])]; + bool variance_213_keep_dims_0 = const()[name = string("variance_213_keep_dims_0"), val = bool(true)]; + tensor variance_213_cast_fp16 = reduce_mean(axes = variance_213_axes_0, keep_dims = variance_213_keep_dims_0, x = var_6400_cast_fp16)[name = string("variance_213_cast_fp16")]; + fp16 var_6403_to_fp16 = const()[name = string("op_6403_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6404_cast_fp16 = add(x = variance_213_cast_fp16, y = var_6403_to_fp16)[name = string("op_6404_cast_fp16")]; + fp32 var_6405_epsilon_0 = const()[name = string("op_6405_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6405_cast_fp16 = rsqrt(epsilon = var_6405_epsilon_0, x = var_6404_cast_fp16)[name = string("op_6405_cast_fp16")]; + tensor var_6406_cast_fp16 = mul(x = x_193_cast_fp16, y = var_6405_cast_fp16)[name = string("op_6406_cast_fp16")]; + tensor k_153_cast_fp16 = mul(x = var_6406_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_153_cast_fp16")]; + tensor var_6408 = const()[name = string("op_6408"), val = tensor([1, 16, 128, 1])]; + tensor z_101_cast_fp16 = reshape(shape = var_6408, x = q_153_cast_fp16)[name = string("z_101_cast_fp16")]; + tensor var_6410 = const()[name = string("op_6410"), val = tensor([1, 8, 128, 1])]; + tensor z_103_cast_fp16 = reshape(shape = var_6410, x = k_153_cast_fp16)[name = string("z_103_cast_fp16")]; + tensor z1_101_begin_0 = const()[name = string("z1_101_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_101_end_0 = const()[name = string("z1_101_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_101_end_mask_0 = const()[name = string("z1_101_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_101_cast_fp16 = slice_by_index(begin = z1_101_begin_0, end = z1_101_end_0, end_mask = z1_101_end_mask_0, x = z_101_cast_fp16)[name = string("z1_101_cast_fp16")]; + tensor z2_101_begin_0 = const()[name = string("z2_101_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_101_end_0 = const()[name = string("z2_101_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_101_end_mask_0 = const()[name = string("z2_101_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_101_cast_fp16 = slice_by_index(begin = z2_101_begin_0, end = z2_101_end_0, end_mask = z2_101_end_mask_0, x = z_101_cast_fp16)[name = string("z2_101_cast_fp16")]; + tensor cos_51_to_fp16 = const()[name = string("cos_51_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636480)))]; + tensor var_6418_cast_fp16 = mul(x = z_101_cast_fp16, y = cos_51_to_fp16)[name = string("op_6418_cast_fp16")]; + fp16 const_56_promoted_to_fp16 = const()[name = string("const_56_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6419_cast_fp16 = mul(x = z2_101_cast_fp16, y = const_56_promoted_to_fp16)[name = string("op_6419_cast_fp16")]; + bool var_6421_interleave_0 = const()[name = string("op_6421_interleave_0"), val = bool(false)]; + tensor var_6421_cast_fp16 = concat(axis = var_6327, interleave = var_6421_interleave_0, values = (var_6419_cast_fp16, z1_101_cast_fp16))[name = string("op_6421_cast_fp16")]; + tensor sin_51_to_fp16 = const()[name = string("sin_51_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141636800)))]; + tensor var_6422_cast_fp16 = mul(x = var_6421_cast_fp16, y = sin_51_to_fp16)[name = string("op_6422_cast_fp16")]; + tensor q_155_cast_fp16 = add(x = var_6418_cast_fp16, y = var_6422_cast_fp16)[name = string("q_155_cast_fp16")]; + tensor z1_103_begin_0 = const()[name = string("z1_103_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_103_end_0 = const()[name = string("z1_103_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_103_end_mask_0 = const()[name = string("z1_103_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_103_cast_fp16 = slice_by_index(begin = z1_103_begin_0, end = z1_103_end_0, end_mask = z1_103_end_mask_0, x = z_103_cast_fp16)[name = string("z1_103_cast_fp16")]; + tensor z2_103_begin_0 = const()[name = string("z2_103_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_103_end_0 = const()[name = string("z2_103_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_103_end_mask_0 = const()[name = string("z2_103_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_103_cast_fp16 = slice_by_index(begin = z2_103_begin_0, end = z2_103_end_0, end_mask = z2_103_end_mask_0, x = z_103_cast_fp16)[name = string("z2_103_cast_fp16")]; + tensor var_6430_cast_fp16 = mul(x = z_103_cast_fp16, y = cos_51_to_fp16)[name = string("op_6430_cast_fp16")]; + fp16 const_57_promoted_to_fp16 = const()[name = string("const_57_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6431_cast_fp16 = mul(x = z2_103_cast_fp16, y = const_57_promoted_to_fp16)[name = string("op_6431_cast_fp16")]; + bool var_6433_interleave_0 = const()[name = string("op_6433_interleave_0"), val = bool(false)]; + tensor var_6433_cast_fp16 = concat(axis = var_6327, interleave = var_6433_interleave_0, values = (var_6431_cast_fp16, z1_103_cast_fp16))[name = string("op_6433_cast_fp16")]; + tensor var_6434_cast_fp16 = mul(x = var_6433_cast_fp16, y = sin_51_to_fp16)[name = string("op_6434_cast_fp16")]; + tensor k_155_cast_fp16 = add(x = var_6430_cast_fp16, y = var_6434_cast_fp16)[name = string("k_155_cast_fp16")]; + tensor var_6436 = const()[name = string("op_6436"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_51_cast_fp16 = reshape(shape = var_6436, x = k_155_cast_fp16)[name = string("cur_key_51_cast_fp16")]; + tensor var_6438_to_fp16 = const()[name = string("op_6438_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637120)))]; + tensor var_6439_cast_fp16 = mul(x = key_cache_51_cast_fp16, y = var_6438_to_fp16)[name = string("op_6439_cast_fp16")]; + tensor upd_51_to_fp16 = const()[name = string("upd_51_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637248)))]; + tensor var_6440_cast_fp16 = mul(x = cur_key_51_cast_fp16, y = upd_51_to_fp16)[name = string("op_6440_cast_fp16")]; + tensor key_51_cast_fp16 = add(x = var_6439_cast_fp16, y = var_6440_cast_fp16)[name = string("key_51_cast_fp16")]; + tensor var_6442_to_fp16 = const()[name = string("op_6442_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637120)))]; + tensor var_6443_cast_fp16 = mul(x = value_cache_51_cast_fp16, y = var_6442_to_fp16)[name = string("op_6443_cast_fp16")]; + tensor var_6444_cast_fp16 = mul(x = v_51_cast_fp16, y = upd_51_to_fp16)[name = string("op_6444_cast_fp16")]; + tensor value_51_cast_fp16 = add(x = var_6443_cast_fp16, y = var_6444_cast_fp16)[name = string("value_51_cast_fp16")]; + tensor var_6446 = const()[name = string("op_6446"), val = tensor([1, 8, 128, 16])]; + tensor kh_101_cast_fp16 = reshape(shape = var_6446, x = key_51_cast_fp16)[name = string("kh_101_cast_fp16")]; + tensor var_6448 = const()[name = string("op_6448"), val = tensor([1, 8, 128, 16])]; + tensor vh_101_cast_fp16 = reshape(shape = var_6448, x = value_51_cast_fp16)[name = string("vh_101_cast_fp16")]; + tensor transpose_100_perm_0 = const()[name = string("transpose_100_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_50_reps_0 = const()[name = string("tile_50_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_100_cast_fp16 = transpose(perm = transpose_100_perm_0, x = kh_101_cast_fp16)[name = string("transpose_329")]; + tensor tile_50_cast_fp16 = tile(reps = tile_50_reps_0, x = transpose_100_cast_fp16)[name = string("tile_50_cast_fp16")]; + tensor concat_128 = const()[name = string("concat_128"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_100_cast_fp16 = reshape(shape = concat_128, x = tile_50_cast_fp16)[name = string("reshape_100_cast_fp16")]; + tensor transpose_101_perm_0 = const()[name = string("transpose_101_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_129 = const()[name = string("concat_129"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_101_cast_fp16 = transpose(perm = transpose_101_perm_0, x = reshape_100_cast_fp16)[name = string("transpose_328")]; + tensor reshape_101_cast_fp16 = reshape(shape = concat_129, x = transpose_101_cast_fp16)[name = string("reshape_101_cast_fp16")]; + tensor transpose_102_perm_0 = const()[name = string("transpose_102_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_51_reps_0 = const()[name = string("tile_51_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_102_cast_fp16 = transpose(perm = transpose_102_perm_0, x = vh_101_cast_fp16)[name = string("transpose_327")]; + tensor tile_51_cast_fp16 = tile(reps = tile_51_reps_0, x = transpose_102_cast_fp16)[name = string("tile_51_cast_fp16")]; + tensor concat_130 = const()[name = string("concat_130"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_102_cast_fp16 = reshape(shape = concat_130, x = tile_51_cast_fp16)[name = string("reshape_102_cast_fp16")]; + tensor transpose_103_perm_0 = const()[name = string("transpose_103_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_131 = const()[name = string("concat_131"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_103_cast_fp16 = transpose(perm = transpose_103_perm_0, x = reshape_102_cast_fp16)[name = string("transpose_326")]; + tensor reshape_103_cast_fp16 = reshape(shape = concat_131, x = transpose_103_cast_fp16)[name = string("reshape_103_cast_fp16")]; + fp16 var_6452_to_fp16 = const()[name = string("op_6452_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_6453_cast_fp16 = mul(x = q_155_cast_fp16, y = var_6452_to_fp16)[name = string("op_6453_cast_fp16")]; + tensor transpose_417_perm_0 = const()[name = string("transpose_417_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_109_transpose_x_1 = const()[name = string("w_109_transpose_x_1"), val = bool(true)]; + bool w_109_transpose_y_1 = const()[name = string("w_109_transpose_y_1"), val = bool(false)]; + tensor transpose_417_cast_fp16 = transpose(perm = transpose_417_perm_0, x = reshape_101_cast_fp16)[name = string("transpose_325")]; + tensor w_109_cast_fp16 = matmul(transpose_x = w_109_transpose_x_1, transpose_y = w_109_transpose_y_1, x = var_6453_cast_fp16, y = transpose_417_cast_fp16)[name = string("w_109_cast_fp16")]; + tensor pad_51_to_fp16 = const()[name = string("pad_51_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637376)))]; + tensor var_6456_cast_fp16 = add(x = w_109_cast_fp16, y = pad_51_to_fp16)[name = string("op_6456_cast_fp16")]; + tensor w_111_cast_fp16 = softmax(axis = var_6331, x = var_6456_cast_fp16)[name = string("w_111_cast_fp16")]; + tensor transpose_418_perm_0 = const()[name = string("transpose_418_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_51_transpose_x_1 = const()[name = string("attn_51_transpose_x_1"), val = bool(false)]; + bool attn_51_transpose_y_1 = const()[name = string("attn_51_transpose_y_1"), val = bool(true)]; + tensor transpose_418_cast_fp16 = transpose(perm = transpose_418_perm_0, x = reshape_103_cast_fp16)[name = string("transpose_324")]; + tensor attn_51_cast_fp16 = matmul(transpose_x = attn_51_transpose_x_1, transpose_y = attn_51_transpose_y_1, x = transpose_418_cast_fp16, y = w_111_cast_fp16)[name = string("attn_51_cast_fp16")]; + tensor var_6460 = const()[name = string("op_6460"), val = tensor([1, 2048, 1, 1])]; + tensor input_269_cast_fp16 = reshape(shape = var_6460, x = attn_51_cast_fp16)[name = string("input_269_cast_fp16")]; + string attn_output_51_pad_type_0 = const()[name = string("attn_output_51_pad_type_0"), val = string("valid")]; + tensor attn_output_51_strides_0 = const()[name = string("attn_output_51_strides_0"), val = tensor([1, 1])]; + tensor attn_output_51_pad_0 = const()[name = string("attn_output_51_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_51_dilations_0 = const()[name = string("attn_output_51_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_51_groups_0 = const()[name = string("attn_output_51_groups_0"), val = int32(1)]; + tensor attn_output_51_cast_fp16 = conv(dilations = attn_output_51_dilations_0, groups = attn_output_51_groups_0, pad = attn_output_51_pad_0, pad_type = attn_output_51_pad_type_0, strides = attn_output_51_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_269_cast_fp16)[name = string("attn_output_51_cast_fp16")]; + tensor x_195_cast_fp16 = add(x = code_embed_15_cast_fp16, y = attn_output_51_cast_fp16)[name = string("x_195_cast_fp16")]; + tensor var_6474_cast_fp16 = mul(x = x_195_cast_fp16, y = x_195_cast_fp16)[name = string("op_6474_cast_fp16")]; + tensor variance_215_axes_0 = const()[name = string("variance_215_axes_0"), val = tensor([1])]; + bool variance_215_keep_dims_0 = const()[name = string("variance_215_keep_dims_0"), val = bool(true)]; + tensor variance_215_cast_fp16 = reduce_mean(axes = variance_215_axes_0, keep_dims = variance_215_keep_dims_0, x = var_6474_cast_fp16)[name = string("variance_215_cast_fp16")]; + fp16 var_6477_to_fp16 = const()[name = string("op_6477_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6478_cast_fp16 = add(x = variance_215_cast_fp16, y = var_6477_to_fp16)[name = string("op_6478_cast_fp16")]; + fp32 var_6479_epsilon_0 = const()[name = string("op_6479_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6479_cast_fp16 = rsqrt(epsilon = var_6479_epsilon_0, x = var_6478_cast_fp16)[name = string("op_6479_cast_fp16")]; + tensor var_6480_cast_fp16 = mul(x = x_195_cast_fp16, y = var_6479_cast_fp16)[name = string("op_6480_cast_fp16")]; + tensor input_271_cast_fp16 = mul(x = var_6480_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_271_cast_fp16")]; + string input_273_pad_type_0 = const()[name = string("input_273_pad_type_0"), val = string("valid")]; + tensor input_273_strides_0 = const()[name = string("input_273_strides_0"), val = tensor([1, 1])]; + tensor input_273_pad_0 = const()[name = string("input_273_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_273_dilations_0 = const()[name = string("input_273_dilations_0"), val = tensor([1, 1])]; + int32 input_273_groups_0 = const()[name = string("input_273_groups_0"), val = int32(1)]; + tensor input_273_cast_fp16 = conv(dilations = input_273_dilations_0, groups = input_273_groups_0, pad = input_273_pad_0, pad_type = input_273_pad_type_0, strides = input_273_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_271_cast_fp16)[name = string("input_273_cast_fp16")]; + tensor var_6488_cast_fp16 = silu(x = input_273_cast_fp16)[name = string("op_6488_cast_fp16")]; + string var_6494_pad_type_0 = const()[name = string("op_6494_pad_type_0"), val = string("valid")]; + tensor var_6494_strides_0 = const()[name = string("op_6494_strides_0"), val = tensor([1, 1])]; + tensor var_6494_pad_0 = const()[name = string("op_6494_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6494_dilations_0 = const()[name = string("op_6494_dilations_0"), val = tensor([1, 1])]; + int32 var_6494_groups_0 = const()[name = string("op_6494_groups_0"), val = int32(1)]; + tensor var_6494_cast_fp16 = conv(dilations = var_6494_dilations_0, groups = var_6494_groups_0, pad = var_6494_pad_0, pad_type = var_6494_pad_type_0, strides = var_6494_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_271_cast_fp16)[name = string("op_6494_cast_fp16")]; + tensor input_275_cast_fp16 = mul(x = var_6488_cast_fp16, y = var_6494_cast_fp16)[name = string("input_275_cast_fp16")]; + string h_51_pad_type_0 = const()[name = string("h_51_pad_type_0"), val = string("valid")]; + tensor h_51_strides_0 = const()[name = string("h_51_strides_0"), val = tensor([1, 1])]; + tensor h_51_pad_0 = const()[name = string("h_51_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_51_dilations_0 = const()[name = string("h_51_dilations_0"), val = tensor([1, 1])]; + int32 h_51_groups_0 = const()[name = string("h_51_groups_0"), val = int32(1)]; + tensor h_51_cast_fp16 = conv(dilations = h_51_dilations_0, groups = h_51_groups_0, pad = h_51_pad_0, pad_type = h_51_pad_type_0, strides = h_51_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_275_cast_fp16)[name = string("h_51_cast_fp16")]; + tensor x_197_cast_fp16 = add(x = x_195_cast_fp16, y = h_51_cast_fp16)[name = string("x_197_cast_fp16")]; + tensor key_cache_53_begin_0 = const()[name = string("key_cache_53_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_53_end_0 = const()[name = string("key_cache_53_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_53_end_mask_0 = const()[name = string("key_cache_53_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_53_cast_fp16 = slice_by_index(begin = key_cache_53_begin_0, end = key_cache_53_end_0, end_mask = key_cache_53_end_mask_0, x = layer_key_caches_11_cast_fp16)[name = string("key_cache_53_cast_fp16")]; + tensor value_cache_53_begin_0 = const()[name = string("value_cache_53_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_53_end_0 = const()[name = string("value_cache_53_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_53_end_mask_0 = const()[name = string("value_cache_53_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_53_cast_fp16 = slice_by_index(begin = value_cache_53_begin_0, end = value_cache_53_end_0, end_mask = value_cache_53_end_mask_0, x = layer_value_caches_11_cast_fp16)[name = string("value_cache_53_cast_fp16")]; + int32 var_6547 = const()[name = string("op_6547"), val = int32(2)]; + int32 var_6551 = const()[name = string("op_6551"), val = int32(3)]; + tensor var_6566_cast_fp16 = mul(x = x_197_cast_fp16, y = x_197_cast_fp16)[name = string("op_6566_cast_fp16")]; + tensor variance_217_axes_0 = const()[name = string("variance_217_axes_0"), val = tensor([1])]; + bool variance_217_keep_dims_0 = const()[name = string("variance_217_keep_dims_0"), val = bool(true)]; + tensor variance_217_cast_fp16 = reduce_mean(axes = variance_217_axes_0, keep_dims = variance_217_keep_dims_0, x = var_6566_cast_fp16)[name = string("variance_217_cast_fp16")]; + fp16 var_6569_to_fp16 = const()[name = string("op_6569_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6570_cast_fp16 = add(x = variance_217_cast_fp16, y = var_6569_to_fp16)[name = string("op_6570_cast_fp16")]; + fp32 var_6571_epsilon_0 = const()[name = string("op_6571_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6571_cast_fp16 = rsqrt(epsilon = var_6571_epsilon_0, x = var_6570_cast_fp16)[name = string("op_6571_cast_fp16")]; + tensor var_6572_cast_fp16 = mul(x = x_197_cast_fp16, y = var_6571_cast_fp16)[name = string("op_6572_cast_fp16")]; + tensor input_277_cast_fp16 = mul(x = var_6572_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_277_cast_fp16")]; + string q_157_pad_type_0 = const()[name = string("q_157_pad_type_0"), val = string("valid")]; + tensor q_157_strides_0 = const()[name = string("q_157_strides_0"), val = tensor([1, 1])]; + tensor q_157_pad_0 = const()[name = string("q_157_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_157_dilations_0 = const()[name = string("q_157_dilations_0"), val = tensor([1, 1])]; + int32 q_157_groups_0 = const()[name = string("q_157_groups_0"), val = int32(1)]; + tensor q_157_cast_fp16 = conv(dilations = q_157_dilations_0, groups = q_157_groups_0, pad = q_157_pad_0, pad_type = q_157_pad_type_0, strides = q_157_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_277_cast_fp16)[name = string("q_157_cast_fp16")]; + string k_157_pad_type_0 = const()[name = string("k_157_pad_type_0"), val = string("valid")]; + tensor k_157_strides_0 = const()[name = string("k_157_strides_0"), val = tensor([1, 1])]; + tensor k_157_pad_0 = const()[name = string("k_157_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_157_dilations_0 = const()[name = string("k_157_dilations_0"), val = tensor([1, 1])]; + int32 k_157_groups_0 = const()[name = string("k_157_groups_0"), val = int32(1)]; + tensor k_157_cast_fp16 = conv(dilations = k_157_dilations_0, groups = k_157_groups_0, pad = k_157_pad_0, pad_type = k_157_pad_type_0, strides = k_157_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_277_cast_fp16)[name = string("k_157_cast_fp16")]; + string v_53_pad_type_0 = const()[name = string("v_53_pad_type_0"), val = string("valid")]; + tensor v_53_strides_0 = const()[name = string("v_53_strides_0"), val = tensor([1, 1])]; + tensor v_53_pad_0 = const()[name = string("v_53_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_53_dilations_0 = const()[name = string("v_53_dilations_0"), val = tensor([1, 1])]; + int32 v_53_groups_0 = const()[name = string("v_53_groups_0"), val = int32(1)]; + tensor v_53_cast_fp16 = conv(dilations = v_53_dilations_0, groups = v_53_groups_0, pad = v_53_pad_0, pad_type = v_53_pad_type_0, strides = v_53_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_277_cast_fp16)[name = string("v_53_cast_fp16")]; + tensor var_6606 = const()[name = string("op_6606"), val = tensor([16, 128, 1, 1])]; + tensor x_199_cast_fp16 = reshape(shape = var_6606, x = q_157_cast_fp16)[name = string("x_199_cast_fp16")]; + tensor var_6609_cast_fp16 = mul(x = x_199_cast_fp16, y = x_199_cast_fp16)[name = string("op_6609_cast_fp16")]; + tensor variance_219_axes_0 = const()[name = string("variance_219_axes_0"), val = tensor([1])]; + bool variance_219_keep_dims_0 = const()[name = string("variance_219_keep_dims_0"), val = bool(true)]; + tensor variance_219_cast_fp16 = reduce_mean(axes = variance_219_axes_0, keep_dims = variance_219_keep_dims_0, x = var_6609_cast_fp16)[name = string("variance_219_cast_fp16")]; + fp16 var_6612_to_fp16 = const()[name = string("op_6612_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6613_cast_fp16 = add(x = variance_219_cast_fp16, y = var_6612_to_fp16)[name = string("op_6613_cast_fp16")]; + fp32 var_6614_epsilon_0 = const()[name = string("op_6614_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6614_cast_fp16 = rsqrt(epsilon = var_6614_epsilon_0, x = var_6613_cast_fp16)[name = string("op_6614_cast_fp16")]; + tensor var_6615_cast_fp16 = mul(x = x_199_cast_fp16, y = var_6614_cast_fp16)[name = string("op_6615_cast_fp16")]; + tensor q_159_cast_fp16 = mul(x = var_6615_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_159_cast_fp16")]; + tensor var_6617 = const()[name = string("op_6617"), val = tensor([8, 128, 1, 1])]; + tensor x_201_cast_fp16 = reshape(shape = var_6617, x = k_157_cast_fp16)[name = string("x_201_cast_fp16")]; + tensor var_6620_cast_fp16 = mul(x = x_201_cast_fp16, y = x_201_cast_fp16)[name = string("op_6620_cast_fp16")]; + tensor variance_221_axes_0 = const()[name = string("variance_221_axes_0"), val = tensor([1])]; + bool variance_221_keep_dims_0 = const()[name = string("variance_221_keep_dims_0"), val = bool(true)]; + tensor variance_221_cast_fp16 = reduce_mean(axes = variance_221_axes_0, keep_dims = variance_221_keep_dims_0, x = var_6620_cast_fp16)[name = string("variance_221_cast_fp16")]; + fp16 var_6623_to_fp16 = const()[name = string("op_6623_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6624_cast_fp16 = add(x = variance_221_cast_fp16, y = var_6623_to_fp16)[name = string("op_6624_cast_fp16")]; + fp32 var_6625_epsilon_0 = const()[name = string("op_6625_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6625_cast_fp16 = rsqrt(epsilon = var_6625_epsilon_0, x = var_6624_cast_fp16)[name = string("op_6625_cast_fp16")]; + tensor var_6626_cast_fp16 = mul(x = x_201_cast_fp16, y = var_6625_cast_fp16)[name = string("op_6626_cast_fp16")]; + tensor k_159_cast_fp16 = mul(x = var_6626_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_159_cast_fp16")]; + tensor var_6628 = const()[name = string("op_6628"), val = tensor([1, 16, 128, 1])]; + tensor z_105_cast_fp16 = reshape(shape = var_6628, x = q_159_cast_fp16)[name = string("z_105_cast_fp16")]; + tensor var_6630 = const()[name = string("op_6630"), val = tensor([1, 8, 128, 1])]; + tensor z_107_cast_fp16 = reshape(shape = var_6630, x = k_159_cast_fp16)[name = string("z_107_cast_fp16")]; + tensor z1_105_begin_0 = const()[name = string("z1_105_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_105_end_0 = const()[name = string("z1_105_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_105_end_mask_0 = const()[name = string("z1_105_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_105_cast_fp16 = slice_by_index(begin = z1_105_begin_0, end = z1_105_end_0, end_mask = z1_105_end_mask_0, x = z_105_cast_fp16)[name = string("z1_105_cast_fp16")]; + tensor z2_105_begin_0 = const()[name = string("z2_105_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_105_end_0 = const()[name = string("z2_105_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_105_end_mask_0 = const()[name = string("z2_105_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_105_cast_fp16 = slice_by_index(begin = z2_105_begin_0, end = z2_105_end_0, end_mask = z2_105_end_mask_0, x = z_105_cast_fp16)[name = string("z2_105_cast_fp16")]; + tensor var_6638_cast_fp16 = mul(x = z_105_cast_fp16, y = cos_51_to_fp16)[name = string("op_6638_cast_fp16")]; + fp16 const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6639_cast_fp16 = mul(x = z2_105_cast_fp16, y = const_58_promoted_to_fp16)[name = string("op_6639_cast_fp16")]; + bool var_6641_interleave_0 = const()[name = string("op_6641_interleave_0"), val = bool(false)]; + tensor var_6641_cast_fp16 = concat(axis = var_6547, interleave = var_6641_interleave_0, values = (var_6639_cast_fp16, z1_105_cast_fp16))[name = string("op_6641_cast_fp16")]; + tensor var_6642_cast_fp16 = mul(x = var_6641_cast_fp16, y = sin_51_to_fp16)[name = string("op_6642_cast_fp16")]; + tensor q_161_cast_fp16 = add(x = var_6638_cast_fp16, y = var_6642_cast_fp16)[name = string("q_161_cast_fp16")]; + tensor z1_107_begin_0 = const()[name = string("z1_107_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_107_end_0 = const()[name = string("z1_107_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_107_end_mask_0 = const()[name = string("z1_107_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_107_cast_fp16 = slice_by_index(begin = z1_107_begin_0, end = z1_107_end_0, end_mask = z1_107_end_mask_0, x = z_107_cast_fp16)[name = string("z1_107_cast_fp16")]; + tensor z2_107_begin_0 = const()[name = string("z2_107_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_107_end_0 = const()[name = string("z2_107_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_107_end_mask_0 = const()[name = string("z2_107_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_107_cast_fp16 = slice_by_index(begin = z2_107_begin_0, end = z2_107_end_0, end_mask = z2_107_end_mask_0, x = z_107_cast_fp16)[name = string("z2_107_cast_fp16")]; + tensor var_6650_cast_fp16 = mul(x = z_107_cast_fp16, y = cos_51_to_fp16)[name = string("op_6650_cast_fp16")]; + fp16 const_59_promoted_to_fp16 = const()[name = string("const_59_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6651_cast_fp16 = mul(x = z2_107_cast_fp16, y = const_59_promoted_to_fp16)[name = string("op_6651_cast_fp16")]; + bool var_6653_interleave_0 = const()[name = string("op_6653_interleave_0"), val = bool(false)]; + tensor var_6653_cast_fp16 = concat(axis = var_6547, interleave = var_6653_interleave_0, values = (var_6651_cast_fp16, z1_107_cast_fp16))[name = string("op_6653_cast_fp16")]; + tensor var_6654_cast_fp16 = mul(x = var_6653_cast_fp16, y = sin_51_to_fp16)[name = string("op_6654_cast_fp16")]; + tensor k_161_cast_fp16 = add(x = var_6650_cast_fp16, y = var_6654_cast_fp16)[name = string("k_161_cast_fp16")]; + tensor var_6656 = const()[name = string("op_6656"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_53_cast_fp16 = reshape(shape = var_6656, x = k_161_cast_fp16)[name = string("cur_key_53_cast_fp16")]; + tensor var_6658_to_fp16 = const()[name = string("op_6658_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637120)))]; + tensor var_6659_cast_fp16 = mul(x = key_cache_53_cast_fp16, y = var_6658_to_fp16)[name = string("op_6659_cast_fp16")]; + tensor upd_53_to_fp16 = const()[name = string("upd_53_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637248)))]; + tensor var_6660_cast_fp16 = mul(x = cur_key_53_cast_fp16, y = upd_53_to_fp16)[name = string("op_6660_cast_fp16")]; + tensor key_53_cast_fp16 = add(x = var_6659_cast_fp16, y = var_6660_cast_fp16)[name = string("key_53_cast_fp16")]; + tensor var_6662_to_fp16 = const()[name = string("op_6662_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637120)))]; + tensor var_6663_cast_fp16 = mul(x = value_cache_53_cast_fp16, y = var_6662_to_fp16)[name = string("op_6663_cast_fp16")]; + tensor var_6664_cast_fp16 = mul(x = v_53_cast_fp16, y = upd_53_to_fp16)[name = string("op_6664_cast_fp16")]; + tensor value_53_cast_fp16 = add(x = var_6663_cast_fp16, y = var_6664_cast_fp16)[name = string("value_53_cast_fp16")]; + tensor var_6666 = const()[name = string("op_6666"), val = tensor([1, 8, 128, 16])]; + tensor kh_105_cast_fp16 = reshape(shape = var_6666, x = key_53_cast_fp16)[name = string("kh_105_cast_fp16")]; + tensor var_6668 = const()[name = string("op_6668"), val = tensor([1, 8, 128, 16])]; + tensor vh_105_cast_fp16 = reshape(shape = var_6668, x = value_53_cast_fp16)[name = string("vh_105_cast_fp16")]; + tensor transpose_104_perm_0 = const()[name = string("transpose_104_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_52_reps_0 = const()[name = string("tile_52_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_104_cast_fp16 = transpose(perm = transpose_104_perm_0, x = kh_105_cast_fp16)[name = string("transpose_323")]; + tensor tile_52_cast_fp16 = tile(reps = tile_52_reps_0, x = transpose_104_cast_fp16)[name = string("tile_52_cast_fp16")]; + tensor concat_132 = const()[name = string("concat_132"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_104_cast_fp16 = reshape(shape = concat_132, x = tile_52_cast_fp16)[name = string("reshape_104_cast_fp16")]; + tensor transpose_105_perm_0 = const()[name = string("transpose_105_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_133 = const()[name = string("concat_133"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_105_cast_fp16 = transpose(perm = transpose_105_perm_0, x = reshape_104_cast_fp16)[name = string("transpose_322")]; + tensor reshape_105_cast_fp16 = reshape(shape = concat_133, x = transpose_105_cast_fp16)[name = string("reshape_105_cast_fp16")]; + tensor transpose_106_perm_0 = const()[name = string("transpose_106_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_53_reps_0 = const()[name = string("tile_53_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_106_cast_fp16 = transpose(perm = transpose_106_perm_0, x = vh_105_cast_fp16)[name = string("transpose_321")]; + tensor tile_53_cast_fp16 = tile(reps = tile_53_reps_0, x = transpose_106_cast_fp16)[name = string("tile_53_cast_fp16")]; + tensor concat_134 = const()[name = string("concat_134"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_106_cast_fp16 = reshape(shape = concat_134, x = tile_53_cast_fp16)[name = string("reshape_106_cast_fp16")]; + tensor transpose_107_perm_0 = const()[name = string("transpose_107_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_135 = const()[name = string("concat_135"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_107_cast_fp16 = transpose(perm = transpose_107_perm_0, x = reshape_106_cast_fp16)[name = string("transpose_320")]; + tensor reshape_107_cast_fp16 = reshape(shape = concat_135, x = transpose_107_cast_fp16)[name = string("reshape_107_cast_fp16")]; + fp16 var_6672_to_fp16 = const()[name = string("op_6672_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_6673_cast_fp16 = mul(x = q_161_cast_fp16, y = var_6672_to_fp16)[name = string("op_6673_cast_fp16")]; + tensor transpose_421_perm_0 = const()[name = string("transpose_421_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_113_transpose_x_1 = const()[name = string("w_113_transpose_x_1"), val = bool(true)]; + bool w_113_transpose_y_1 = const()[name = string("w_113_transpose_y_1"), val = bool(false)]; + tensor transpose_421_cast_fp16 = transpose(perm = transpose_421_perm_0, x = reshape_105_cast_fp16)[name = string("transpose_319")]; + tensor w_113_cast_fp16 = matmul(transpose_x = w_113_transpose_x_1, transpose_y = w_113_transpose_y_1, x = var_6673_cast_fp16, y = transpose_421_cast_fp16)[name = string("w_113_cast_fp16")]; + tensor pad_53_to_fp16 = const()[name = string("pad_53_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637376)))]; + tensor var_6676_cast_fp16 = add(x = w_113_cast_fp16, y = pad_53_to_fp16)[name = string("op_6676_cast_fp16")]; + tensor w_115_cast_fp16 = softmax(axis = var_6551, x = var_6676_cast_fp16)[name = string("w_115_cast_fp16")]; + tensor transpose_422_perm_0 = const()[name = string("transpose_422_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_53_transpose_x_1 = const()[name = string("attn_53_transpose_x_1"), val = bool(false)]; + bool attn_53_transpose_y_1 = const()[name = string("attn_53_transpose_y_1"), val = bool(true)]; + tensor transpose_422_cast_fp16 = transpose(perm = transpose_422_perm_0, x = reshape_107_cast_fp16)[name = string("transpose_318")]; + tensor attn_53_cast_fp16 = matmul(transpose_x = attn_53_transpose_x_1, transpose_y = attn_53_transpose_y_1, x = transpose_422_cast_fp16, y = w_115_cast_fp16)[name = string("attn_53_cast_fp16")]; + tensor var_6680 = const()[name = string("op_6680"), val = tensor([1, 2048, 1, 1])]; + tensor input_279_cast_fp16 = reshape(shape = var_6680, x = attn_53_cast_fp16)[name = string("input_279_cast_fp16")]; + string attn_output_53_pad_type_0 = const()[name = string("attn_output_53_pad_type_0"), val = string("valid")]; + tensor attn_output_53_strides_0 = const()[name = string("attn_output_53_strides_0"), val = tensor([1, 1])]; + tensor attn_output_53_pad_0 = const()[name = string("attn_output_53_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_53_dilations_0 = const()[name = string("attn_output_53_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_53_groups_0 = const()[name = string("attn_output_53_groups_0"), val = int32(1)]; + tensor attn_output_53_cast_fp16 = conv(dilations = attn_output_53_dilations_0, groups = attn_output_53_groups_0, pad = attn_output_53_pad_0, pad_type = attn_output_53_pad_type_0, strides = attn_output_53_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_279_cast_fp16)[name = string("attn_output_53_cast_fp16")]; + tensor x_203_cast_fp16 = add(x = x_197_cast_fp16, y = attn_output_53_cast_fp16)[name = string("x_203_cast_fp16")]; + tensor var_6694_cast_fp16 = mul(x = x_203_cast_fp16, y = x_203_cast_fp16)[name = string("op_6694_cast_fp16")]; + tensor variance_223_axes_0 = const()[name = string("variance_223_axes_0"), val = tensor([1])]; + bool variance_223_keep_dims_0 = const()[name = string("variance_223_keep_dims_0"), val = bool(true)]; + tensor variance_223_cast_fp16 = reduce_mean(axes = variance_223_axes_0, keep_dims = variance_223_keep_dims_0, x = var_6694_cast_fp16)[name = string("variance_223_cast_fp16")]; + fp16 var_6697_to_fp16 = const()[name = string("op_6697_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6698_cast_fp16 = add(x = variance_223_cast_fp16, y = var_6697_to_fp16)[name = string("op_6698_cast_fp16")]; + fp32 var_6699_epsilon_0 = const()[name = string("op_6699_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6699_cast_fp16 = rsqrt(epsilon = var_6699_epsilon_0, x = var_6698_cast_fp16)[name = string("op_6699_cast_fp16")]; + tensor var_6700_cast_fp16 = mul(x = x_203_cast_fp16, y = var_6699_cast_fp16)[name = string("op_6700_cast_fp16")]; + tensor input_281_cast_fp16 = mul(x = var_6700_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_281_cast_fp16")]; + string input_283_pad_type_0 = const()[name = string("input_283_pad_type_0"), val = string("valid")]; + tensor input_283_strides_0 = const()[name = string("input_283_strides_0"), val = tensor([1, 1])]; + tensor input_283_pad_0 = const()[name = string("input_283_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_283_dilations_0 = const()[name = string("input_283_dilations_0"), val = tensor([1, 1])]; + int32 input_283_groups_0 = const()[name = string("input_283_groups_0"), val = int32(1)]; + tensor input_283_cast_fp16 = conv(dilations = input_283_dilations_0, groups = input_283_groups_0, pad = input_283_pad_0, pad_type = input_283_pad_type_0, strides = input_283_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_281_cast_fp16)[name = string("input_283_cast_fp16")]; + tensor var_6708_cast_fp16 = silu(x = input_283_cast_fp16)[name = string("op_6708_cast_fp16")]; + string var_6714_pad_type_0 = const()[name = string("op_6714_pad_type_0"), val = string("valid")]; + tensor var_6714_strides_0 = const()[name = string("op_6714_strides_0"), val = tensor([1, 1])]; + tensor var_6714_pad_0 = const()[name = string("op_6714_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6714_dilations_0 = const()[name = string("op_6714_dilations_0"), val = tensor([1, 1])]; + int32 var_6714_groups_0 = const()[name = string("op_6714_groups_0"), val = int32(1)]; + tensor var_6714_cast_fp16 = conv(dilations = var_6714_dilations_0, groups = var_6714_groups_0, pad = var_6714_pad_0, pad_type = var_6714_pad_type_0, strides = var_6714_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_281_cast_fp16)[name = string("op_6714_cast_fp16")]; + tensor input_285_cast_fp16 = mul(x = var_6708_cast_fp16, y = var_6714_cast_fp16)[name = string("input_285_cast_fp16")]; + string h_53_pad_type_0 = const()[name = string("h_53_pad_type_0"), val = string("valid")]; + tensor h_53_strides_0 = const()[name = string("h_53_strides_0"), val = tensor([1, 1])]; + tensor h_53_pad_0 = const()[name = string("h_53_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_53_dilations_0 = const()[name = string("h_53_dilations_0"), val = tensor([1, 1])]; + int32 h_53_groups_0 = const()[name = string("h_53_groups_0"), val = int32(1)]; + tensor h_53_cast_fp16 = conv(dilations = h_53_dilations_0, groups = h_53_groups_0, pad = h_53_pad_0, pad_type = h_53_pad_type_0, strides = h_53_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_285_cast_fp16)[name = string("h_53_cast_fp16")]; + tensor x_205_cast_fp16 = add(x = x_203_cast_fp16, y = h_53_cast_fp16)[name = string("x_205_cast_fp16")]; + tensor key_cache_55_begin_0 = const()[name = string("key_cache_55_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_55_end_0 = const()[name = string("key_cache_55_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_55_end_mask_0 = const()[name = string("key_cache_55_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_55_cast_fp16 = slice_by_index(begin = key_cache_55_begin_0, end = key_cache_55_end_0, end_mask = key_cache_55_end_mask_0, x = layer_key_caches_11_cast_fp16)[name = string("key_cache_55_cast_fp16")]; + tensor value_cache_55_begin_0 = const()[name = string("value_cache_55_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_55_end_0 = const()[name = string("value_cache_55_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_55_end_mask_0 = const()[name = string("value_cache_55_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_55_cast_fp16 = slice_by_index(begin = value_cache_55_begin_0, end = value_cache_55_end_0, end_mask = value_cache_55_end_mask_0, x = layer_value_caches_11_cast_fp16)[name = string("value_cache_55_cast_fp16")]; + int32 var_6767 = const()[name = string("op_6767"), val = int32(2)]; + int32 var_6771 = const()[name = string("op_6771"), val = int32(3)]; + tensor var_6786_cast_fp16 = mul(x = x_205_cast_fp16, y = x_205_cast_fp16)[name = string("op_6786_cast_fp16")]; + tensor variance_225_axes_0 = const()[name = string("variance_225_axes_0"), val = tensor([1])]; + bool variance_225_keep_dims_0 = const()[name = string("variance_225_keep_dims_0"), val = bool(true)]; + tensor variance_225_cast_fp16 = reduce_mean(axes = variance_225_axes_0, keep_dims = variance_225_keep_dims_0, x = var_6786_cast_fp16)[name = string("variance_225_cast_fp16")]; + fp16 var_6789_to_fp16 = const()[name = string("op_6789_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6790_cast_fp16 = add(x = variance_225_cast_fp16, y = var_6789_to_fp16)[name = string("op_6790_cast_fp16")]; + fp32 var_6791_epsilon_0 = const()[name = string("op_6791_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6791_cast_fp16 = rsqrt(epsilon = var_6791_epsilon_0, x = var_6790_cast_fp16)[name = string("op_6791_cast_fp16")]; + tensor var_6792_cast_fp16 = mul(x = x_205_cast_fp16, y = var_6791_cast_fp16)[name = string("op_6792_cast_fp16")]; + tensor input_287_cast_fp16 = mul(x = var_6792_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_287_cast_fp16")]; + string q_163_pad_type_0 = const()[name = string("q_163_pad_type_0"), val = string("valid")]; + tensor q_163_strides_0 = const()[name = string("q_163_strides_0"), val = tensor([1, 1])]; + tensor q_163_pad_0 = const()[name = string("q_163_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_163_dilations_0 = const()[name = string("q_163_dilations_0"), val = tensor([1, 1])]; + int32 q_163_groups_0 = const()[name = string("q_163_groups_0"), val = int32(1)]; + tensor q_163_cast_fp16 = conv(dilations = q_163_dilations_0, groups = q_163_groups_0, pad = q_163_pad_0, pad_type = q_163_pad_type_0, strides = q_163_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_287_cast_fp16)[name = string("q_163_cast_fp16")]; + string k_163_pad_type_0 = const()[name = string("k_163_pad_type_0"), val = string("valid")]; + tensor k_163_strides_0 = const()[name = string("k_163_strides_0"), val = tensor([1, 1])]; + tensor k_163_pad_0 = const()[name = string("k_163_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_163_dilations_0 = const()[name = string("k_163_dilations_0"), val = tensor([1, 1])]; + int32 k_163_groups_0 = const()[name = string("k_163_groups_0"), val = int32(1)]; + tensor k_163_cast_fp16 = conv(dilations = k_163_dilations_0, groups = k_163_groups_0, pad = k_163_pad_0, pad_type = k_163_pad_type_0, strides = k_163_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_287_cast_fp16)[name = string("k_163_cast_fp16")]; + string v_55_pad_type_0 = const()[name = string("v_55_pad_type_0"), val = string("valid")]; + tensor v_55_strides_0 = const()[name = string("v_55_strides_0"), val = tensor([1, 1])]; + tensor v_55_pad_0 = const()[name = string("v_55_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_55_dilations_0 = const()[name = string("v_55_dilations_0"), val = tensor([1, 1])]; + int32 v_55_groups_0 = const()[name = string("v_55_groups_0"), val = int32(1)]; + tensor v_55_cast_fp16 = conv(dilations = v_55_dilations_0, groups = v_55_groups_0, pad = v_55_pad_0, pad_type = v_55_pad_type_0, strides = v_55_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_287_cast_fp16)[name = string("v_55_cast_fp16")]; + tensor var_6826 = const()[name = string("op_6826"), val = tensor([16, 128, 1, 1])]; + tensor x_207_cast_fp16 = reshape(shape = var_6826, x = q_163_cast_fp16)[name = string("x_207_cast_fp16")]; + tensor var_6829_cast_fp16 = mul(x = x_207_cast_fp16, y = x_207_cast_fp16)[name = string("op_6829_cast_fp16")]; + tensor variance_227_axes_0 = const()[name = string("variance_227_axes_0"), val = tensor([1])]; + bool variance_227_keep_dims_0 = const()[name = string("variance_227_keep_dims_0"), val = bool(true)]; + tensor variance_227_cast_fp16 = reduce_mean(axes = variance_227_axes_0, keep_dims = variance_227_keep_dims_0, x = var_6829_cast_fp16)[name = string("variance_227_cast_fp16")]; + fp16 var_6832_to_fp16 = const()[name = string("op_6832_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6833_cast_fp16 = add(x = variance_227_cast_fp16, y = var_6832_to_fp16)[name = string("op_6833_cast_fp16")]; + fp32 var_6834_epsilon_0 = const()[name = string("op_6834_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6834_cast_fp16 = rsqrt(epsilon = var_6834_epsilon_0, x = var_6833_cast_fp16)[name = string("op_6834_cast_fp16")]; + tensor var_6835_cast_fp16 = mul(x = x_207_cast_fp16, y = var_6834_cast_fp16)[name = string("op_6835_cast_fp16")]; + tensor q_165_cast_fp16 = mul(x = var_6835_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_165_cast_fp16")]; + tensor var_6837 = const()[name = string("op_6837"), val = tensor([8, 128, 1, 1])]; + tensor x_209_cast_fp16 = reshape(shape = var_6837, x = k_163_cast_fp16)[name = string("x_209_cast_fp16")]; + tensor var_6840_cast_fp16 = mul(x = x_209_cast_fp16, y = x_209_cast_fp16)[name = string("op_6840_cast_fp16")]; + tensor variance_229_axes_0 = const()[name = string("variance_229_axes_0"), val = tensor([1])]; + bool variance_229_keep_dims_0 = const()[name = string("variance_229_keep_dims_0"), val = bool(true)]; + tensor variance_229_cast_fp16 = reduce_mean(axes = variance_229_axes_0, keep_dims = variance_229_keep_dims_0, x = var_6840_cast_fp16)[name = string("variance_229_cast_fp16")]; + fp16 var_6843_to_fp16 = const()[name = string("op_6843_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6844_cast_fp16 = add(x = variance_229_cast_fp16, y = var_6843_to_fp16)[name = string("op_6844_cast_fp16")]; + fp32 var_6845_epsilon_0 = const()[name = string("op_6845_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6845_cast_fp16 = rsqrt(epsilon = var_6845_epsilon_0, x = var_6844_cast_fp16)[name = string("op_6845_cast_fp16")]; + tensor var_6846_cast_fp16 = mul(x = x_209_cast_fp16, y = var_6845_cast_fp16)[name = string("op_6846_cast_fp16")]; + tensor k_165_cast_fp16 = mul(x = var_6846_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_165_cast_fp16")]; + tensor var_6848 = const()[name = string("op_6848"), val = tensor([1, 16, 128, 1])]; + tensor z_109_cast_fp16 = reshape(shape = var_6848, x = q_165_cast_fp16)[name = string("z_109_cast_fp16")]; + tensor var_6850 = const()[name = string("op_6850"), val = tensor([1, 8, 128, 1])]; + tensor z_111_cast_fp16 = reshape(shape = var_6850, x = k_165_cast_fp16)[name = string("z_111_cast_fp16")]; + tensor z1_109_begin_0 = const()[name = string("z1_109_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_109_end_0 = const()[name = string("z1_109_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_109_end_mask_0 = const()[name = string("z1_109_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_109_cast_fp16 = slice_by_index(begin = z1_109_begin_0, end = z1_109_end_0, end_mask = z1_109_end_mask_0, x = z_109_cast_fp16)[name = string("z1_109_cast_fp16")]; + tensor z2_109_begin_0 = const()[name = string("z2_109_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_109_end_0 = const()[name = string("z2_109_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_109_end_mask_0 = const()[name = string("z2_109_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_109_cast_fp16 = slice_by_index(begin = z2_109_begin_0, end = z2_109_end_0, end_mask = z2_109_end_mask_0, x = z_109_cast_fp16)[name = string("z2_109_cast_fp16")]; + tensor var_6858_cast_fp16 = mul(x = z_109_cast_fp16, y = cos_51_to_fp16)[name = string("op_6858_cast_fp16")]; + fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6859_cast_fp16 = mul(x = z2_109_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_6859_cast_fp16")]; + bool var_6861_interleave_0 = const()[name = string("op_6861_interleave_0"), val = bool(false)]; + tensor var_6861_cast_fp16 = concat(axis = var_6767, interleave = var_6861_interleave_0, values = (var_6859_cast_fp16, z1_109_cast_fp16))[name = string("op_6861_cast_fp16")]; + tensor var_6862_cast_fp16 = mul(x = var_6861_cast_fp16, y = sin_51_to_fp16)[name = string("op_6862_cast_fp16")]; + tensor q_167_cast_fp16 = add(x = var_6858_cast_fp16, y = var_6862_cast_fp16)[name = string("q_167_cast_fp16")]; + tensor z1_111_begin_0 = const()[name = string("z1_111_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_111_end_0 = const()[name = string("z1_111_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_111_end_mask_0 = const()[name = string("z1_111_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_111_cast_fp16 = slice_by_index(begin = z1_111_begin_0, end = z1_111_end_0, end_mask = z1_111_end_mask_0, x = z_111_cast_fp16)[name = string("z1_111_cast_fp16")]; + tensor z2_111_begin_0 = const()[name = string("z2_111_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_111_end_0 = const()[name = string("z2_111_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_111_end_mask_0 = const()[name = string("z2_111_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_111_cast_fp16 = slice_by_index(begin = z2_111_begin_0, end = z2_111_end_0, end_mask = z2_111_end_mask_0, x = z_111_cast_fp16)[name = string("z2_111_cast_fp16")]; + tensor var_6870_cast_fp16 = mul(x = z_111_cast_fp16, y = cos_51_to_fp16)[name = string("op_6870_cast_fp16")]; + fp16 const_61_promoted_to_fp16 = const()[name = string("const_61_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6871_cast_fp16 = mul(x = z2_111_cast_fp16, y = const_61_promoted_to_fp16)[name = string("op_6871_cast_fp16")]; + bool var_6873_interleave_0 = const()[name = string("op_6873_interleave_0"), val = bool(false)]; + tensor var_6873_cast_fp16 = concat(axis = var_6767, interleave = var_6873_interleave_0, values = (var_6871_cast_fp16, z1_111_cast_fp16))[name = string("op_6873_cast_fp16")]; + tensor var_6874_cast_fp16 = mul(x = var_6873_cast_fp16, y = sin_51_to_fp16)[name = string("op_6874_cast_fp16")]; + tensor k_167_cast_fp16 = add(x = var_6870_cast_fp16, y = var_6874_cast_fp16)[name = string("k_167_cast_fp16")]; + tensor var_6876 = const()[name = string("op_6876"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_55_cast_fp16 = reshape(shape = var_6876, x = k_167_cast_fp16)[name = string("cur_key_55_cast_fp16")]; + tensor var_6878_to_fp16 = const()[name = string("op_6878_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637120)))]; + tensor var_6879_cast_fp16 = mul(x = key_cache_55_cast_fp16, y = var_6878_to_fp16)[name = string("op_6879_cast_fp16")]; + tensor upd_55_to_fp16 = const()[name = string("upd_55_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637248)))]; + tensor var_6880_cast_fp16 = mul(x = cur_key_55_cast_fp16, y = upd_55_to_fp16)[name = string("op_6880_cast_fp16")]; + tensor key_55_cast_fp16 = add(x = var_6879_cast_fp16, y = var_6880_cast_fp16)[name = string("key_55_cast_fp16")]; + tensor var_6882_to_fp16 = const()[name = string("op_6882_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637120)))]; + tensor var_6883_cast_fp16 = mul(x = value_cache_55_cast_fp16, y = var_6882_to_fp16)[name = string("op_6883_cast_fp16")]; + tensor var_6884_cast_fp16 = mul(x = v_55_cast_fp16, y = upd_55_to_fp16)[name = string("op_6884_cast_fp16")]; + tensor value_55_cast_fp16 = add(x = var_6883_cast_fp16, y = var_6884_cast_fp16)[name = string("value_55_cast_fp16")]; + tensor var_6886 = const()[name = string("op_6886"), val = tensor([1, 8, 128, 16])]; + tensor kh_109_cast_fp16 = reshape(shape = var_6886, x = key_55_cast_fp16)[name = string("kh_109_cast_fp16")]; + tensor var_6888 = const()[name = string("op_6888"), val = tensor([1, 8, 128, 16])]; + tensor vh_109_cast_fp16 = reshape(shape = var_6888, x = value_55_cast_fp16)[name = string("vh_109_cast_fp16")]; + tensor transpose_108_perm_0 = const()[name = string("transpose_108_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_54_reps_0 = const()[name = string("tile_54_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_108_cast_fp16 = transpose(perm = transpose_108_perm_0, x = kh_109_cast_fp16)[name = string("transpose_317")]; + tensor tile_54_cast_fp16 = tile(reps = tile_54_reps_0, x = transpose_108_cast_fp16)[name = string("tile_54_cast_fp16")]; + tensor concat_136 = const()[name = string("concat_136"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_108_cast_fp16 = reshape(shape = concat_136, x = tile_54_cast_fp16)[name = string("reshape_108_cast_fp16")]; + tensor transpose_109_perm_0 = const()[name = string("transpose_109_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_137 = const()[name = string("concat_137"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_109_cast_fp16 = transpose(perm = transpose_109_perm_0, x = reshape_108_cast_fp16)[name = string("transpose_316")]; + tensor reshape_109_cast_fp16 = reshape(shape = concat_137, x = transpose_109_cast_fp16)[name = string("reshape_109_cast_fp16")]; + tensor transpose_110_perm_0 = const()[name = string("transpose_110_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_55_reps_0 = const()[name = string("tile_55_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_110_cast_fp16 = transpose(perm = transpose_110_perm_0, x = vh_109_cast_fp16)[name = string("transpose_315")]; + tensor tile_55_cast_fp16 = tile(reps = tile_55_reps_0, x = transpose_110_cast_fp16)[name = string("tile_55_cast_fp16")]; + tensor concat_138 = const()[name = string("concat_138"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_110_cast_fp16 = reshape(shape = concat_138, x = tile_55_cast_fp16)[name = string("reshape_110_cast_fp16")]; + tensor transpose_111_perm_0 = const()[name = string("transpose_111_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_139 = const()[name = string("concat_139"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_111_cast_fp16 = transpose(perm = transpose_111_perm_0, x = reshape_110_cast_fp16)[name = string("transpose_314")]; + tensor reshape_111_cast_fp16 = reshape(shape = concat_139, x = transpose_111_cast_fp16)[name = string("reshape_111_cast_fp16")]; + fp16 var_6892_to_fp16 = const()[name = string("op_6892_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_6893_cast_fp16 = mul(x = q_167_cast_fp16, y = var_6892_to_fp16)[name = string("op_6893_cast_fp16")]; + tensor transpose_425_perm_0 = const()[name = string("transpose_425_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_117_transpose_x_1 = const()[name = string("w_117_transpose_x_1"), val = bool(true)]; + bool w_117_transpose_y_1 = const()[name = string("w_117_transpose_y_1"), val = bool(false)]; + tensor transpose_425_cast_fp16 = transpose(perm = transpose_425_perm_0, x = reshape_109_cast_fp16)[name = string("transpose_313")]; + tensor w_117_cast_fp16 = matmul(transpose_x = w_117_transpose_x_1, transpose_y = w_117_transpose_y_1, x = var_6893_cast_fp16, y = transpose_425_cast_fp16)[name = string("w_117_cast_fp16")]; + tensor pad_55_to_fp16 = const()[name = string("pad_55_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637376)))]; + tensor var_6896_cast_fp16 = add(x = w_117_cast_fp16, y = pad_55_to_fp16)[name = string("op_6896_cast_fp16")]; + tensor w_119_cast_fp16 = softmax(axis = var_6771, x = var_6896_cast_fp16)[name = string("w_119_cast_fp16")]; + tensor transpose_426_perm_0 = const()[name = string("transpose_426_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_55_transpose_x_1 = const()[name = string("attn_55_transpose_x_1"), val = bool(false)]; + bool attn_55_transpose_y_1 = const()[name = string("attn_55_transpose_y_1"), val = bool(true)]; + tensor transpose_426_cast_fp16 = transpose(perm = transpose_426_perm_0, x = reshape_111_cast_fp16)[name = string("transpose_312")]; + tensor attn_55_cast_fp16 = matmul(transpose_x = attn_55_transpose_x_1, transpose_y = attn_55_transpose_y_1, x = transpose_426_cast_fp16, y = w_119_cast_fp16)[name = string("attn_55_cast_fp16")]; + tensor var_6900 = const()[name = string("op_6900"), val = tensor([1, 2048, 1, 1])]; + tensor input_289_cast_fp16 = reshape(shape = var_6900, x = attn_55_cast_fp16)[name = string("input_289_cast_fp16")]; + string attn_output_55_pad_type_0 = const()[name = string("attn_output_55_pad_type_0"), val = string("valid")]; + tensor attn_output_55_strides_0 = const()[name = string("attn_output_55_strides_0"), val = tensor([1, 1])]; + tensor attn_output_55_pad_0 = const()[name = string("attn_output_55_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_55_dilations_0 = const()[name = string("attn_output_55_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_55_groups_0 = const()[name = string("attn_output_55_groups_0"), val = int32(1)]; + tensor attn_output_55_cast_fp16 = conv(dilations = attn_output_55_dilations_0, groups = attn_output_55_groups_0, pad = attn_output_55_pad_0, pad_type = attn_output_55_pad_type_0, strides = attn_output_55_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_289_cast_fp16)[name = string("attn_output_55_cast_fp16")]; + tensor x_211_cast_fp16 = add(x = x_205_cast_fp16, y = attn_output_55_cast_fp16)[name = string("x_211_cast_fp16")]; + tensor var_6914_cast_fp16 = mul(x = x_211_cast_fp16, y = x_211_cast_fp16)[name = string("op_6914_cast_fp16")]; + tensor variance_231_axes_0 = const()[name = string("variance_231_axes_0"), val = tensor([1])]; + bool variance_231_keep_dims_0 = const()[name = string("variance_231_keep_dims_0"), val = bool(true)]; + tensor variance_231_cast_fp16 = reduce_mean(axes = variance_231_axes_0, keep_dims = variance_231_keep_dims_0, x = var_6914_cast_fp16)[name = string("variance_231_cast_fp16")]; + fp16 var_6917_to_fp16 = const()[name = string("op_6917_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6918_cast_fp16 = add(x = variance_231_cast_fp16, y = var_6917_to_fp16)[name = string("op_6918_cast_fp16")]; + fp32 var_6919_epsilon_0 = const()[name = string("op_6919_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6919_cast_fp16 = rsqrt(epsilon = var_6919_epsilon_0, x = var_6918_cast_fp16)[name = string("op_6919_cast_fp16")]; + tensor var_6920_cast_fp16 = mul(x = x_211_cast_fp16, y = var_6919_cast_fp16)[name = string("op_6920_cast_fp16")]; + tensor input_291_cast_fp16 = mul(x = var_6920_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_291_cast_fp16")]; + string input_293_pad_type_0 = const()[name = string("input_293_pad_type_0"), val = string("valid")]; + tensor input_293_strides_0 = const()[name = string("input_293_strides_0"), val = tensor([1, 1])]; + tensor input_293_pad_0 = const()[name = string("input_293_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_293_dilations_0 = const()[name = string("input_293_dilations_0"), val = tensor([1, 1])]; + int32 input_293_groups_0 = const()[name = string("input_293_groups_0"), val = int32(1)]; + tensor input_293_cast_fp16 = conv(dilations = input_293_dilations_0, groups = input_293_groups_0, pad = input_293_pad_0, pad_type = input_293_pad_type_0, strides = input_293_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_291_cast_fp16)[name = string("input_293_cast_fp16")]; + tensor var_6928_cast_fp16 = silu(x = input_293_cast_fp16)[name = string("op_6928_cast_fp16")]; + string var_6934_pad_type_0 = const()[name = string("op_6934_pad_type_0"), val = string("valid")]; + tensor var_6934_strides_0 = const()[name = string("op_6934_strides_0"), val = tensor([1, 1])]; + tensor var_6934_pad_0 = const()[name = string("op_6934_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6934_dilations_0 = const()[name = string("op_6934_dilations_0"), val = tensor([1, 1])]; + int32 var_6934_groups_0 = const()[name = string("op_6934_groups_0"), val = int32(1)]; + tensor var_6934_cast_fp16 = conv(dilations = var_6934_dilations_0, groups = var_6934_groups_0, pad = var_6934_pad_0, pad_type = var_6934_pad_type_0, strides = var_6934_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_291_cast_fp16)[name = string("op_6934_cast_fp16")]; + tensor input_295_cast_fp16 = mul(x = var_6928_cast_fp16, y = var_6934_cast_fp16)[name = string("input_295_cast_fp16")]; + string h_55_pad_type_0 = const()[name = string("h_55_pad_type_0"), val = string("valid")]; + tensor h_55_strides_0 = const()[name = string("h_55_strides_0"), val = tensor([1, 1])]; + tensor h_55_pad_0 = const()[name = string("h_55_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_55_dilations_0 = const()[name = string("h_55_dilations_0"), val = tensor([1, 1])]; + int32 h_55_groups_0 = const()[name = string("h_55_groups_0"), val = int32(1)]; + tensor h_55_cast_fp16 = conv(dilations = h_55_dilations_0, groups = h_55_groups_0, pad = h_55_pad_0, pad_type = h_55_pad_type_0, strides = h_55_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_295_cast_fp16)[name = string("h_55_cast_fp16")]; + tensor x_213_cast_fp16 = add(x = x_211_cast_fp16, y = h_55_cast_fp16)[name = string("x_213_cast_fp16")]; + tensor key_cache_57_begin_0 = const()[name = string("key_cache_57_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_57_end_0 = const()[name = string("key_cache_57_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_57_end_mask_0 = const()[name = string("key_cache_57_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_57_cast_fp16 = slice_by_index(begin = key_cache_57_begin_0, end = key_cache_57_end_0, end_mask = key_cache_57_end_mask_0, x = layer_key_caches_11_cast_fp16)[name = string("key_cache_57_cast_fp16")]; + tensor value_cache_57_begin_0 = const()[name = string("value_cache_57_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_57_end_0 = const()[name = string("value_cache_57_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_57_end_mask_0 = const()[name = string("value_cache_57_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_57_cast_fp16 = slice_by_index(begin = value_cache_57_begin_0, end = value_cache_57_end_0, end_mask = value_cache_57_end_mask_0, x = layer_value_caches_11_cast_fp16)[name = string("value_cache_57_cast_fp16")]; + int32 var_6987 = const()[name = string("op_6987"), val = int32(2)]; + int32 var_6991 = const()[name = string("op_6991"), val = int32(3)]; + tensor var_7006_cast_fp16 = mul(x = x_213_cast_fp16, y = x_213_cast_fp16)[name = string("op_7006_cast_fp16")]; + tensor variance_233_axes_0 = const()[name = string("variance_233_axes_0"), val = tensor([1])]; + bool variance_233_keep_dims_0 = const()[name = string("variance_233_keep_dims_0"), val = bool(true)]; + tensor variance_233_cast_fp16 = reduce_mean(axes = variance_233_axes_0, keep_dims = variance_233_keep_dims_0, x = var_7006_cast_fp16)[name = string("variance_233_cast_fp16")]; + fp16 var_7009_to_fp16 = const()[name = string("op_7009_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7010_cast_fp16 = add(x = variance_233_cast_fp16, y = var_7009_to_fp16)[name = string("op_7010_cast_fp16")]; + fp32 var_7011_epsilon_0 = const()[name = string("op_7011_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7011_cast_fp16 = rsqrt(epsilon = var_7011_epsilon_0, x = var_7010_cast_fp16)[name = string("op_7011_cast_fp16")]; + tensor var_7012_cast_fp16 = mul(x = x_213_cast_fp16, y = var_7011_cast_fp16)[name = string("op_7012_cast_fp16")]; + tensor input_297_cast_fp16 = mul(x = var_7012_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_297_cast_fp16")]; + string q_169_pad_type_0 = const()[name = string("q_169_pad_type_0"), val = string("valid")]; + tensor q_169_strides_0 = const()[name = string("q_169_strides_0"), val = tensor([1, 1])]; + tensor q_169_pad_0 = const()[name = string("q_169_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_169_dilations_0 = const()[name = string("q_169_dilations_0"), val = tensor([1, 1])]; + int32 q_169_groups_0 = const()[name = string("q_169_groups_0"), val = int32(1)]; + tensor q_169_cast_fp16 = conv(dilations = q_169_dilations_0, groups = q_169_groups_0, pad = q_169_pad_0, pad_type = q_169_pad_type_0, strides = q_169_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_297_cast_fp16)[name = string("q_169_cast_fp16")]; + string k_169_pad_type_0 = const()[name = string("k_169_pad_type_0"), val = string("valid")]; + tensor k_169_strides_0 = const()[name = string("k_169_strides_0"), val = tensor([1, 1])]; + tensor k_169_pad_0 = const()[name = string("k_169_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_169_dilations_0 = const()[name = string("k_169_dilations_0"), val = tensor([1, 1])]; + int32 k_169_groups_0 = const()[name = string("k_169_groups_0"), val = int32(1)]; + tensor k_169_cast_fp16 = conv(dilations = k_169_dilations_0, groups = k_169_groups_0, pad = k_169_pad_0, pad_type = k_169_pad_type_0, strides = k_169_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_297_cast_fp16)[name = string("k_169_cast_fp16")]; + string v_57_pad_type_0 = const()[name = string("v_57_pad_type_0"), val = string("valid")]; + tensor v_57_strides_0 = const()[name = string("v_57_strides_0"), val = tensor([1, 1])]; + tensor v_57_pad_0 = const()[name = string("v_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_57_dilations_0 = const()[name = string("v_57_dilations_0"), val = tensor([1, 1])]; + int32 v_57_groups_0 = const()[name = string("v_57_groups_0"), val = int32(1)]; + tensor v_57_cast_fp16 = conv(dilations = v_57_dilations_0, groups = v_57_groups_0, pad = v_57_pad_0, pad_type = v_57_pad_type_0, strides = v_57_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_297_cast_fp16)[name = string("v_57_cast_fp16")]; + tensor var_7046 = const()[name = string("op_7046"), val = tensor([16, 128, 1, 1])]; + tensor x_215_cast_fp16 = reshape(shape = var_7046, x = q_169_cast_fp16)[name = string("x_215_cast_fp16")]; + tensor var_7049_cast_fp16 = mul(x = x_215_cast_fp16, y = x_215_cast_fp16)[name = string("op_7049_cast_fp16")]; + tensor variance_235_axes_0 = const()[name = string("variance_235_axes_0"), val = tensor([1])]; + bool variance_235_keep_dims_0 = const()[name = string("variance_235_keep_dims_0"), val = bool(true)]; + tensor variance_235_cast_fp16 = reduce_mean(axes = variance_235_axes_0, keep_dims = variance_235_keep_dims_0, x = var_7049_cast_fp16)[name = string("variance_235_cast_fp16")]; + fp16 var_7052_to_fp16 = const()[name = string("op_7052_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7053_cast_fp16 = add(x = variance_235_cast_fp16, y = var_7052_to_fp16)[name = string("op_7053_cast_fp16")]; + fp32 var_7054_epsilon_0 = const()[name = string("op_7054_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7054_cast_fp16 = rsqrt(epsilon = var_7054_epsilon_0, x = var_7053_cast_fp16)[name = string("op_7054_cast_fp16")]; + tensor var_7055_cast_fp16 = mul(x = x_215_cast_fp16, y = var_7054_cast_fp16)[name = string("op_7055_cast_fp16")]; + tensor q_171_cast_fp16 = mul(x = var_7055_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_171_cast_fp16")]; + tensor var_7057 = const()[name = string("op_7057"), val = tensor([8, 128, 1, 1])]; + tensor x_217_cast_fp16 = reshape(shape = var_7057, x = k_169_cast_fp16)[name = string("x_217_cast_fp16")]; + tensor var_7060_cast_fp16 = mul(x = x_217_cast_fp16, y = x_217_cast_fp16)[name = string("op_7060_cast_fp16")]; + tensor variance_237_axes_0 = const()[name = string("variance_237_axes_0"), val = tensor([1])]; + bool variance_237_keep_dims_0 = const()[name = string("variance_237_keep_dims_0"), val = bool(true)]; + tensor variance_237_cast_fp16 = reduce_mean(axes = variance_237_axes_0, keep_dims = variance_237_keep_dims_0, x = var_7060_cast_fp16)[name = string("variance_237_cast_fp16")]; + fp16 var_7063_to_fp16 = const()[name = string("op_7063_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7064_cast_fp16 = add(x = variance_237_cast_fp16, y = var_7063_to_fp16)[name = string("op_7064_cast_fp16")]; + fp32 var_7065_epsilon_0 = const()[name = string("op_7065_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7065_cast_fp16 = rsqrt(epsilon = var_7065_epsilon_0, x = var_7064_cast_fp16)[name = string("op_7065_cast_fp16")]; + tensor var_7066_cast_fp16 = mul(x = x_217_cast_fp16, y = var_7065_cast_fp16)[name = string("op_7066_cast_fp16")]; + tensor k_171_cast_fp16 = mul(x = var_7066_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_171_cast_fp16")]; + tensor var_7068 = const()[name = string("op_7068"), val = tensor([1, 16, 128, 1])]; + tensor z_113_cast_fp16 = reshape(shape = var_7068, x = q_171_cast_fp16)[name = string("z_113_cast_fp16")]; + tensor var_7070 = const()[name = string("op_7070"), val = tensor([1, 8, 128, 1])]; + tensor z_115_cast_fp16 = reshape(shape = var_7070, x = k_171_cast_fp16)[name = string("z_115_cast_fp16")]; + tensor z1_113_begin_0 = const()[name = string("z1_113_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_113_end_0 = const()[name = string("z1_113_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_113_end_mask_0 = const()[name = string("z1_113_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_113_cast_fp16 = slice_by_index(begin = z1_113_begin_0, end = z1_113_end_0, end_mask = z1_113_end_mask_0, x = z_113_cast_fp16)[name = string("z1_113_cast_fp16")]; + tensor z2_113_begin_0 = const()[name = string("z2_113_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_113_end_0 = const()[name = string("z2_113_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_113_end_mask_0 = const()[name = string("z2_113_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_113_cast_fp16 = slice_by_index(begin = z2_113_begin_0, end = z2_113_end_0, end_mask = z2_113_end_mask_0, x = z_113_cast_fp16)[name = string("z2_113_cast_fp16")]; + tensor var_7078_cast_fp16 = mul(x = z_113_cast_fp16, y = cos_51_to_fp16)[name = string("op_7078_cast_fp16")]; + fp16 const_62_promoted_to_fp16 = const()[name = string("const_62_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7079_cast_fp16 = mul(x = z2_113_cast_fp16, y = const_62_promoted_to_fp16)[name = string("op_7079_cast_fp16")]; + bool var_7081_interleave_0 = const()[name = string("op_7081_interleave_0"), val = bool(false)]; + tensor var_7081_cast_fp16 = concat(axis = var_6987, interleave = var_7081_interleave_0, values = (var_7079_cast_fp16, z1_113_cast_fp16))[name = string("op_7081_cast_fp16")]; + tensor var_7082_cast_fp16 = mul(x = var_7081_cast_fp16, y = sin_51_to_fp16)[name = string("op_7082_cast_fp16")]; + tensor q_173_cast_fp16 = add(x = var_7078_cast_fp16, y = var_7082_cast_fp16)[name = string("q_173_cast_fp16")]; + tensor z1_115_begin_0 = const()[name = string("z1_115_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_115_end_0 = const()[name = string("z1_115_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_115_end_mask_0 = const()[name = string("z1_115_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_115_cast_fp16 = slice_by_index(begin = z1_115_begin_0, end = z1_115_end_0, end_mask = z1_115_end_mask_0, x = z_115_cast_fp16)[name = string("z1_115_cast_fp16")]; + tensor z2_115_begin_0 = const()[name = string("z2_115_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_115_end_0 = const()[name = string("z2_115_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_115_end_mask_0 = const()[name = string("z2_115_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_115_cast_fp16 = slice_by_index(begin = z2_115_begin_0, end = z2_115_end_0, end_mask = z2_115_end_mask_0, x = z_115_cast_fp16)[name = string("z2_115_cast_fp16")]; + tensor var_7090_cast_fp16 = mul(x = z_115_cast_fp16, y = cos_51_to_fp16)[name = string("op_7090_cast_fp16")]; + fp16 const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7091_cast_fp16 = mul(x = z2_115_cast_fp16, y = const_63_promoted_to_fp16)[name = string("op_7091_cast_fp16")]; + bool var_7093_interleave_0 = const()[name = string("op_7093_interleave_0"), val = bool(false)]; + tensor var_7093_cast_fp16 = concat(axis = var_6987, interleave = var_7093_interleave_0, values = (var_7091_cast_fp16, z1_115_cast_fp16))[name = string("op_7093_cast_fp16")]; + tensor var_7094_cast_fp16 = mul(x = var_7093_cast_fp16, y = sin_51_to_fp16)[name = string("op_7094_cast_fp16")]; + tensor k_173_cast_fp16 = add(x = var_7090_cast_fp16, y = var_7094_cast_fp16)[name = string("k_173_cast_fp16")]; + tensor var_7096 = const()[name = string("op_7096"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_57_cast_fp16 = reshape(shape = var_7096, x = k_173_cast_fp16)[name = string("cur_key_57_cast_fp16")]; + tensor var_7098_to_fp16 = const()[name = string("op_7098_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637120)))]; + tensor var_7099_cast_fp16 = mul(x = key_cache_57_cast_fp16, y = var_7098_to_fp16)[name = string("op_7099_cast_fp16")]; + tensor upd_57_to_fp16 = const()[name = string("upd_57_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637248)))]; + tensor var_7100_cast_fp16 = mul(x = cur_key_57_cast_fp16, y = upd_57_to_fp16)[name = string("op_7100_cast_fp16")]; + tensor key_57_cast_fp16 = add(x = var_7099_cast_fp16, y = var_7100_cast_fp16)[name = string("key_57_cast_fp16")]; + tensor var_7102_to_fp16 = const()[name = string("op_7102_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637120)))]; + tensor var_7103_cast_fp16 = mul(x = value_cache_57_cast_fp16, y = var_7102_to_fp16)[name = string("op_7103_cast_fp16")]; + tensor var_7104_cast_fp16 = mul(x = v_57_cast_fp16, y = upd_57_to_fp16)[name = string("op_7104_cast_fp16")]; + tensor value_57_cast_fp16 = add(x = var_7103_cast_fp16, y = var_7104_cast_fp16)[name = string("value_57_cast_fp16")]; + tensor var_7106 = const()[name = string("op_7106"), val = tensor([1, 8, 128, 16])]; + tensor kh_113_cast_fp16 = reshape(shape = var_7106, x = key_57_cast_fp16)[name = string("kh_113_cast_fp16")]; + tensor var_7108 = const()[name = string("op_7108"), val = tensor([1, 8, 128, 16])]; + tensor vh_113_cast_fp16 = reshape(shape = var_7108, x = value_57_cast_fp16)[name = string("vh_113_cast_fp16")]; + tensor transpose_112_perm_0 = const()[name = string("transpose_112_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_56_reps_0 = const()[name = string("tile_56_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_112_cast_fp16 = transpose(perm = transpose_112_perm_0, x = kh_113_cast_fp16)[name = string("transpose_311")]; + tensor tile_56_cast_fp16 = tile(reps = tile_56_reps_0, x = transpose_112_cast_fp16)[name = string("tile_56_cast_fp16")]; + tensor concat_140 = const()[name = string("concat_140"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_112_cast_fp16 = reshape(shape = concat_140, x = tile_56_cast_fp16)[name = string("reshape_112_cast_fp16")]; + tensor transpose_113_perm_0 = const()[name = string("transpose_113_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_141 = const()[name = string("concat_141"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_113_cast_fp16 = transpose(perm = transpose_113_perm_0, x = reshape_112_cast_fp16)[name = string("transpose_310")]; + tensor reshape_113_cast_fp16 = reshape(shape = concat_141, x = transpose_113_cast_fp16)[name = string("reshape_113_cast_fp16")]; + tensor transpose_114_perm_0 = const()[name = string("transpose_114_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_57_reps_0 = const()[name = string("tile_57_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_114_cast_fp16 = transpose(perm = transpose_114_perm_0, x = vh_113_cast_fp16)[name = string("transpose_309")]; + tensor tile_57_cast_fp16 = tile(reps = tile_57_reps_0, x = transpose_114_cast_fp16)[name = string("tile_57_cast_fp16")]; + tensor concat_142 = const()[name = string("concat_142"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_114_cast_fp16 = reshape(shape = concat_142, x = tile_57_cast_fp16)[name = string("reshape_114_cast_fp16")]; + tensor transpose_115_perm_0 = const()[name = string("transpose_115_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_143 = const()[name = string("concat_143"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_115_cast_fp16 = transpose(perm = transpose_115_perm_0, x = reshape_114_cast_fp16)[name = string("transpose_308")]; + tensor reshape_115_cast_fp16 = reshape(shape = concat_143, x = transpose_115_cast_fp16)[name = string("reshape_115_cast_fp16")]; + fp16 var_7112_to_fp16 = const()[name = string("op_7112_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_7113_cast_fp16 = mul(x = q_173_cast_fp16, y = var_7112_to_fp16)[name = string("op_7113_cast_fp16")]; + tensor transpose_429_perm_0 = const()[name = string("transpose_429_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_121_transpose_x_1 = const()[name = string("w_121_transpose_x_1"), val = bool(true)]; + bool w_121_transpose_y_1 = const()[name = string("w_121_transpose_y_1"), val = bool(false)]; + tensor transpose_429_cast_fp16 = transpose(perm = transpose_429_perm_0, x = reshape_113_cast_fp16)[name = string("transpose_307")]; + tensor w_121_cast_fp16 = matmul(transpose_x = w_121_transpose_x_1, transpose_y = w_121_transpose_y_1, x = var_7113_cast_fp16, y = transpose_429_cast_fp16)[name = string("w_121_cast_fp16")]; + tensor pad_57_to_fp16 = const()[name = string("pad_57_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637376)))]; + tensor var_7116_cast_fp16 = add(x = w_121_cast_fp16, y = pad_57_to_fp16)[name = string("op_7116_cast_fp16")]; + tensor w_123_cast_fp16 = softmax(axis = var_6991, x = var_7116_cast_fp16)[name = string("w_123_cast_fp16")]; + tensor transpose_430_perm_0 = const()[name = string("transpose_430_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_57_transpose_x_1 = const()[name = string("attn_57_transpose_x_1"), val = bool(false)]; + bool attn_57_transpose_y_1 = const()[name = string("attn_57_transpose_y_1"), val = bool(true)]; + tensor transpose_430_cast_fp16 = transpose(perm = transpose_430_perm_0, x = reshape_115_cast_fp16)[name = string("transpose_306")]; + tensor attn_57_cast_fp16 = matmul(transpose_x = attn_57_transpose_x_1, transpose_y = attn_57_transpose_y_1, x = transpose_430_cast_fp16, y = w_123_cast_fp16)[name = string("attn_57_cast_fp16")]; + tensor var_7120 = const()[name = string("op_7120"), val = tensor([1, 2048, 1, 1])]; + tensor input_299_cast_fp16 = reshape(shape = var_7120, x = attn_57_cast_fp16)[name = string("input_299_cast_fp16")]; + string attn_output_57_pad_type_0 = const()[name = string("attn_output_57_pad_type_0"), val = string("valid")]; + tensor attn_output_57_strides_0 = const()[name = string("attn_output_57_strides_0"), val = tensor([1, 1])]; + tensor attn_output_57_pad_0 = const()[name = string("attn_output_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_57_dilations_0 = const()[name = string("attn_output_57_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_57_groups_0 = const()[name = string("attn_output_57_groups_0"), val = int32(1)]; + tensor attn_output_57_cast_fp16 = conv(dilations = attn_output_57_dilations_0, groups = attn_output_57_groups_0, pad = attn_output_57_pad_0, pad_type = attn_output_57_pad_type_0, strides = attn_output_57_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_299_cast_fp16)[name = string("attn_output_57_cast_fp16")]; + tensor x_219_cast_fp16 = add(x = x_213_cast_fp16, y = attn_output_57_cast_fp16)[name = string("x_219_cast_fp16")]; + tensor var_7134_cast_fp16 = mul(x = x_219_cast_fp16, y = x_219_cast_fp16)[name = string("op_7134_cast_fp16")]; + tensor variance_239_axes_0 = const()[name = string("variance_239_axes_0"), val = tensor([1])]; + bool variance_239_keep_dims_0 = const()[name = string("variance_239_keep_dims_0"), val = bool(true)]; + tensor variance_239_cast_fp16 = reduce_mean(axes = variance_239_axes_0, keep_dims = variance_239_keep_dims_0, x = var_7134_cast_fp16)[name = string("variance_239_cast_fp16")]; + fp16 var_7137_to_fp16 = const()[name = string("op_7137_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7138_cast_fp16 = add(x = variance_239_cast_fp16, y = var_7137_to_fp16)[name = string("op_7138_cast_fp16")]; + fp32 var_7139_epsilon_0 = const()[name = string("op_7139_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7139_cast_fp16 = rsqrt(epsilon = var_7139_epsilon_0, x = var_7138_cast_fp16)[name = string("op_7139_cast_fp16")]; + tensor var_7140_cast_fp16 = mul(x = x_219_cast_fp16, y = var_7139_cast_fp16)[name = string("op_7140_cast_fp16")]; + tensor input_301_cast_fp16 = mul(x = var_7140_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_301_cast_fp16")]; + string input_303_pad_type_0 = const()[name = string("input_303_pad_type_0"), val = string("valid")]; + tensor input_303_strides_0 = const()[name = string("input_303_strides_0"), val = tensor([1, 1])]; + tensor input_303_pad_0 = const()[name = string("input_303_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_303_dilations_0 = const()[name = string("input_303_dilations_0"), val = tensor([1, 1])]; + int32 input_303_groups_0 = const()[name = string("input_303_groups_0"), val = int32(1)]; + tensor input_303_cast_fp16 = conv(dilations = input_303_dilations_0, groups = input_303_groups_0, pad = input_303_pad_0, pad_type = input_303_pad_type_0, strides = input_303_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_301_cast_fp16)[name = string("input_303_cast_fp16")]; + tensor var_7148_cast_fp16 = silu(x = input_303_cast_fp16)[name = string("op_7148_cast_fp16")]; + string var_7154_pad_type_0 = const()[name = string("op_7154_pad_type_0"), val = string("valid")]; + tensor var_7154_strides_0 = const()[name = string("op_7154_strides_0"), val = tensor([1, 1])]; + tensor var_7154_pad_0 = const()[name = string("op_7154_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7154_dilations_0 = const()[name = string("op_7154_dilations_0"), val = tensor([1, 1])]; + int32 var_7154_groups_0 = const()[name = string("op_7154_groups_0"), val = int32(1)]; + tensor var_7154_cast_fp16 = conv(dilations = var_7154_dilations_0, groups = var_7154_groups_0, pad = var_7154_pad_0, pad_type = var_7154_pad_type_0, strides = var_7154_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_301_cast_fp16)[name = string("op_7154_cast_fp16")]; + tensor input_305_cast_fp16 = mul(x = var_7148_cast_fp16, y = var_7154_cast_fp16)[name = string("input_305_cast_fp16")]; + string h_57_pad_type_0 = const()[name = string("h_57_pad_type_0"), val = string("valid")]; + tensor h_57_strides_0 = const()[name = string("h_57_strides_0"), val = tensor([1, 1])]; + tensor h_57_pad_0 = const()[name = string("h_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_57_dilations_0 = const()[name = string("h_57_dilations_0"), val = tensor([1, 1])]; + int32 h_57_groups_0 = const()[name = string("h_57_groups_0"), val = int32(1)]; + tensor h_57_cast_fp16 = conv(dilations = h_57_dilations_0, groups = h_57_groups_0, pad = h_57_pad_0, pad_type = h_57_pad_type_0, strides = h_57_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_305_cast_fp16)[name = string("h_57_cast_fp16")]; + tensor x_221_cast_fp16 = add(x = x_219_cast_fp16, y = h_57_cast_fp16)[name = string("x_221_cast_fp16")]; + tensor key_cache_59_begin_0 = const()[name = string("key_cache_59_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_59_end_0 = const()[name = string("key_cache_59_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_59_end_mask_0 = const()[name = string("key_cache_59_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_59_cast_fp16 = slice_by_index(begin = key_cache_59_begin_0, end = key_cache_59_end_0, end_mask = key_cache_59_end_mask_0, x = layer_key_caches_11_cast_fp16)[name = string("key_cache_59_cast_fp16")]; + tensor value_cache_59_begin_0 = const()[name = string("value_cache_59_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_59_end_0 = const()[name = string("value_cache_59_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_59_end_mask_0 = const()[name = string("value_cache_59_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_59_cast_fp16 = slice_by_index(begin = value_cache_59_begin_0, end = value_cache_59_end_0, end_mask = value_cache_59_end_mask_0, x = layer_value_caches_11_cast_fp16)[name = string("value_cache_59_cast_fp16")]; + int32 var_7207 = const()[name = string("op_7207"), val = int32(2)]; + int32 var_7211 = const()[name = string("op_7211"), val = int32(3)]; + tensor var_7226_cast_fp16 = mul(x = x_221_cast_fp16, y = x_221_cast_fp16)[name = string("op_7226_cast_fp16")]; + tensor variance_241_axes_0 = const()[name = string("variance_241_axes_0"), val = tensor([1])]; + bool variance_241_keep_dims_0 = const()[name = string("variance_241_keep_dims_0"), val = bool(true)]; + tensor variance_241_cast_fp16 = reduce_mean(axes = variance_241_axes_0, keep_dims = variance_241_keep_dims_0, x = var_7226_cast_fp16)[name = string("variance_241_cast_fp16")]; + fp16 var_7229_to_fp16 = const()[name = string("op_7229_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7230_cast_fp16 = add(x = variance_241_cast_fp16, y = var_7229_to_fp16)[name = string("op_7230_cast_fp16")]; + fp32 var_7231_epsilon_0 = const()[name = string("op_7231_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7231_cast_fp16 = rsqrt(epsilon = var_7231_epsilon_0, x = var_7230_cast_fp16)[name = string("op_7231_cast_fp16")]; + tensor var_7232_cast_fp16 = mul(x = x_221_cast_fp16, y = var_7231_cast_fp16)[name = string("op_7232_cast_fp16")]; + tensor input_307_cast_fp16 = mul(x = var_7232_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_307_cast_fp16")]; + string q_175_pad_type_0 = const()[name = string("q_175_pad_type_0"), val = string("valid")]; + tensor q_175_strides_0 = const()[name = string("q_175_strides_0"), val = tensor([1, 1])]; + tensor q_175_pad_0 = const()[name = string("q_175_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_175_dilations_0 = const()[name = string("q_175_dilations_0"), val = tensor([1, 1])]; + int32 q_175_groups_0 = const()[name = string("q_175_groups_0"), val = int32(1)]; + tensor q_175_cast_fp16 = conv(dilations = q_175_dilations_0, groups = q_175_groups_0, pad = q_175_pad_0, pad_type = q_175_pad_type_0, strides = q_175_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_307_cast_fp16)[name = string("q_175_cast_fp16")]; + string k_175_pad_type_0 = const()[name = string("k_175_pad_type_0"), val = string("valid")]; + tensor k_175_strides_0 = const()[name = string("k_175_strides_0"), val = tensor([1, 1])]; + tensor k_175_pad_0 = const()[name = string("k_175_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_175_dilations_0 = const()[name = string("k_175_dilations_0"), val = tensor([1, 1])]; + int32 k_175_groups_0 = const()[name = string("k_175_groups_0"), val = int32(1)]; + tensor k_175_cast_fp16 = conv(dilations = k_175_dilations_0, groups = k_175_groups_0, pad = k_175_pad_0, pad_type = k_175_pad_type_0, strides = k_175_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_307_cast_fp16)[name = string("k_175_cast_fp16")]; + string v_59_pad_type_0 = const()[name = string("v_59_pad_type_0"), val = string("valid")]; + tensor v_59_strides_0 = const()[name = string("v_59_strides_0"), val = tensor([1, 1])]; + tensor v_59_pad_0 = const()[name = string("v_59_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_59_dilations_0 = const()[name = string("v_59_dilations_0"), val = tensor([1, 1])]; + int32 v_59_groups_0 = const()[name = string("v_59_groups_0"), val = int32(1)]; + tensor v_59_cast_fp16 = conv(dilations = v_59_dilations_0, groups = v_59_groups_0, pad = v_59_pad_0, pad_type = v_59_pad_type_0, strides = v_59_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_307_cast_fp16)[name = string("v_59_cast_fp16")]; + tensor var_7266 = const()[name = string("op_7266"), val = tensor([16, 128, 1, 1])]; + tensor x_223_cast_fp16 = reshape(shape = var_7266, x = q_175_cast_fp16)[name = string("x_223_cast_fp16")]; + tensor var_7269_cast_fp16 = mul(x = x_223_cast_fp16, y = x_223_cast_fp16)[name = string("op_7269_cast_fp16")]; + tensor variance_243_axes_0 = const()[name = string("variance_243_axes_0"), val = tensor([1])]; + bool variance_243_keep_dims_0 = const()[name = string("variance_243_keep_dims_0"), val = bool(true)]; + tensor variance_243_cast_fp16 = reduce_mean(axes = variance_243_axes_0, keep_dims = variance_243_keep_dims_0, x = var_7269_cast_fp16)[name = string("variance_243_cast_fp16")]; + fp16 var_7272_to_fp16 = const()[name = string("op_7272_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7273_cast_fp16 = add(x = variance_243_cast_fp16, y = var_7272_to_fp16)[name = string("op_7273_cast_fp16")]; + fp32 var_7274_epsilon_0 = const()[name = string("op_7274_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7274_cast_fp16 = rsqrt(epsilon = var_7274_epsilon_0, x = var_7273_cast_fp16)[name = string("op_7274_cast_fp16")]; + tensor var_7275_cast_fp16 = mul(x = x_223_cast_fp16, y = var_7274_cast_fp16)[name = string("op_7275_cast_fp16")]; + tensor q_177_cast_fp16 = mul(x = var_7275_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_177_cast_fp16")]; + tensor var_7277 = const()[name = string("op_7277"), val = tensor([8, 128, 1, 1])]; + tensor x_225_cast_fp16 = reshape(shape = var_7277, x = k_175_cast_fp16)[name = string("x_225_cast_fp16")]; + tensor var_7280_cast_fp16 = mul(x = x_225_cast_fp16, y = x_225_cast_fp16)[name = string("op_7280_cast_fp16")]; + tensor variance_245_axes_0 = const()[name = string("variance_245_axes_0"), val = tensor([1])]; + bool variance_245_keep_dims_0 = const()[name = string("variance_245_keep_dims_0"), val = bool(true)]; + tensor variance_245_cast_fp16 = reduce_mean(axes = variance_245_axes_0, keep_dims = variance_245_keep_dims_0, x = var_7280_cast_fp16)[name = string("variance_245_cast_fp16")]; + fp16 var_7283_to_fp16 = const()[name = string("op_7283_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7284_cast_fp16 = add(x = variance_245_cast_fp16, y = var_7283_to_fp16)[name = string("op_7284_cast_fp16")]; + fp32 var_7285_epsilon_0 = const()[name = string("op_7285_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7285_cast_fp16 = rsqrt(epsilon = var_7285_epsilon_0, x = var_7284_cast_fp16)[name = string("op_7285_cast_fp16")]; + tensor var_7286_cast_fp16 = mul(x = x_225_cast_fp16, y = var_7285_cast_fp16)[name = string("op_7286_cast_fp16")]; + tensor k_177_cast_fp16 = mul(x = var_7286_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_177_cast_fp16")]; + tensor var_7288 = const()[name = string("op_7288"), val = tensor([1, 16, 128, 1])]; + tensor z_117_cast_fp16 = reshape(shape = var_7288, x = q_177_cast_fp16)[name = string("z_117_cast_fp16")]; + tensor var_7290 = const()[name = string("op_7290"), val = tensor([1, 8, 128, 1])]; + tensor z_119_cast_fp16 = reshape(shape = var_7290, x = k_177_cast_fp16)[name = string("z_119_cast_fp16")]; + tensor z1_117_begin_0 = const()[name = string("z1_117_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_117_end_0 = const()[name = string("z1_117_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_117_end_mask_0 = const()[name = string("z1_117_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_117_cast_fp16 = slice_by_index(begin = z1_117_begin_0, end = z1_117_end_0, end_mask = z1_117_end_mask_0, x = z_117_cast_fp16)[name = string("z1_117_cast_fp16")]; + tensor z2_117_begin_0 = const()[name = string("z2_117_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_117_end_0 = const()[name = string("z2_117_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_117_end_mask_0 = const()[name = string("z2_117_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_117_cast_fp16 = slice_by_index(begin = z2_117_begin_0, end = z2_117_end_0, end_mask = z2_117_end_mask_0, x = z_117_cast_fp16)[name = string("z2_117_cast_fp16")]; + tensor var_7298_cast_fp16 = mul(x = z_117_cast_fp16, y = cos_51_to_fp16)[name = string("op_7298_cast_fp16")]; + fp16 const_64_promoted_to_fp16 = const()[name = string("const_64_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7299_cast_fp16 = mul(x = z2_117_cast_fp16, y = const_64_promoted_to_fp16)[name = string("op_7299_cast_fp16")]; + bool var_7301_interleave_0 = const()[name = string("op_7301_interleave_0"), val = bool(false)]; + tensor var_7301_cast_fp16 = concat(axis = var_7207, interleave = var_7301_interleave_0, values = (var_7299_cast_fp16, z1_117_cast_fp16))[name = string("op_7301_cast_fp16")]; + tensor var_7302_cast_fp16 = mul(x = var_7301_cast_fp16, y = sin_51_to_fp16)[name = string("op_7302_cast_fp16")]; + tensor q_179_cast_fp16 = add(x = var_7298_cast_fp16, y = var_7302_cast_fp16)[name = string("q_179_cast_fp16")]; + tensor z1_119_begin_0 = const()[name = string("z1_119_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_119_end_0 = const()[name = string("z1_119_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_119_end_mask_0 = const()[name = string("z1_119_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_119_cast_fp16 = slice_by_index(begin = z1_119_begin_0, end = z1_119_end_0, end_mask = z1_119_end_mask_0, x = z_119_cast_fp16)[name = string("z1_119_cast_fp16")]; + tensor z2_119_begin_0 = const()[name = string("z2_119_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_119_end_0 = const()[name = string("z2_119_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_119_end_mask_0 = const()[name = string("z2_119_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_119_cast_fp16 = slice_by_index(begin = z2_119_begin_0, end = z2_119_end_0, end_mask = z2_119_end_mask_0, x = z_119_cast_fp16)[name = string("z2_119_cast_fp16")]; + tensor var_7310_cast_fp16 = mul(x = z_119_cast_fp16, y = cos_51_to_fp16)[name = string("op_7310_cast_fp16")]; + fp16 const_65_promoted_to_fp16 = const()[name = string("const_65_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7311_cast_fp16 = mul(x = z2_119_cast_fp16, y = const_65_promoted_to_fp16)[name = string("op_7311_cast_fp16")]; + bool var_7313_interleave_0 = const()[name = string("op_7313_interleave_0"), val = bool(false)]; + tensor var_7313_cast_fp16 = concat(axis = var_7207, interleave = var_7313_interleave_0, values = (var_7311_cast_fp16, z1_119_cast_fp16))[name = string("op_7313_cast_fp16")]; + tensor var_7314_cast_fp16 = mul(x = var_7313_cast_fp16, y = sin_51_to_fp16)[name = string("op_7314_cast_fp16")]; + tensor k_179_cast_fp16 = add(x = var_7310_cast_fp16, y = var_7314_cast_fp16)[name = string("k_179_cast_fp16")]; + tensor var_7316 = const()[name = string("op_7316"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_59_cast_fp16 = reshape(shape = var_7316, x = k_179_cast_fp16)[name = string("cur_key_59_cast_fp16")]; + tensor var_7318_to_fp16 = const()[name = string("op_7318_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637120)))]; + tensor var_7319_cast_fp16 = mul(x = key_cache_59_cast_fp16, y = var_7318_to_fp16)[name = string("op_7319_cast_fp16")]; + tensor upd_59_to_fp16 = const()[name = string("upd_59_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637248)))]; + tensor var_7320_cast_fp16 = mul(x = cur_key_59_cast_fp16, y = upd_59_to_fp16)[name = string("op_7320_cast_fp16")]; + tensor key_59_cast_fp16 = add(x = var_7319_cast_fp16, y = var_7320_cast_fp16)[name = string("key_59_cast_fp16")]; + tensor var_7322_to_fp16 = const()[name = string("op_7322_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637120)))]; + tensor var_7323_cast_fp16 = mul(x = value_cache_59_cast_fp16, y = var_7322_to_fp16)[name = string("op_7323_cast_fp16")]; + tensor var_7324_cast_fp16 = mul(x = v_59_cast_fp16, y = upd_59_to_fp16)[name = string("op_7324_cast_fp16")]; + tensor value_59_cast_fp16 = add(x = var_7323_cast_fp16, y = var_7324_cast_fp16)[name = string("value_59_cast_fp16")]; + tensor var_7326 = const()[name = string("op_7326"), val = tensor([1, 8, 128, 16])]; + tensor kh_117_cast_fp16 = reshape(shape = var_7326, x = key_59_cast_fp16)[name = string("kh_117_cast_fp16")]; + tensor var_7328 = const()[name = string("op_7328"), val = tensor([1, 8, 128, 16])]; + tensor vh_117_cast_fp16 = reshape(shape = var_7328, x = value_59_cast_fp16)[name = string("vh_117_cast_fp16")]; + tensor transpose_116_perm_0 = const()[name = string("transpose_116_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_58_reps_0 = const()[name = string("tile_58_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_116_cast_fp16 = transpose(perm = transpose_116_perm_0, x = kh_117_cast_fp16)[name = string("transpose_305")]; + tensor tile_58_cast_fp16 = tile(reps = tile_58_reps_0, x = transpose_116_cast_fp16)[name = string("tile_58_cast_fp16")]; + tensor concat_144 = const()[name = string("concat_144"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_116_cast_fp16 = reshape(shape = concat_144, x = tile_58_cast_fp16)[name = string("reshape_116_cast_fp16")]; + tensor transpose_117_perm_0 = const()[name = string("transpose_117_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_145 = const()[name = string("concat_145"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_117_cast_fp16 = transpose(perm = transpose_117_perm_0, x = reshape_116_cast_fp16)[name = string("transpose_304")]; + tensor reshape_117_cast_fp16 = reshape(shape = concat_145, x = transpose_117_cast_fp16)[name = string("reshape_117_cast_fp16")]; + tensor transpose_118_perm_0 = const()[name = string("transpose_118_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_59_reps_0 = const()[name = string("tile_59_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_118_cast_fp16 = transpose(perm = transpose_118_perm_0, x = vh_117_cast_fp16)[name = string("transpose_303")]; + tensor tile_59_cast_fp16 = tile(reps = tile_59_reps_0, x = transpose_118_cast_fp16)[name = string("tile_59_cast_fp16")]; + tensor concat_146 = const()[name = string("concat_146"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_118_cast_fp16 = reshape(shape = concat_146, x = tile_59_cast_fp16)[name = string("reshape_118_cast_fp16")]; + tensor transpose_119_perm_0 = const()[name = string("transpose_119_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_147 = const()[name = string("concat_147"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_119_cast_fp16 = transpose(perm = transpose_119_perm_0, x = reshape_118_cast_fp16)[name = string("transpose_302")]; + tensor reshape_119_cast_fp16 = reshape(shape = concat_147, x = transpose_119_cast_fp16)[name = string("reshape_119_cast_fp16")]; + fp16 var_7332_to_fp16 = const()[name = string("op_7332_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_7333_cast_fp16 = mul(x = q_179_cast_fp16, y = var_7332_to_fp16)[name = string("op_7333_cast_fp16")]; + tensor transpose_433_perm_0 = const()[name = string("transpose_433_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_125_transpose_x_1 = const()[name = string("w_125_transpose_x_1"), val = bool(true)]; + bool w_125_transpose_y_1 = const()[name = string("w_125_transpose_y_1"), val = bool(false)]; + tensor transpose_433_cast_fp16 = transpose(perm = transpose_433_perm_0, x = reshape_117_cast_fp16)[name = string("transpose_301")]; + tensor w_125_cast_fp16 = matmul(transpose_x = w_125_transpose_x_1, transpose_y = w_125_transpose_y_1, x = var_7333_cast_fp16, y = transpose_433_cast_fp16)[name = string("w_125_cast_fp16")]; + tensor pad_59_to_fp16 = const()[name = string("pad_59_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637376)))]; + tensor var_7336_cast_fp16 = add(x = w_125_cast_fp16, y = pad_59_to_fp16)[name = string("op_7336_cast_fp16")]; + tensor w_127_cast_fp16 = softmax(axis = var_7211, x = var_7336_cast_fp16)[name = string("w_127_cast_fp16")]; + tensor transpose_434_perm_0 = const()[name = string("transpose_434_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_59_transpose_x_1 = const()[name = string("attn_59_transpose_x_1"), val = bool(false)]; + bool attn_59_transpose_y_1 = const()[name = string("attn_59_transpose_y_1"), val = bool(true)]; + tensor transpose_434_cast_fp16 = transpose(perm = transpose_434_perm_0, x = reshape_119_cast_fp16)[name = string("transpose_300")]; + tensor attn_59_cast_fp16 = matmul(transpose_x = attn_59_transpose_x_1, transpose_y = attn_59_transpose_y_1, x = transpose_434_cast_fp16, y = w_127_cast_fp16)[name = string("attn_59_cast_fp16")]; + tensor var_7340 = const()[name = string("op_7340"), val = tensor([1, 2048, 1, 1])]; + tensor input_309_cast_fp16 = reshape(shape = var_7340, x = attn_59_cast_fp16)[name = string("input_309_cast_fp16")]; + string attn_output_59_pad_type_0 = const()[name = string("attn_output_59_pad_type_0"), val = string("valid")]; + tensor attn_output_59_strides_0 = const()[name = string("attn_output_59_strides_0"), val = tensor([1, 1])]; + tensor attn_output_59_pad_0 = const()[name = string("attn_output_59_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_59_dilations_0 = const()[name = string("attn_output_59_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_59_groups_0 = const()[name = string("attn_output_59_groups_0"), val = int32(1)]; + tensor attn_output_59_cast_fp16 = conv(dilations = attn_output_59_dilations_0, groups = attn_output_59_groups_0, pad = attn_output_59_pad_0, pad_type = attn_output_59_pad_type_0, strides = attn_output_59_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_309_cast_fp16)[name = string("attn_output_59_cast_fp16")]; + tensor x_227_cast_fp16 = add(x = x_221_cast_fp16, y = attn_output_59_cast_fp16)[name = string("x_227_cast_fp16")]; + tensor var_7354_cast_fp16 = mul(x = x_227_cast_fp16, y = x_227_cast_fp16)[name = string("op_7354_cast_fp16")]; + tensor variance_247_axes_0 = const()[name = string("variance_247_axes_0"), val = tensor([1])]; + bool variance_247_keep_dims_0 = const()[name = string("variance_247_keep_dims_0"), val = bool(true)]; + tensor variance_247_cast_fp16 = reduce_mean(axes = variance_247_axes_0, keep_dims = variance_247_keep_dims_0, x = var_7354_cast_fp16)[name = string("variance_247_cast_fp16")]; + fp16 var_7357_to_fp16 = const()[name = string("op_7357_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7358_cast_fp16 = add(x = variance_247_cast_fp16, y = var_7357_to_fp16)[name = string("op_7358_cast_fp16")]; + fp32 var_7359_epsilon_0 = const()[name = string("op_7359_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7359_cast_fp16 = rsqrt(epsilon = var_7359_epsilon_0, x = var_7358_cast_fp16)[name = string("op_7359_cast_fp16")]; + tensor var_7360_cast_fp16 = mul(x = x_227_cast_fp16, y = var_7359_cast_fp16)[name = string("op_7360_cast_fp16")]; + tensor input_311_cast_fp16 = mul(x = var_7360_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_311_cast_fp16")]; + string input_313_pad_type_0 = const()[name = string("input_313_pad_type_0"), val = string("valid")]; + tensor input_313_strides_0 = const()[name = string("input_313_strides_0"), val = tensor([1, 1])]; + tensor input_313_pad_0 = const()[name = string("input_313_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_313_dilations_0 = const()[name = string("input_313_dilations_0"), val = tensor([1, 1])]; + int32 input_313_groups_0 = const()[name = string("input_313_groups_0"), val = int32(1)]; + tensor input_313_cast_fp16 = conv(dilations = input_313_dilations_0, groups = input_313_groups_0, pad = input_313_pad_0, pad_type = input_313_pad_type_0, strides = input_313_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_311_cast_fp16)[name = string("input_313_cast_fp16")]; + tensor var_7368_cast_fp16 = silu(x = input_313_cast_fp16)[name = string("op_7368_cast_fp16")]; + string var_7374_pad_type_0 = const()[name = string("op_7374_pad_type_0"), val = string("valid")]; + tensor var_7374_strides_0 = const()[name = string("op_7374_strides_0"), val = tensor([1, 1])]; + tensor var_7374_pad_0 = const()[name = string("op_7374_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7374_dilations_0 = const()[name = string("op_7374_dilations_0"), val = tensor([1, 1])]; + int32 var_7374_groups_0 = const()[name = string("op_7374_groups_0"), val = int32(1)]; + tensor var_7374_cast_fp16 = conv(dilations = var_7374_dilations_0, groups = var_7374_groups_0, pad = var_7374_pad_0, pad_type = var_7374_pad_type_0, strides = var_7374_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_311_cast_fp16)[name = string("op_7374_cast_fp16")]; + tensor input_315_cast_fp16 = mul(x = var_7368_cast_fp16, y = var_7374_cast_fp16)[name = string("input_315_cast_fp16")]; + string h_59_pad_type_0 = const()[name = string("h_59_pad_type_0"), val = string("valid")]; + tensor h_59_strides_0 = const()[name = string("h_59_strides_0"), val = tensor([1, 1])]; + tensor h_59_pad_0 = const()[name = string("h_59_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_59_dilations_0 = const()[name = string("h_59_dilations_0"), val = tensor([1, 1])]; + int32 h_59_groups_0 = const()[name = string("h_59_groups_0"), val = int32(1)]; + tensor h_59_cast_fp16 = conv(dilations = h_59_dilations_0, groups = h_59_groups_0, pad = h_59_pad_0, pad_type = h_59_pad_type_0, strides = h_59_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_315_cast_fp16)[name = string("h_59_cast_fp16")]; + tensor inputs_9_cast_fp16 = add(x = x_227_cast_fp16, y = h_59_cast_fp16)[name = string("inputs_9_cast_fp16")]; + int32 var_7402 = const()[name = string("op_7402"), val = int32(1)]; + bool layer_key_caches_13_interleave_0 = const()[name = string("layer_key_caches_13_interleave_0"), val = bool(false)]; + tensor layer_key_caches_13_cast_fp16 = concat(axis = var_7402, interleave = layer_key_caches_13_interleave_0, values = (key_51_cast_fp16, key_53_cast_fp16, key_55_cast_fp16, key_57_cast_fp16, key_59_cast_fp16))[name = string("layer_key_caches_13_cast_fp16")]; + int32 var_7405 = const()[name = string("op_7405"), val = int32(1)]; + bool layer_value_caches_13_interleave_0 = const()[name = string("layer_value_caches_13_interleave_0"), val = bool(false)]; + tensor layer_value_caches_13_cast_fp16 = concat(axis = var_7405, interleave = layer_value_caches_13_interleave_0, values = (value_51_cast_fp16, value_53_cast_fp16, value_55_cast_fp16, value_57_cast_fp16, value_59_cast_fp16))[name = string("layer_value_caches_13_cast_fp16")]; + tensor inputs_sq_9_cast_fp16 = mul(x = inputs_9_cast_fp16, y = inputs_9_cast_fp16)[name = string("inputs_sq_9_cast_fp16")]; + tensor variance_249_axes_0 = const()[name = string("variance_249_axes_0"), val = tensor([1])]; + bool variance_249_keep_dims_0 = const()[name = string("variance_249_keep_dims_0"), val = bool(true)]; + tensor variance_249_cast_fp16 = reduce_mean(axes = variance_249_axes_0, keep_dims = variance_249_keep_dims_0, x = inputs_sq_9_cast_fp16)[name = string("variance_249_cast_fp16")]; + fp16 var_7415_to_fp16 = const()[name = string("op_7415_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7416_cast_fp16 = add(x = variance_249_cast_fp16, y = var_7415_to_fp16)[name = string("op_7416_cast_fp16")]; + fp32 var_7417_epsilon_0 = const()[name = string("op_7417_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7417_cast_fp16 = rsqrt(epsilon = var_7417_epsilon_0, x = var_7416_cast_fp16)[name = string("op_7417_cast_fp16")]; + tensor hidden_states_9_cast_fp16 = mul(x = inputs_9_cast_fp16, y = var_7417_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; + tensor input_317_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_9_cast_fp16)[name = string("input_317_cast_fp16")]; + string logits_17_pad_type_0 = const()[name = string("logits_17_pad_type_0"), val = string("valid")]; + tensor logits_17_strides_0 = const()[name = string("logits_17_strides_0"), val = tensor([1, 1])]; + tensor logits_17_pad_0 = const()[name = string("logits_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_17_dilations_0 = const()[name = string("logits_17_dilations_0"), val = tensor([1, 1])]; + int32 logits_17_groups_0 = const()[name = string("logits_17_groups_0"), val = int32(1)]; + tensor lm_heads_4_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87097856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89195072))))[name = string("lm_heads_4_weight_to_fp16_palettized")]; + tensor logits_17_cast_fp16 = conv(dilations = logits_17_dilations_0, groups = logits_17_groups_0, pad = logits_17_pad_0, pad_type = logits_17_pad_type_0, strides = logits_17_strides_0, weight = lm_heads_4_weight_to_fp16_palettized, x = input_317_cast_fp16)[name = string("logits_17_cast_fp16")]; + tensor var_7435 = const()[name = string("op_7435"), val = tensor([1, 2048])]; + tensor logits_19_cast_fp16 = reshape(shape = var_7435, x = logits_17_cast_fp16)[name = string("logits_19_cast_fp16")]; + tensor scaled_logits_9_cast_fp16 = real_div(x = logits_19_cast_fp16, y = temperature)[name = string("scaled_logits_9_cast_fp16")]; + int32 var_7445 = const()[name = string("op_7445"), val = int32(100)]; + int32 top_values_9_axis_0 = const()[name = string("top_values_9_axis_0"), val = int32(1)]; + bool top_values_9_ascending_0 = const()[name = string("top_values_9_ascending_0"), val = bool(false)]; + bool top_values_9_sort_0 = const()[name = string("top_values_9_sort_0"), val = bool(true)]; + bool top_values_9_return_indices_0 = const()[name = string("top_values_9_return_indices_0"), val = bool(true)]; + string top_values_9_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_9_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_9_cast_fp16_cast_uint16_0, tensor top_values_9_cast_fp16_cast_uint16_1 = topk(ascending = top_values_9_ascending_0, axis = top_values_9_axis_0, k = var_7445, output_indices_dtype = top_values_9_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_9_return_indices_0, sort = top_values_9_sort_0, x = scaled_logits_9_cast_fp16)[name = string("top_values_9_cast_fp16_cast_uint16")]; + tensor var_7451_cast_fp16 = mul(x = top_values_9_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_7451_cast_fp16")]; + tensor var_7455_cast_fp16 = add(x = var_7451_cast_fp16, y = var_2438_cast_fp16)[name = string("op_7455_cast_fp16")]; + tensor reduce_min_4_axes_0 = const()[name = string("reduce_min_4_axes_0"), val = tensor([1])]; + bool reduce_min_4_keep_dims_0 = const()[name = string("reduce_min_4_keep_dims_0"), val = bool(true)]; + tensor reduce_min_4_cast_fp16 = reduce_min(axes = reduce_min_4_axes_0, keep_dims = reduce_min_4_keep_dims_0, x = var_7455_cast_fp16)[name = string("reduce_min_4_cast_fp16")]; + tensor var_7458_cast_fp16 = greater_equal(x = scaled_logits_9_cast_fp16, y = reduce_min_4_cast_fp16)[name = string("op_7458_cast_fp16")]; + fp16 var_7459_value_0_to_fp16 = const()[name = string("op_7459_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_7459_cast_fp16 = fill_like(ref_tensor = scaled_logits_9_cast_fp16, value = var_7459_value_0_to_fp16)[name = string("op_7459_cast_fp16")]; + tensor masked_logits_9_cast_fp16 = select(a = scaled_logits_9_cast_fp16, b = var_7459_cast_fp16, cond = var_7458_cast_fp16)[name = string("masked_logits_9_cast_fp16")]; + tensor var_7463_begin_0 = const()[name = string("op_7463_begin_0"), val = tensor([4, 0])]; + tensor var_7463_end_0 = const()[name = string("op_7463_end_0"), val = tensor([5, 2048])]; + tensor var_7463_end_mask_0 = const()[name = string("op_7463_end_mask_0"), val = tensor([false, true])]; + tensor var_7463_squeeze_mask_0 = const()[name = string("op_7463_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_7463_cast_fp16 = slice_by_index(begin = var_7463_begin_0, end = var_7463_end_0, end_mask = var_7463_end_mask_0, squeeze_mask = var_7463_squeeze_mask_0, x = gumbel)[name = string("op_7463_cast_fp16")]; + tensor var_7466 = const()[name = string("op_7466"), val = tensor([1, 2048])]; + tensor var_7467_cast_fp16 = reshape(shape = var_7466, x = var_7463_cast_fp16)[name = string("op_7467_cast_fp16")]; + tensor noisy_logits_9_cast_fp16 = add(x = masked_logits_9_cast_fp16, y = var_7467_cast_fp16)[name = string("noisy_logits_9_cast_fp16")]; + int32 code_9_axis_0 = const()[name = string("code_9_axis_0"), val = int32(1)]; + bool code_9_keep_dims_0 = const()[name = string("code_9_keep_dims_0"), val = bool(false)]; + string code_9_output_dtype_0 = const()[name = string("code_9_output_dtype_0"), val = string("int32")]; + tensor code_9_cast_fp16 = reduce_argmax(axis = code_9_axis_0, keep_dims = code_9_keep_dims_0, output_dtype = code_9_output_dtype_0, x = noisy_logits_9_cast_fp16)[name = string("code_9_cast_fp16")]; + int32 var_7478 = const()[name = string("op_7478"), val = int32(8192)]; + tensor input_319 = add(x = code_9_cast_fp16, y = var_7478)[name = string("input_319")]; + int32 code_embed_17_axis_0 = const()[name = string("code_embed_17_axis_0"), val = int32(0)]; + int32 code_embed_17_batch_dims_0 = const()[name = string("code_embed_17_batch_dims_0"), val = int32(0)]; + bool code_embed_17_validate_indices_0 = const()[name = string("code_embed_17_validate_indices_0"), val = bool(false)]; + string input_319_to_uint16_dtype_0 = const()[name = string("input_319_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_319_to_uint16 = cast(dtype = input_319_to_uint16_dtype_0, x = input_319)[name = string("cast_10")]; + tensor code_embed_17_cast_fp16_cast_uint16 = gather(axis = code_embed_17_axis_0, batch_dims = code_embed_17_batch_dims_0, indices = input_319_to_uint16, validate_indices = code_embed_17_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_17_cast_fp16_cast_uint16")]; + tensor var_7482 = const()[name = string("op_7482"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_19_cast_fp16 = reshape(shape = var_7482, x = code_embed_17_cast_fp16_cast_uint16)[name = string("code_embed_19_cast_fp16")]; + tensor embed_sum_11_cast_fp16 = add(x = embed_sum_9_cast_fp16, y = code_embed_19_cast_fp16)[name = string("embed_sum_11_cast_fp16")]; + tensor key_cache_61_begin_0 = const()[name = string("key_cache_61_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_61_end_0 = const()[name = string("key_cache_61_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_61_end_mask_0 = const()[name = string("key_cache_61_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_61_cast_fp16 = slice_by_index(begin = key_cache_61_begin_0, end = key_cache_61_end_0, end_mask = key_cache_61_end_mask_0, x = layer_key_caches_13_cast_fp16)[name = string("key_cache_61_cast_fp16")]; + tensor value_cache_61_begin_0 = const()[name = string("value_cache_61_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_61_end_0 = const()[name = string("value_cache_61_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_61_end_mask_0 = const()[name = string("value_cache_61_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_61_cast_fp16 = slice_by_index(begin = value_cache_61_begin_0, end = value_cache_61_end_0, end_mask = value_cache_61_end_mask_0, x = layer_value_caches_13_cast_fp16)[name = string("value_cache_61_cast_fp16")]; + int32 var_7581 = const()[name = string("op_7581"), val = int32(2)]; + int32 var_7585 = const()[name = string("op_7585"), val = int32(3)]; + tensor var_7600_cast_fp16 = mul(x = code_embed_19_cast_fp16, y = code_embed_19_cast_fp16)[name = string("op_7600_cast_fp16")]; + tensor variance_251_axes_0 = const()[name = string("variance_251_axes_0"), val = tensor([1])]; + bool variance_251_keep_dims_0 = const()[name = string("variance_251_keep_dims_0"), val = bool(true)]; + tensor variance_251_cast_fp16 = reduce_mean(axes = variance_251_axes_0, keep_dims = variance_251_keep_dims_0, x = var_7600_cast_fp16)[name = string("variance_251_cast_fp16")]; + fp16 var_7603_to_fp16 = const()[name = string("op_7603_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7604_cast_fp16 = add(x = variance_251_cast_fp16, y = var_7603_to_fp16)[name = string("op_7604_cast_fp16")]; + fp32 var_7605_epsilon_0 = const()[name = string("op_7605_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7605_cast_fp16 = rsqrt(epsilon = var_7605_epsilon_0, x = var_7604_cast_fp16)[name = string("op_7605_cast_fp16")]; + tensor var_7606_cast_fp16 = mul(x = code_embed_19_cast_fp16, y = var_7605_cast_fp16)[name = string("op_7606_cast_fp16")]; + tensor input_321_cast_fp16 = mul(x = var_7606_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_321_cast_fp16")]; + string q_181_pad_type_0 = const()[name = string("q_181_pad_type_0"), val = string("valid")]; + tensor q_181_strides_0 = const()[name = string("q_181_strides_0"), val = tensor([1, 1])]; + tensor q_181_pad_0 = const()[name = string("q_181_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_181_dilations_0 = const()[name = string("q_181_dilations_0"), val = tensor([1, 1])]; + int32 q_181_groups_0 = const()[name = string("q_181_groups_0"), val = int32(1)]; + tensor q_181_cast_fp16 = conv(dilations = q_181_dilations_0, groups = q_181_groups_0, pad = q_181_pad_0, pad_type = q_181_pad_type_0, strides = q_181_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_321_cast_fp16)[name = string("q_181_cast_fp16")]; + string k_181_pad_type_0 = const()[name = string("k_181_pad_type_0"), val = string("valid")]; + tensor k_181_strides_0 = const()[name = string("k_181_strides_0"), val = tensor([1, 1])]; + tensor k_181_pad_0 = const()[name = string("k_181_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_181_dilations_0 = const()[name = string("k_181_dilations_0"), val = tensor([1, 1])]; + int32 k_181_groups_0 = const()[name = string("k_181_groups_0"), val = int32(1)]; + tensor k_181_cast_fp16 = conv(dilations = k_181_dilations_0, groups = k_181_groups_0, pad = k_181_pad_0, pad_type = k_181_pad_type_0, strides = k_181_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_321_cast_fp16)[name = string("k_181_cast_fp16")]; + string v_61_pad_type_0 = const()[name = string("v_61_pad_type_0"), val = string("valid")]; + tensor v_61_strides_0 = const()[name = string("v_61_strides_0"), val = tensor([1, 1])]; + tensor v_61_pad_0 = const()[name = string("v_61_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_61_dilations_0 = const()[name = string("v_61_dilations_0"), val = tensor([1, 1])]; + int32 v_61_groups_0 = const()[name = string("v_61_groups_0"), val = int32(1)]; + tensor v_61_cast_fp16 = conv(dilations = v_61_dilations_0, groups = v_61_groups_0, pad = v_61_pad_0, pad_type = v_61_pad_type_0, strides = v_61_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_321_cast_fp16)[name = string("v_61_cast_fp16")]; + tensor var_7640 = const()[name = string("op_7640"), val = tensor([16, 128, 1, 1])]; + tensor x_229_cast_fp16 = reshape(shape = var_7640, x = q_181_cast_fp16)[name = string("x_229_cast_fp16")]; + tensor var_7643_cast_fp16 = mul(x = x_229_cast_fp16, y = x_229_cast_fp16)[name = string("op_7643_cast_fp16")]; + tensor variance_253_axes_0 = const()[name = string("variance_253_axes_0"), val = tensor([1])]; + bool variance_253_keep_dims_0 = const()[name = string("variance_253_keep_dims_0"), val = bool(true)]; + tensor variance_253_cast_fp16 = reduce_mean(axes = variance_253_axes_0, keep_dims = variance_253_keep_dims_0, x = var_7643_cast_fp16)[name = string("variance_253_cast_fp16")]; + fp16 var_7646_to_fp16 = const()[name = string("op_7646_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7647_cast_fp16 = add(x = variance_253_cast_fp16, y = var_7646_to_fp16)[name = string("op_7647_cast_fp16")]; + fp32 var_7648_epsilon_0 = const()[name = string("op_7648_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7648_cast_fp16 = rsqrt(epsilon = var_7648_epsilon_0, x = var_7647_cast_fp16)[name = string("op_7648_cast_fp16")]; + tensor var_7649_cast_fp16 = mul(x = x_229_cast_fp16, y = var_7648_cast_fp16)[name = string("op_7649_cast_fp16")]; + tensor q_183_cast_fp16 = mul(x = var_7649_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_183_cast_fp16")]; + tensor var_7651 = const()[name = string("op_7651"), val = tensor([8, 128, 1, 1])]; + tensor x_231_cast_fp16 = reshape(shape = var_7651, x = k_181_cast_fp16)[name = string("x_231_cast_fp16")]; + tensor var_7654_cast_fp16 = mul(x = x_231_cast_fp16, y = x_231_cast_fp16)[name = string("op_7654_cast_fp16")]; + tensor variance_255_axes_0 = const()[name = string("variance_255_axes_0"), val = tensor([1])]; + bool variance_255_keep_dims_0 = const()[name = string("variance_255_keep_dims_0"), val = bool(true)]; + tensor variance_255_cast_fp16 = reduce_mean(axes = variance_255_axes_0, keep_dims = variance_255_keep_dims_0, x = var_7654_cast_fp16)[name = string("variance_255_cast_fp16")]; + fp16 var_7657_to_fp16 = const()[name = string("op_7657_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7658_cast_fp16 = add(x = variance_255_cast_fp16, y = var_7657_to_fp16)[name = string("op_7658_cast_fp16")]; + fp32 var_7659_epsilon_0 = const()[name = string("op_7659_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7659_cast_fp16 = rsqrt(epsilon = var_7659_epsilon_0, x = var_7658_cast_fp16)[name = string("op_7659_cast_fp16")]; + tensor var_7660_cast_fp16 = mul(x = x_231_cast_fp16, y = var_7659_cast_fp16)[name = string("op_7660_cast_fp16")]; + tensor k_183_cast_fp16 = mul(x = var_7660_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_183_cast_fp16")]; + tensor var_7662 = const()[name = string("op_7662"), val = tensor([1, 16, 128, 1])]; + tensor z_121_cast_fp16 = reshape(shape = var_7662, x = q_183_cast_fp16)[name = string("z_121_cast_fp16")]; + tensor var_7664 = const()[name = string("op_7664"), val = tensor([1, 8, 128, 1])]; + tensor z_123_cast_fp16 = reshape(shape = var_7664, x = k_183_cast_fp16)[name = string("z_123_cast_fp16")]; + tensor z1_121_begin_0 = const()[name = string("z1_121_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_121_end_0 = const()[name = string("z1_121_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_121_end_mask_0 = const()[name = string("z1_121_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_121_cast_fp16 = slice_by_index(begin = z1_121_begin_0, end = z1_121_end_0, end_mask = z1_121_end_mask_0, x = z_121_cast_fp16)[name = string("z1_121_cast_fp16")]; + tensor z2_121_begin_0 = const()[name = string("z2_121_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_121_end_0 = const()[name = string("z2_121_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_121_end_mask_0 = const()[name = string("z2_121_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_121_cast_fp16 = slice_by_index(begin = z2_121_begin_0, end = z2_121_end_0, end_mask = z2_121_end_mask_0, x = z_121_cast_fp16)[name = string("z2_121_cast_fp16")]; + tensor cos_61_to_fp16 = const()[name = string("cos_61_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637504)))]; + tensor var_7672_cast_fp16 = mul(x = z_121_cast_fp16, y = cos_61_to_fp16)[name = string("op_7672_cast_fp16")]; + fp16 const_67_promoted_to_fp16 = const()[name = string("const_67_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7673_cast_fp16 = mul(x = z2_121_cast_fp16, y = const_67_promoted_to_fp16)[name = string("op_7673_cast_fp16")]; + bool var_7675_interleave_0 = const()[name = string("op_7675_interleave_0"), val = bool(false)]; + tensor var_7675_cast_fp16 = concat(axis = var_7581, interleave = var_7675_interleave_0, values = (var_7673_cast_fp16, z1_121_cast_fp16))[name = string("op_7675_cast_fp16")]; + tensor sin_61_to_fp16 = const()[name = string("sin_61_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141637824)))]; + tensor var_7676_cast_fp16 = mul(x = var_7675_cast_fp16, y = sin_61_to_fp16)[name = string("op_7676_cast_fp16")]; + tensor q_185_cast_fp16 = add(x = var_7672_cast_fp16, y = var_7676_cast_fp16)[name = string("q_185_cast_fp16")]; + tensor z1_123_begin_0 = const()[name = string("z1_123_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_123_end_0 = const()[name = string("z1_123_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_123_end_mask_0 = const()[name = string("z1_123_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_123_cast_fp16 = slice_by_index(begin = z1_123_begin_0, end = z1_123_end_0, end_mask = z1_123_end_mask_0, x = z_123_cast_fp16)[name = string("z1_123_cast_fp16")]; + tensor z2_123_begin_0 = const()[name = string("z2_123_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_123_end_0 = const()[name = string("z2_123_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_123_end_mask_0 = const()[name = string("z2_123_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_123_cast_fp16 = slice_by_index(begin = z2_123_begin_0, end = z2_123_end_0, end_mask = z2_123_end_mask_0, x = z_123_cast_fp16)[name = string("z2_123_cast_fp16")]; + tensor var_7684_cast_fp16 = mul(x = z_123_cast_fp16, y = cos_61_to_fp16)[name = string("op_7684_cast_fp16")]; + fp16 const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7685_cast_fp16 = mul(x = z2_123_cast_fp16, y = const_68_promoted_to_fp16)[name = string("op_7685_cast_fp16")]; + bool var_7687_interleave_0 = const()[name = string("op_7687_interleave_0"), val = bool(false)]; + tensor var_7687_cast_fp16 = concat(axis = var_7581, interleave = var_7687_interleave_0, values = (var_7685_cast_fp16, z1_123_cast_fp16))[name = string("op_7687_cast_fp16")]; + tensor var_7688_cast_fp16 = mul(x = var_7687_cast_fp16, y = sin_61_to_fp16)[name = string("op_7688_cast_fp16")]; + tensor k_185_cast_fp16 = add(x = var_7684_cast_fp16, y = var_7688_cast_fp16)[name = string("k_185_cast_fp16")]; + tensor var_7690 = const()[name = string("op_7690"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_61_cast_fp16 = reshape(shape = var_7690, x = k_185_cast_fp16)[name = string("cur_key_61_cast_fp16")]; + tensor var_7692_to_fp16 = const()[name = string("op_7692_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638144)))]; + tensor var_7693_cast_fp16 = mul(x = key_cache_61_cast_fp16, y = var_7692_to_fp16)[name = string("op_7693_cast_fp16")]; + tensor upd_61_to_fp16 = const()[name = string("upd_61_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638272)))]; + tensor var_7694_cast_fp16 = mul(x = cur_key_61_cast_fp16, y = upd_61_to_fp16)[name = string("op_7694_cast_fp16")]; + tensor key_61_cast_fp16 = add(x = var_7693_cast_fp16, y = var_7694_cast_fp16)[name = string("key_61_cast_fp16")]; + tensor var_7696_to_fp16 = const()[name = string("op_7696_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638144)))]; + tensor var_7697_cast_fp16 = mul(x = value_cache_61_cast_fp16, y = var_7696_to_fp16)[name = string("op_7697_cast_fp16")]; + tensor var_7698_cast_fp16 = mul(x = v_61_cast_fp16, y = upd_61_to_fp16)[name = string("op_7698_cast_fp16")]; + tensor value_61_cast_fp16 = add(x = var_7697_cast_fp16, y = var_7698_cast_fp16)[name = string("value_61_cast_fp16")]; + tensor var_7700 = const()[name = string("op_7700"), val = tensor([1, 8, 128, 16])]; + tensor kh_121_cast_fp16 = reshape(shape = var_7700, x = key_61_cast_fp16)[name = string("kh_121_cast_fp16")]; + tensor var_7702 = const()[name = string("op_7702"), val = tensor([1, 8, 128, 16])]; + tensor vh_121_cast_fp16 = reshape(shape = var_7702, x = value_61_cast_fp16)[name = string("vh_121_cast_fp16")]; + tensor transpose_120_perm_0 = const()[name = string("transpose_120_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_60_reps_0 = const()[name = string("tile_60_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_120_cast_fp16 = transpose(perm = transpose_120_perm_0, x = kh_121_cast_fp16)[name = string("transpose_299")]; + tensor tile_60_cast_fp16 = tile(reps = tile_60_reps_0, x = transpose_120_cast_fp16)[name = string("tile_60_cast_fp16")]; + tensor concat_153 = const()[name = string("concat_153"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_120_cast_fp16 = reshape(shape = concat_153, x = tile_60_cast_fp16)[name = string("reshape_120_cast_fp16")]; + tensor transpose_121_perm_0 = const()[name = string("transpose_121_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_154 = const()[name = string("concat_154"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_121_cast_fp16 = transpose(perm = transpose_121_perm_0, x = reshape_120_cast_fp16)[name = string("transpose_298")]; + tensor reshape_121_cast_fp16 = reshape(shape = concat_154, x = transpose_121_cast_fp16)[name = string("reshape_121_cast_fp16")]; + tensor transpose_122_perm_0 = const()[name = string("transpose_122_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_61_reps_0 = const()[name = string("tile_61_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_122_cast_fp16 = transpose(perm = transpose_122_perm_0, x = vh_121_cast_fp16)[name = string("transpose_297")]; + tensor tile_61_cast_fp16 = tile(reps = tile_61_reps_0, x = transpose_122_cast_fp16)[name = string("tile_61_cast_fp16")]; + tensor concat_155 = const()[name = string("concat_155"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_122_cast_fp16 = reshape(shape = concat_155, x = tile_61_cast_fp16)[name = string("reshape_122_cast_fp16")]; + tensor transpose_123_perm_0 = const()[name = string("transpose_123_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_156 = const()[name = string("concat_156"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_123_cast_fp16 = transpose(perm = transpose_123_perm_0, x = reshape_122_cast_fp16)[name = string("transpose_296")]; + tensor reshape_123_cast_fp16 = reshape(shape = concat_156, x = transpose_123_cast_fp16)[name = string("reshape_123_cast_fp16")]; + fp16 var_7706_to_fp16 = const()[name = string("op_7706_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_7707_cast_fp16 = mul(x = q_185_cast_fp16, y = var_7706_to_fp16)[name = string("op_7707_cast_fp16")]; + tensor transpose_437_perm_0 = const()[name = string("transpose_437_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_131_transpose_x_1 = const()[name = string("w_131_transpose_x_1"), val = bool(true)]; + bool w_131_transpose_y_1 = const()[name = string("w_131_transpose_y_1"), val = bool(false)]; + tensor transpose_437_cast_fp16 = transpose(perm = transpose_437_perm_0, x = reshape_121_cast_fp16)[name = string("transpose_295")]; + tensor w_131_cast_fp16 = matmul(transpose_x = w_131_transpose_x_1, transpose_y = w_131_transpose_y_1, x = var_7707_cast_fp16, y = transpose_437_cast_fp16)[name = string("w_131_cast_fp16")]; + tensor pad_61_to_fp16 = const()[name = string("pad_61_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638400)))]; + tensor var_7710_cast_fp16 = add(x = w_131_cast_fp16, y = pad_61_to_fp16)[name = string("op_7710_cast_fp16")]; + tensor w_133_cast_fp16 = softmax(axis = var_7585, x = var_7710_cast_fp16)[name = string("w_133_cast_fp16")]; + tensor transpose_438_perm_0 = const()[name = string("transpose_438_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_61_transpose_x_1 = const()[name = string("attn_61_transpose_x_1"), val = bool(false)]; + bool attn_61_transpose_y_1 = const()[name = string("attn_61_transpose_y_1"), val = bool(true)]; + tensor transpose_438_cast_fp16 = transpose(perm = transpose_438_perm_0, x = reshape_123_cast_fp16)[name = string("transpose_294")]; + tensor attn_61_cast_fp16 = matmul(transpose_x = attn_61_transpose_x_1, transpose_y = attn_61_transpose_y_1, x = transpose_438_cast_fp16, y = w_133_cast_fp16)[name = string("attn_61_cast_fp16")]; + tensor var_7714 = const()[name = string("op_7714"), val = tensor([1, 2048, 1, 1])]; + tensor input_323_cast_fp16 = reshape(shape = var_7714, x = attn_61_cast_fp16)[name = string("input_323_cast_fp16")]; + string attn_output_61_pad_type_0 = const()[name = string("attn_output_61_pad_type_0"), val = string("valid")]; + tensor attn_output_61_strides_0 = const()[name = string("attn_output_61_strides_0"), val = tensor([1, 1])]; + tensor attn_output_61_pad_0 = const()[name = string("attn_output_61_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_61_dilations_0 = const()[name = string("attn_output_61_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_61_groups_0 = const()[name = string("attn_output_61_groups_0"), val = int32(1)]; + tensor attn_output_61_cast_fp16 = conv(dilations = attn_output_61_dilations_0, groups = attn_output_61_groups_0, pad = attn_output_61_pad_0, pad_type = attn_output_61_pad_type_0, strides = attn_output_61_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_323_cast_fp16)[name = string("attn_output_61_cast_fp16")]; + tensor x_233_cast_fp16 = add(x = code_embed_19_cast_fp16, y = attn_output_61_cast_fp16)[name = string("x_233_cast_fp16")]; + tensor var_7728_cast_fp16 = mul(x = x_233_cast_fp16, y = x_233_cast_fp16)[name = string("op_7728_cast_fp16")]; + tensor variance_257_axes_0 = const()[name = string("variance_257_axes_0"), val = tensor([1])]; + bool variance_257_keep_dims_0 = const()[name = string("variance_257_keep_dims_0"), val = bool(true)]; + tensor variance_257_cast_fp16 = reduce_mean(axes = variance_257_axes_0, keep_dims = variance_257_keep_dims_0, x = var_7728_cast_fp16)[name = string("variance_257_cast_fp16")]; + fp16 var_7731_to_fp16 = const()[name = string("op_7731_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7732_cast_fp16 = add(x = variance_257_cast_fp16, y = var_7731_to_fp16)[name = string("op_7732_cast_fp16")]; + fp32 var_7733_epsilon_0 = const()[name = string("op_7733_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7733_cast_fp16 = rsqrt(epsilon = var_7733_epsilon_0, x = var_7732_cast_fp16)[name = string("op_7733_cast_fp16")]; + tensor var_7734_cast_fp16 = mul(x = x_233_cast_fp16, y = var_7733_cast_fp16)[name = string("op_7734_cast_fp16")]; + tensor input_325_cast_fp16 = mul(x = var_7734_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_325_cast_fp16")]; + string input_327_pad_type_0 = const()[name = string("input_327_pad_type_0"), val = string("valid")]; + tensor input_327_strides_0 = const()[name = string("input_327_strides_0"), val = tensor([1, 1])]; + tensor input_327_pad_0 = const()[name = string("input_327_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_327_dilations_0 = const()[name = string("input_327_dilations_0"), val = tensor([1, 1])]; + int32 input_327_groups_0 = const()[name = string("input_327_groups_0"), val = int32(1)]; + tensor input_327_cast_fp16 = conv(dilations = input_327_dilations_0, groups = input_327_groups_0, pad = input_327_pad_0, pad_type = input_327_pad_type_0, strides = input_327_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_325_cast_fp16)[name = string("input_327_cast_fp16")]; + tensor var_7742_cast_fp16 = silu(x = input_327_cast_fp16)[name = string("op_7742_cast_fp16")]; + string var_7748_pad_type_0 = const()[name = string("op_7748_pad_type_0"), val = string("valid")]; + tensor var_7748_strides_0 = const()[name = string("op_7748_strides_0"), val = tensor([1, 1])]; + tensor var_7748_pad_0 = const()[name = string("op_7748_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7748_dilations_0 = const()[name = string("op_7748_dilations_0"), val = tensor([1, 1])]; + int32 var_7748_groups_0 = const()[name = string("op_7748_groups_0"), val = int32(1)]; + tensor var_7748_cast_fp16 = conv(dilations = var_7748_dilations_0, groups = var_7748_groups_0, pad = var_7748_pad_0, pad_type = var_7748_pad_type_0, strides = var_7748_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_325_cast_fp16)[name = string("op_7748_cast_fp16")]; + tensor input_329_cast_fp16 = mul(x = var_7742_cast_fp16, y = var_7748_cast_fp16)[name = string("input_329_cast_fp16")]; + string h_61_pad_type_0 = const()[name = string("h_61_pad_type_0"), val = string("valid")]; + tensor h_61_strides_0 = const()[name = string("h_61_strides_0"), val = tensor([1, 1])]; + tensor h_61_pad_0 = const()[name = string("h_61_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_61_dilations_0 = const()[name = string("h_61_dilations_0"), val = tensor([1, 1])]; + int32 h_61_groups_0 = const()[name = string("h_61_groups_0"), val = int32(1)]; + tensor h_61_cast_fp16 = conv(dilations = h_61_dilations_0, groups = h_61_groups_0, pad = h_61_pad_0, pad_type = h_61_pad_type_0, strides = h_61_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_329_cast_fp16)[name = string("h_61_cast_fp16")]; + tensor x_235_cast_fp16 = add(x = x_233_cast_fp16, y = h_61_cast_fp16)[name = string("x_235_cast_fp16")]; + tensor key_cache_63_begin_0 = const()[name = string("key_cache_63_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_63_end_0 = const()[name = string("key_cache_63_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_63_end_mask_0 = const()[name = string("key_cache_63_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_63_cast_fp16 = slice_by_index(begin = key_cache_63_begin_0, end = key_cache_63_end_0, end_mask = key_cache_63_end_mask_0, x = layer_key_caches_13_cast_fp16)[name = string("key_cache_63_cast_fp16")]; + tensor value_cache_63_begin_0 = const()[name = string("value_cache_63_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_63_end_0 = const()[name = string("value_cache_63_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_63_end_mask_0 = const()[name = string("value_cache_63_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_63_cast_fp16 = slice_by_index(begin = value_cache_63_begin_0, end = value_cache_63_end_0, end_mask = value_cache_63_end_mask_0, x = layer_value_caches_13_cast_fp16)[name = string("value_cache_63_cast_fp16")]; + int32 var_7801 = const()[name = string("op_7801"), val = int32(2)]; + int32 var_7805 = const()[name = string("op_7805"), val = int32(3)]; + tensor var_7820_cast_fp16 = mul(x = x_235_cast_fp16, y = x_235_cast_fp16)[name = string("op_7820_cast_fp16")]; + tensor variance_259_axes_0 = const()[name = string("variance_259_axes_0"), val = tensor([1])]; + bool variance_259_keep_dims_0 = const()[name = string("variance_259_keep_dims_0"), val = bool(true)]; + tensor variance_259_cast_fp16 = reduce_mean(axes = variance_259_axes_0, keep_dims = variance_259_keep_dims_0, x = var_7820_cast_fp16)[name = string("variance_259_cast_fp16")]; + fp16 var_7823_to_fp16 = const()[name = string("op_7823_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7824_cast_fp16 = add(x = variance_259_cast_fp16, y = var_7823_to_fp16)[name = string("op_7824_cast_fp16")]; + fp32 var_7825_epsilon_0 = const()[name = string("op_7825_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7825_cast_fp16 = rsqrt(epsilon = var_7825_epsilon_0, x = var_7824_cast_fp16)[name = string("op_7825_cast_fp16")]; + tensor var_7826_cast_fp16 = mul(x = x_235_cast_fp16, y = var_7825_cast_fp16)[name = string("op_7826_cast_fp16")]; + tensor input_331_cast_fp16 = mul(x = var_7826_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_331_cast_fp16")]; + string q_187_pad_type_0 = const()[name = string("q_187_pad_type_0"), val = string("valid")]; + tensor q_187_strides_0 = const()[name = string("q_187_strides_0"), val = tensor([1, 1])]; + tensor q_187_pad_0 = const()[name = string("q_187_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_187_dilations_0 = const()[name = string("q_187_dilations_0"), val = tensor([1, 1])]; + int32 q_187_groups_0 = const()[name = string("q_187_groups_0"), val = int32(1)]; + tensor q_187_cast_fp16 = conv(dilations = q_187_dilations_0, groups = q_187_groups_0, pad = q_187_pad_0, pad_type = q_187_pad_type_0, strides = q_187_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_331_cast_fp16)[name = string("q_187_cast_fp16")]; + string k_187_pad_type_0 = const()[name = string("k_187_pad_type_0"), val = string("valid")]; + tensor k_187_strides_0 = const()[name = string("k_187_strides_0"), val = tensor([1, 1])]; + tensor k_187_pad_0 = const()[name = string("k_187_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_187_dilations_0 = const()[name = string("k_187_dilations_0"), val = tensor([1, 1])]; + int32 k_187_groups_0 = const()[name = string("k_187_groups_0"), val = int32(1)]; + tensor k_187_cast_fp16 = conv(dilations = k_187_dilations_0, groups = k_187_groups_0, pad = k_187_pad_0, pad_type = k_187_pad_type_0, strides = k_187_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_331_cast_fp16)[name = string("k_187_cast_fp16")]; + string v_63_pad_type_0 = const()[name = string("v_63_pad_type_0"), val = string("valid")]; + tensor v_63_strides_0 = const()[name = string("v_63_strides_0"), val = tensor([1, 1])]; + tensor v_63_pad_0 = const()[name = string("v_63_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_63_dilations_0 = const()[name = string("v_63_dilations_0"), val = tensor([1, 1])]; + int32 v_63_groups_0 = const()[name = string("v_63_groups_0"), val = int32(1)]; + tensor v_63_cast_fp16 = conv(dilations = v_63_dilations_0, groups = v_63_groups_0, pad = v_63_pad_0, pad_type = v_63_pad_type_0, strides = v_63_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_331_cast_fp16)[name = string("v_63_cast_fp16")]; + tensor var_7860 = const()[name = string("op_7860"), val = tensor([16, 128, 1, 1])]; + tensor x_237_cast_fp16 = reshape(shape = var_7860, x = q_187_cast_fp16)[name = string("x_237_cast_fp16")]; + tensor var_7863_cast_fp16 = mul(x = x_237_cast_fp16, y = x_237_cast_fp16)[name = string("op_7863_cast_fp16")]; + tensor variance_261_axes_0 = const()[name = string("variance_261_axes_0"), val = tensor([1])]; + bool variance_261_keep_dims_0 = const()[name = string("variance_261_keep_dims_0"), val = bool(true)]; + tensor variance_261_cast_fp16 = reduce_mean(axes = variance_261_axes_0, keep_dims = variance_261_keep_dims_0, x = var_7863_cast_fp16)[name = string("variance_261_cast_fp16")]; + fp16 var_7866_to_fp16 = const()[name = string("op_7866_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7867_cast_fp16 = add(x = variance_261_cast_fp16, y = var_7866_to_fp16)[name = string("op_7867_cast_fp16")]; + fp32 var_7868_epsilon_0 = const()[name = string("op_7868_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7868_cast_fp16 = rsqrt(epsilon = var_7868_epsilon_0, x = var_7867_cast_fp16)[name = string("op_7868_cast_fp16")]; + tensor var_7869_cast_fp16 = mul(x = x_237_cast_fp16, y = var_7868_cast_fp16)[name = string("op_7869_cast_fp16")]; + tensor q_189_cast_fp16 = mul(x = var_7869_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_189_cast_fp16")]; + tensor var_7871 = const()[name = string("op_7871"), val = tensor([8, 128, 1, 1])]; + tensor x_239_cast_fp16 = reshape(shape = var_7871, x = k_187_cast_fp16)[name = string("x_239_cast_fp16")]; + tensor var_7874_cast_fp16 = mul(x = x_239_cast_fp16, y = x_239_cast_fp16)[name = string("op_7874_cast_fp16")]; + tensor variance_263_axes_0 = const()[name = string("variance_263_axes_0"), val = tensor([1])]; + bool variance_263_keep_dims_0 = const()[name = string("variance_263_keep_dims_0"), val = bool(true)]; + tensor variance_263_cast_fp16 = reduce_mean(axes = variance_263_axes_0, keep_dims = variance_263_keep_dims_0, x = var_7874_cast_fp16)[name = string("variance_263_cast_fp16")]; + fp16 var_7877_to_fp16 = const()[name = string("op_7877_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7878_cast_fp16 = add(x = variance_263_cast_fp16, y = var_7877_to_fp16)[name = string("op_7878_cast_fp16")]; + fp32 var_7879_epsilon_0 = const()[name = string("op_7879_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7879_cast_fp16 = rsqrt(epsilon = var_7879_epsilon_0, x = var_7878_cast_fp16)[name = string("op_7879_cast_fp16")]; + tensor var_7880_cast_fp16 = mul(x = x_239_cast_fp16, y = var_7879_cast_fp16)[name = string("op_7880_cast_fp16")]; + tensor k_189_cast_fp16 = mul(x = var_7880_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_189_cast_fp16")]; + tensor var_7882 = const()[name = string("op_7882"), val = tensor([1, 16, 128, 1])]; + tensor z_125_cast_fp16 = reshape(shape = var_7882, x = q_189_cast_fp16)[name = string("z_125_cast_fp16")]; + tensor var_7884 = const()[name = string("op_7884"), val = tensor([1, 8, 128, 1])]; + tensor z_127_cast_fp16 = reshape(shape = var_7884, x = k_189_cast_fp16)[name = string("z_127_cast_fp16")]; + tensor z1_125_begin_0 = const()[name = string("z1_125_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_125_end_0 = const()[name = string("z1_125_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_125_end_mask_0 = const()[name = string("z1_125_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_125_cast_fp16 = slice_by_index(begin = z1_125_begin_0, end = z1_125_end_0, end_mask = z1_125_end_mask_0, x = z_125_cast_fp16)[name = string("z1_125_cast_fp16")]; + tensor z2_125_begin_0 = const()[name = string("z2_125_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_125_end_0 = const()[name = string("z2_125_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_125_end_mask_0 = const()[name = string("z2_125_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_125_cast_fp16 = slice_by_index(begin = z2_125_begin_0, end = z2_125_end_0, end_mask = z2_125_end_mask_0, x = z_125_cast_fp16)[name = string("z2_125_cast_fp16")]; + tensor var_7892_cast_fp16 = mul(x = z_125_cast_fp16, y = cos_61_to_fp16)[name = string("op_7892_cast_fp16")]; + fp16 const_69_promoted_to_fp16 = const()[name = string("const_69_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7893_cast_fp16 = mul(x = z2_125_cast_fp16, y = const_69_promoted_to_fp16)[name = string("op_7893_cast_fp16")]; + bool var_7895_interleave_0 = const()[name = string("op_7895_interleave_0"), val = bool(false)]; + tensor var_7895_cast_fp16 = concat(axis = var_7801, interleave = var_7895_interleave_0, values = (var_7893_cast_fp16, z1_125_cast_fp16))[name = string("op_7895_cast_fp16")]; + tensor var_7896_cast_fp16 = mul(x = var_7895_cast_fp16, y = sin_61_to_fp16)[name = string("op_7896_cast_fp16")]; + tensor q_191_cast_fp16 = add(x = var_7892_cast_fp16, y = var_7896_cast_fp16)[name = string("q_191_cast_fp16")]; + tensor z1_127_begin_0 = const()[name = string("z1_127_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_127_end_0 = const()[name = string("z1_127_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_127_end_mask_0 = const()[name = string("z1_127_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_127_cast_fp16 = slice_by_index(begin = z1_127_begin_0, end = z1_127_end_0, end_mask = z1_127_end_mask_0, x = z_127_cast_fp16)[name = string("z1_127_cast_fp16")]; + tensor z2_127_begin_0 = const()[name = string("z2_127_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_127_end_0 = const()[name = string("z2_127_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_127_end_mask_0 = const()[name = string("z2_127_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_127_cast_fp16 = slice_by_index(begin = z2_127_begin_0, end = z2_127_end_0, end_mask = z2_127_end_mask_0, x = z_127_cast_fp16)[name = string("z2_127_cast_fp16")]; + tensor var_7904_cast_fp16 = mul(x = z_127_cast_fp16, y = cos_61_to_fp16)[name = string("op_7904_cast_fp16")]; + fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7905_cast_fp16 = mul(x = z2_127_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_7905_cast_fp16")]; + bool var_7907_interleave_0 = const()[name = string("op_7907_interleave_0"), val = bool(false)]; + tensor var_7907_cast_fp16 = concat(axis = var_7801, interleave = var_7907_interleave_0, values = (var_7905_cast_fp16, z1_127_cast_fp16))[name = string("op_7907_cast_fp16")]; + tensor var_7908_cast_fp16 = mul(x = var_7907_cast_fp16, y = sin_61_to_fp16)[name = string("op_7908_cast_fp16")]; + tensor k_191_cast_fp16 = add(x = var_7904_cast_fp16, y = var_7908_cast_fp16)[name = string("k_191_cast_fp16")]; + tensor var_7910 = const()[name = string("op_7910"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_63_cast_fp16 = reshape(shape = var_7910, x = k_191_cast_fp16)[name = string("cur_key_63_cast_fp16")]; + tensor var_7912_to_fp16 = const()[name = string("op_7912_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638144)))]; + tensor var_7913_cast_fp16 = mul(x = key_cache_63_cast_fp16, y = var_7912_to_fp16)[name = string("op_7913_cast_fp16")]; + tensor upd_63_to_fp16 = const()[name = string("upd_63_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638272)))]; + tensor var_7914_cast_fp16 = mul(x = cur_key_63_cast_fp16, y = upd_63_to_fp16)[name = string("op_7914_cast_fp16")]; + tensor key_63_cast_fp16 = add(x = var_7913_cast_fp16, y = var_7914_cast_fp16)[name = string("key_63_cast_fp16")]; + tensor var_7916_to_fp16 = const()[name = string("op_7916_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638144)))]; + tensor var_7917_cast_fp16 = mul(x = value_cache_63_cast_fp16, y = var_7916_to_fp16)[name = string("op_7917_cast_fp16")]; + tensor var_7918_cast_fp16 = mul(x = v_63_cast_fp16, y = upd_63_to_fp16)[name = string("op_7918_cast_fp16")]; + tensor value_63_cast_fp16 = add(x = var_7917_cast_fp16, y = var_7918_cast_fp16)[name = string("value_63_cast_fp16")]; + tensor var_7920 = const()[name = string("op_7920"), val = tensor([1, 8, 128, 16])]; + tensor kh_125_cast_fp16 = reshape(shape = var_7920, x = key_63_cast_fp16)[name = string("kh_125_cast_fp16")]; + tensor var_7922 = const()[name = string("op_7922"), val = tensor([1, 8, 128, 16])]; + tensor vh_125_cast_fp16 = reshape(shape = var_7922, x = value_63_cast_fp16)[name = string("vh_125_cast_fp16")]; + tensor transpose_124_perm_0 = const()[name = string("transpose_124_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_62_reps_0 = const()[name = string("tile_62_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_124_cast_fp16 = transpose(perm = transpose_124_perm_0, x = kh_125_cast_fp16)[name = string("transpose_293")]; + tensor tile_62_cast_fp16 = tile(reps = tile_62_reps_0, x = transpose_124_cast_fp16)[name = string("tile_62_cast_fp16")]; + tensor concat_157 = const()[name = string("concat_157"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_124_cast_fp16 = reshape(shape = concat_157, x = tile_62_cast_fp16)[name = string("reshape_124_cast_fp16")]; + tensor transpose_125_perm_0 = const()[name = string("transpose_125_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_158 = const()[name = string("concat_158"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_125_cast_fp16 = transpose(perm = transpose_125_perm_0, x = reshape_124_cast_fp16)[name = string("transpose_292")]; + tensor reshape_125_cast_fp16 = reshape(shape = concat_158, x = transpose_125_cast_fp16)[name = string("reshape_125_cast_fp16")]; + tensor transpose_126_perm_0 = const()[name = string("transpose_126_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_63_reps_0 = const()[name = string("tile_63_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_126_cast_fp16 = transpose(perm = transpose_126_perm_0, x = vh_125_cast_fp16)[name = string("transpose_291")]; + tensor tile_63_cast_fp16 = tile(reps = tile_63_reps_0, x = transpose_126_cast_fp16)[name = string("tile_63_cast_fp16")]; + tensor concat_159 = const()[name = string("concat_159"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_126_cast_fp16 = reshape(shape = concat_159, x = tile_63_cast_fp16)[name = string("reshape_126_cast_fp16")]; + tensor transpose_127_perm_0 = const()[name = string("transpose_127_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_160 = const()[name = string("concat_160"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_127_cast_fp16 = transpose(perm = transpose_127_perm_0, x = reshape_126_cast_fp16)[name = string("transpose_290")]; + tensor reshape_127_cast_fp16 = reshape(shape = concat_160, x = transpose_127_cast_fp16)[name = string("reshape_127_cast_fp16")]; + fp16 var_7926_to_fp16 = const()[name = string("op_7926_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_7927_cast_fp16 = mul(x = q_191_cast_fp16, y = var_7926_to_fp16)[name = string("op_7927_cast_fp16")]; + tensor transpose_441_perm_0 = const()[name = string("transpose_441_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_135_transpose_x_1 = const()[name = string("w_135_transpose_x_1"), val = bool(true)]; + bool w_135_transpose_y_1 = const()[name = string("w_135_transpose_y_1"), val = bool(false)]; + tensor transpose_441_cast_fp16 = transpose(perm = transpose_441_perm_0, x = reshape_125_cast_fp16)[name = string("transpose_289")]; + tensor w_135_cast_fp16 = matmul(transpose_x = w_135_transpose_x_1, transpose_y = w_135_transpose_y_1, x = var_7927_cast_fp16, y = transpose_441_cast_fp16)[name = string("w_135_cast_fp16")]; + tensor pad_63_to_fp16 = const()[name = string("pad_63_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638400)))]; + tensor var_7930_cast_fp16 = add(x = w_135_cast_fp16, y = pad_63_to_fp16)[name = string("op_7930_cast_fp16")]; + tensor w_137_cast_fp16 = softmax(axis = var_7805, x = var_7930_cast_fp16)[name = string("w_137_cast_fp16")]; + tensor transpose_442_perm_0 = const()[name = string("transpose_442_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_63_transpose_x_1 = const()[name = string("attn_63_transpose_x_1"), val = bool(false)]; + bool attn_63_transpose_y_1 = const()[name = string("attn_63_transpose_y_1"), val = bool(true)]; + tensor transpose_442_cast_fp16 = transpose(perm = transpose_442_perm_0, x = reshape_127_cast_fp16)[name = string("transpose_288")]; + tensor attn_63_cast_fp16 = matmul(transpose_x = attn_63_transpose_x_1, transpose_y = attn_63_transpose_y_1, x = transpose_442_cast_fp16, y = w_137_cast_fp16)[name = string("attn_63_cast_fp16")]; + tensor var_7934 = const()[name = string("op_7934"), val = tensor([1, 2048, 1, 1])]; + tensor input_333_cast_fp16 = reshape(shape = var_7934, x = attn_63_cast_fp16)[name = string("input_333_cast_fp16")]; + string attn_output_63_pad_type_0 = const()[name = string("attn_output_63_pad_type_0"), val = string("valid")]; + tensor attn_output_63_strides_0 = const()[name = string("attn_output_63_strides_0"), val = tensor([1, 1])]; + tensor attn_output_63_pad_0 = const()[name = string("attn_output_63_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_63_dilations_0 = const()[name = string("attn_output_63_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_63_groups_0 = const()[name = string("attn_output_63_groups_0"), val = int32(1)]; + tensor attn_output_63_cast_fp16 = conv(dilations = attn_output_63_dilations_0, groups = attn_output_63_groups_0, pad = attn_output_63_pad_0, pad_type = attn_output_63_pad_type_0, strides = attn_output_63_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_333_cast_fp16)[name = string("attn_output_63_cast_fp16")]; + tensor x_241_cast_fp16 = add(x = x_235_cast_fp16, y = attn_output_63_cast_fp16)[name = string("x_241_cast_fp16")]; + tensor var_7948_cast_fp16 = mul(x = x_241_cast_fp16, y = x_241_cast_fp16)[name = string("op_7948_cast_fp16")]; + tensor variance_265_axes_0 = const()[name = string("variance_265_axes_0"), val = tensor([1])]; + bool variance_265_keep_dims_0 = const()[name = string("variance_265_keep_dims_0"), val = bool(true)]; + tensor variance_265_cast_fp16 = reduce_mean(axes = variance_265_axes_0, keep_dims = variance_265_keep_dims_0, x = var_7948_cast_fp16)[name = string("variance_265_cast_fp16")]; + fp16 var_7951_to_fp16 = const()[name = string("op_7951_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7952_cast_fp16 = add(x = variance_265_cast_fp16, y = var_7951_to_fp16)[name = string("op_7952_cast_fp16")]; + fp32 var_7953_epsilon_0 = const()[name = string("op_7953_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7953_cast_fp16 = rsqrt(epsilon = var_7953_epsilon_0, x = var_7952_cast_fp16)[name = string("op_7953_cast_fp16")]; + tensor var_7954_cast_fp16 = mul(x = x_241_cast_fp16, y = var_7953_cast_fp16)[name = string("op_7954_cast_fp16")]; + tensor input_335_cast_fp16 = mul(x = var_7954_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_335_cast_fp16")]; + string input_337_pad_type_0 = const()[name = string("input_337_pad_type_0"), val = string("valid")]; + tensor input_337_strides_0 = const()[name = string("input_337_strides_0"), val = tensor([1, 1])]; + tensor input_337_pad_0 = const()[name = string("input_337_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_337_dilations_0 = const()[name = string("input_337_dilations_0"), val = tensor([1, 1])]; + int32 input_337_groups_0 = const()[name = string("input_337_groups_0"), val = int32(1)]; + tensor input_337_cast_fp16 = conv(dilations = input_337_dilations_0, groups = input_337_groups_0, pad = input_337_pad_0, pad_type = input_337_pad_type_0, strides = input_337_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_335_cast_fp16)[name = string("input_337_cast_fp16")]; + tensor var_7962_cast_fp16 = silu(x = input_337_cast_fp16)[name = string("op_7962_cast_fp16")]; + string var_7968_pad_type_0 = const()[name = string("op_7968_pad_type_0"), val = string("valid")]; + tensor var_7968_strides_0 = const()[name = string("op_7968_strides_0"), val = tensor([1, 1])]; + tensor var_7968_pad_0 = const()[name = string("op_7968_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7968_dilations_0 = const()[name = string("op_7968_dilations_0"), val = tensor([1, 1])]; + int32 var_7968_groups_0 = const()[name = string("op_7968_groups_0"), val = int32(1)]; + tensor var_7968_cast_fp16 = conv(dilations = var_7968_dilations_0, groups = var_7968_groups_0, pad = var_7968_pad_0, pad_type = var_7968_pad_type_0, strides = var_7968_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_335_cast_fp16)[name = string("op_7968_cast_fp16")]; + tensor input_339_cast_fp16 = mul(x = var_7962_cast_fp16, y = var_7968_cast_fp16)[name = string("input_339_cast_fp16")]; + string h_63_pad_type_0 = const()[name = string("h_63_pad_type_0"), val = string("valid")]; + tensor h_63_strides_0 = const()[name = string("h_63_strides_0"), val = tensor([1, 1])]; + tensor h_63_pad_0 = const()[name = string("h_63_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_63_dilations_0 = const()[name = string("h_63_dilations_0"), val = tensor([1, 1])]; + int32 h_63_groups_0 = const()[name = string("h_63_groups_0"), val = int32(1)]; + tensor h_63_cast_fp16 = conv(dilations = h_63_dilations_0, groups = h_63_groups_0, pad = h_63_pad_0, pad_type = h_63_pad_type_0, strides = h_63_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_339_cast_fp16)[name = string("h_63_cast_fp16")]; + tensor x_243_cast_fp16 = add(x = x_241_cast_fp16, y = h_63_cast_fp16)[name = string("x_243_cast_fp16")]; + tensor key_cache_65_begin_0 = const()[name = string("key_cache_65_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_65_end_0 = const()[name = string("key_cache_65_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_65_end_mask_0 = const()[name = string("key_cache_65_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_65_cast_fp16 = slice_by_index(begin = key_cache_65_begin_0, end = key_cache_65_end_0, end_mask = key_cache_65_end_mask_0, x = layer_key_caches_13_cast_fp16)[name = string("key_cache_65_cast_fp16")]; + tensor value_cache_65_begin_0 = const()[name = string("value_cache_65_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_65_end_0 = const()[name = string("value_cache_65_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_65_end_mask_0 = const()[name = string("value_cache_65_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_65_cast_fp16 = slice_by_index(begin = value_cache_65_begin_0, end = value_cache_65_end_0, end_mask = value_cache_65_end_mask_0, x = layer_value_caches_13_cast_fp16)[name = string("value_cache_65_cast_fp16")]; + int32 var_8021 = const()[name = string("op_8021"), val = int32(2)]; + int32 var_8025 = const()[name = string("op_8025"), val = int32(3)]; + tensor var_8040_cast_fp16 = mul(x = x_243_cast_fp16, y = x_243_cast_fp16)[name = string("op_8040_cast_fp16")]; + tensor variance_267_axes_0 = const()[name = string("variance_267_axes_0"), val = tensor([1])]; + bool variance_267_keep_dims_0 = const()[name = string("variance_267_keep_dims_0"), val = bool(true)]; + tensor variance_267_cast_fp16 = reduce_mean(axes = variance_267_axes_0, keep_dims = variance_267_keep_dims_0, x = var_8040_cast_fp16)[name = string("variance_267_cast_fp16")]; + fp16 var_8043_to_fp16 = const()[name = string("op_8043_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8044_cast_fp16 = add(x = variance_267_cast_fp16, y = var_8043_to_fp16)[name = string("op_8044_cast_fp16")]; + fp32 var_8045_epsilon_0 = const()[name = string("op_8045_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8045_cast_fp16 = rsqrt(epsilon = var_8045_epsilon_0, x = var_8044_cast_fp16)[name = string("op_8045_cast_fp16")]; + tensor var_8046_cast_fp16 = mul(x = x_243_cast_fp16, y = var_8045_cast_fp16)[name = string("op_8046_cast_fp16")]; + tensor input_341_cast_fp16 = mul(x = var_8046_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_341_cast_fp16")]; + string q_193_pad_type_0 = const()[name = string("q_193_pad_type_0"), val = string("valid")]; + tensor q_193_strides_0 = const()[name = string("q_193_strides_0"), val = tensor([1, 1])]; + tensor q_193_pad_0 = const()[name = string("q_193_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_193_dilations_0 = const()[name = string("q_193_dilations_0"), val = tensor([1, 1])]; + int32 q_193_groups_0 = const()[name = string("q_193_groups_0"), val = int32(1)]; + tensor q_193_cast_fp16 = conv(dilations = q_193_dilations_0, groups = q_193_groups_0, pad = q_193_pad_0, pad_type = q_193_pad_type_0, strides = q_193_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_341_cast_fp16)[name = string("q_193_cast_fp16")]; + string k_193_pad_type_0 = const()[name = string("k_193_pad_type_0"), val = string("valid")]; + tensor k_193_strides_0 = const()[name = string("k_193_strides_0"), val = tensor([1, 1])]; + tensor k_193_pad_0 = const()[name = string("k_193_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_193_dilations_0 = const()[name = string("k_193_dilations_0"), val = tensor([1, 1])]; + int32 k_193_groups_0 = const()[name = string("k_193_groups_0"), val = int32(1)]; + tensor k_193_cast_fp16 = conv(dilations = k_193_dilations_0, groups = k_193_groups_0, pad = k_193_pad_0, pad_type = k_193_pad_type_0, strides = k_193_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_341_cast_fp16)[name = string("k_193_cast_fp16")]; + string v_65_pad_type_0 = const()[name = string("v_65_pad_type_0"), val = string("valid")]; + tensor v_65_strides_0 = const()[name = string("v_65_strides_0"), val = tensor([1, 1])]; + tensor v_65_pad_0 = const()[name = string("v_65_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_65_dilations_0 = const()[name = string("v_65_dilations_0"), val = tensor([1, 1])]; + int32 v_65_groups_0 = const()[name = string("v_65_groups_0"), val = int32(1)]; + tensor v_65_cast_fp16 = conv(dilations = v_65_dilations_0, groups = v_65_groups_0, pad = v_65_pad_0, pad_type = v_65_pad_type_0, strides = v_65_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_341_cast_fp16)[name = string("v_65_cast_fp16")]; + tensor var_8080 = const()[name = string("op_8080"), val = tensor([16, 128, 1, 1])]; + tensor x_245_cast_fp16 = reshape(shape = var_8080, x = q_193_cast_fp16)[name = string("x_245_cast_fp16")]; + tensor var_8083_cast_fp16 = mul(x = x_245_cast_fp16, y = x_245_cast_fp16)[name = string("op_8083_cast_fp16")]; + tensor variance_269_axes_0 = const()[name = string("variance_269_axes_0"), val = tensor([1])]; + bool variance_269_keep_dims_0 = const()[name = string("variance_269_keep_dims_0"), val = bool(true)]; + tensor variance_269_cast_fp16 = reduce_mean(axes = variance_269_axes_0, keep_dims = variance_269_keep_dims_0, x = var_8083_cast_fp16)[name = string("variance_269_cast_fp16")]; + fp16 var_8086_to_fp16 = const()[name = string("op_8086_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8087_cast_fp16 = add(x = variance_269_cast_fp16, y = var_8086_to_fp16)[name = string("op_8087_cast_fp16")]; + fp32 var_8088_epsilon_0 = const()[name = string("op_8088_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8088_cast_fp16 = rsqrt(epsilon = var_8088_epsilon_0, x = var_8087_cast_fp16)[name = string("op_8088_cast_fp16")]; + tensor var_8089_cast_fp16 = mul(x = x_245_cast_fp16, y = var_8088_cast_fp16)[name = string("op_8089_cast_fp16")]; + tensor q_195_cast_fp16 = mul(x = var_8089_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_195_cast_fp16")]; + tensor var_8091 = const()[name = string("op_8091"), val = tensor([8, 128, 1, 1])]; + tensor x_247_cast_fp16 = reshape(shape = var_8091, x = k_193_cast_fp16)[name = string("x_247_cast_fp16")]; + tensor var_8094_cast_fp16 = mul(x = x_247_cast_fp16, y = x_247_cast_fp16)[name = string("op_8094_cast_fp16")]; + tensor variance_271_axes_0 = const()[name = string("variance_271_axes_0"), val = tensor([1])]; + bool variance_271_keep_dims_0 = const()[name = string("variance_271_keep_dims_0"), val = bool(true)]; + tensor variance_271_cast_fp16 = reduce_mean(axes = variance_271_axes_0, keep_dims = variance_271_keep_dims_0, x = var_8094_cast_fp16)[name = string("variance_271_cast_fp16")]; + fp16 var_8097_to_fp16 = const()[name = string("op_8097_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8098_cast_fp16 = add(x = variance_271_cast_fp16, y = var_8097_to_fp16)[name = string("op_8098_cast_fp16")]; + fp32 var_8099_epsilon_0 = const()[name = string("op_8099_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8099_cast_fp16 = rsqrt(epsilon = var_8099_epsilon_0, x = var_8098_cast_fp16)[name = string("op_8099_cast_fp16")]; + tensor var_8100_cast_fp16 = mul(x = x_247_cast_fp16, y = var_8099_cast_fp16)[name = string("op_8100_cast_fp16")]; + tensor k_195_cast_fp16 = mul(x = var_8100_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_195_cast_fp16")]; + tensor var_8102 = const()[name = string("op_8102"), val = tensor([1, 16, 128, 1])]; + tensor z_129_cast_fp16 = reshape(shape = var_8102, x = q_195_cast_fp16)[name = string("z_129_cast_fp16")]; + tensor var_8104 = const()[name = string("op_8104"), val = tensor([1, 8, 128, 1])]; + tensor z_131_cast_fp16 = reshape(shape = var_8104, x = k_195_cast_fp16)[name = string("z_131_cast_fp16")]; + tensor z1_129_begin_0 = const()[name = string("z1_129_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_129_end_0 = const()[name = string("z1_129_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_129_end_mask_0 = const()[name = string("z1_129_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_129_cast_fp16 = slice_by_index(begin = z1_129_begin_0, end = z1_129_end_0, end_mask = z1_129_end_mask_0, x = z_129_cast_fp16)[name = string("z1_129_cast_fp16")]; + tensor z2_129_begin_0 = const()[name = string("z2_129_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_129_end_0 = const()[name = string("z2_129_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_129_end_mask_0 = const()[name = string("z2_129_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_129_cast_fp16 = slice_by_index(begin = z2_129_begin_0, end = z2_129_end_0, end_mask = z2_129_end_mask_0, x = z_129_cast_fp16)[name = string("z2_129_cast_fp16")]; + tensor var_8112_cast_fp16 = mul(x = z_129_cast_fp16, y = cos_61_to_fp16)[name = string("op_8112_cast_fp16")]; + fp16 const_71_promoted_to_fp16 = const()[name = string("const_71_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8113_cast_fp16 = mul(x = z2_129_cast_fp16, y = const_71_promoted_to_fp16)[name = string("op_8113_cast_fp16")]; + bool var_8115_interleave_0 = const()[name = string("op_8115_interleave_0"), val = bool(false)]; + tensor var_8115_cast_fp16 = concat(axis = var_8021, interleave = var_8115_interleave_0, values = (var_8113_cast_fp16, z1_129_cast_fp16))[name = string("op_8115_cast_fp16")]; + tensor var_8116_cast_fp16 = mul(x = var_8115_cast_fp16, y = sin_61_to_fp16)[name = string("op_8116_cast_fp16")]; + tensor q_197_cast_fp16 = add(x = var_8112_cast_fp16, y = var_8116_cast_fp16)[name = string("q_197_cast_fp16")]; + tensor z1_131_begin_0 = const()[name = string("z1_131_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_131_end_0 = const()[name = string("z1_131_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_131_end_mask_0 = const()[name = string("z1_131_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_131_cast_fp16 = slice_by_index(begin = z1_131_begin_0, end = z1_131_end_0, end_mask = z1_131_end_mask_0, x = z_131_cast_fp16)[name = string("z1_131_cast_fp16")]; + tensor z2_131_begin_0 = const()[name = string("z2_131_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_131_end_0 = const()[name = string("z2_131_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_131_end_mask_0 = const()[name = string("z2_131_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_131_cast_fp16 = slice_by_index(begin = z2_131_begin_0, end = z2_131_end_0, end_mask = z2_131_end_mask_0, x = z_131_cast_fp16)[name = string("z2_131_cast_fp16")]; + tensor var_8124_cast_fp16 = mul(x = z_131_cast_fp16, y = cos_61_to_fp16)[name = string("op_8124_cast_fp16")]; + fp16 const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8125_cast_fp16 = mul(x = z2_131_cast_fp16, y = const_72_promoted_to_fp16)[name = string("op_8125_cast_fp16")]; + bool var_8127_interleave_0 = const()[name = string("op_8127_interleave_0"), val = bool(false)]; + tensor var_8127_cast_fp16 = concat(axis = var_8021, interleave = var_8127_interleave_0, values = (var_8125_cast_fp16, z1_131_cast_fp16))[name = string("op_8127_cast_fp16")]; + tensor var_8128_cast_fp16 = mul(x = var_8127_cast_fp16, y = sin_61_to_fp16)[name = string("op_8128_cast_fp16")]; + tensor k_197_cast_fp16 = add(x = var_8124_cast_fp16, y = var_8128_cast_fp16)[name = string("k_197_cast_fp16")]; + tensor var_8130 = const()[name = string("op_8130"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_65_cast_fp16 = reshape(shape = var_8130, x = k_197_cast_fp16)[name = string("cur_key_65_cast_fp16")]; + tensor var_8132_to_fp16 = const()[name = string("op_8132_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638144)))]; + tensor var_8133_cast_fp16 = mul(x = key_cache_65_cast_fp16, y = var_8132_to_fp16)[name = string("op_8133_cast_fp16")]; + tensor upd_65_to_fp16 = const()[name = string("upd_65_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638272)))]; + tensor var_8134_cast_fp16 = mul(x = cur_key_65_cast_fp16, y = upd_65_to_fp16)[name = string("op_8134_cast_fp16")]; + tensor key_65_cast_fp16 = add(x = var_8133_cast_fp16, y = var_8134_cast_fp16)[name = string("key_65_cast_fp16")]; + tensor var_8136_to_fp16 = const()[name = string("op_8136_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638144)))]; + tensor var_8137_cast_fp16 = mul(x = value_cache_65_cast_fp16, y = var_8136_to_fp16)[name = string("op_8137_cast_fp16")]; + tensor var_8138_cast_fp16 = mul(x = v_65_cast_fp16, y = upd_65_to_fp16)[name = string("op_8138_cast_fp16")]; + tensor value_65_cast_fp16 = add(x = var_8137_cast_fp16, y = var_8138_cast_fp16)[name = string("value_65_cast_fp16")]; + tensor var_8140 = const()[name = string("op_8140"), val = tensor([1, 8, 128, 16])]; + tensor kh_129_cast_fp16 = reshape(shape = var_8140, x = key_65_cast_fp16)[name = string("kh_129_cast_fp16")]; + tensor var_8142 = const()[name = string("op_8142"), val = tensor([1, 8, 128, 16])]; + tensor vh_129_cast_fp16 = reshape(shape = var_8142, x = value_65_cast_fp16)[name = string("vh_129_cast_fp16")]; + tensor transpose_128_perm_0 = const()[name = string("transpose_128_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_64_reps_0 = const()[name = string("tile_64_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_128_cast_fp16 = transpose(perm = transpose_128_perm_0, x = kh_129_cast_fp16)[name = string("transpose_287")]; + tensor tile_64_cast_fp16 = tile(reps = tile_64_reps_0, x = transpose_128_cast_fp16)[name = string("tile_64_cast_fp16")]; + tensor concat_161 = const()[name = string("concat_161"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_128_cast_fp16 = reshape(shape = concat_161, x = tile_64_cast_fp16)[name = string("reshape_128_cast_fp16")]; + tensor transpose_129_perm_0 = const()[name = string("transpose_129_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_162 = const()[name = string("concat_162"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_129_cast_fp16 = transpose(perm = transpose_129_perm_0, x = reshape_128_cast_fp16)[name = string("transpose_286")]; + tensor reshape_129_cast_fp16 = reshape(shape = concat_162, x = transpose_129_cast_fp16)[name = string("reshape_129_cast_fp16")]; + tensor transpose_130_perm_0 = const()[name = string("transpose_130_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_65_reps_0 = const()[name = string("tile_65_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_130_cast_fp16 = transpose(perm = transpose_130_perm_0, x = vh_129_cast_fp16)[name = string("transpose_285")]; + tensor tile_65_cast_fp16 = tile(reps = tile_65_reps_0, x = transpose_130_cast_fp16)[name = string("tile_65_cast_fp16")]; + tensor concat_163 = const()[name = string("concat_163"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_130_cast_fp16 = reshape(shape = concat_163, x = tile_65_cast_fp16)[name = string("reshape_130_cast_fp16")]; + tensor transpose_131_perm_0 = const()[name = string("transpose_131_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_164 = const()[name = string("concat_164"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_131_cast_fp16 = transpose(perm = transpose_131_perm_0, x = reshape_130_cast_fp16)[name = string("transpose_284")]; + tensor reshape_131_cast_fp16 = reshape(shape = concat_164, x = transpose_131_cast_fp16)[name = string("reshape_131_cast_fp16")]; + fp16 var_8146_to_fp16 = const()[name = string("op_8146_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_8147_cast_fp16 = mul(x = q_197_cast_fp16, y = var_8146_to_fp16)[name = string("op_8147_cast_fp16")]; + tensor transpose_445_perm_0 = const()[name = string("transpose_445_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_139_transpose_x_1 = const()[name = string("w_139_transpose_x_1"), val = bool(true)]; + bool w_139_transpose_y_1 = const()[name = string("w_139_transpose_y_1"), val = bool(false)]; + tensor transpose_445_cast_fp16 = transpose(perm = transpose_445_perm_0, x = reshape_129_cast_fp16)[name = string("transpose_283")]; + tensor w_139_cast_fp16 = matmul(transpose_x = w_139_transpose_x_1, transpose_y = w_139_transpose_y_1, x = var_8147_cast_fp16, y = transpose_445_cast_fp16)[name = string("w_139_cast_fp16")]; + tensor pad_65_to_fp16 = const()[name = string("pad_65_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638400)))]; + tensor var_8150_cast_fp16 = add(x = w_139_cast_fp16, y = pad_65_to_fp16)[name = string("op_8150_cast_fp16")]; + tensor w_141_cast_fp16 = softmax(axis = var_8025, x = var_8150_cast_fp16)[name = string("w_141_cast_fp16")]; + tensor transpose_446_perm_0 = const()[name = string("transpose_446_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_65_transpose_x_1 = const()[name = string("attn_65_transpose_x_1"), val = bool(false)]; + bool attn_65_transpose_y_1 = const()[name = string("attn_65_transpose_y_1"), val = bool(true)]; + tensor transpose_446_cast_fp16 = transpose(perm = transpose_446_perm_0, x = reshape_131_cast_fp16)[name = string("transpose_282")]; + tensor attn_65_cast_fp16 = matmul(transpose_x = attn_65_transpose_x_1, transpose_y = attn_65_transpose_y_1, x = transpose_446_cast_fp16, y = w_141_cast_fp16)[name = string("attn_65_cast_fp16")]; + tensor var_8154 = const()[name = string("op_8154"), val = tensor([1, 2048, 1, 1])]; + tensor input_343_cast_fp16 = reshape(shape = var_8154, x = attn_65_cast_fp16)[name = string("input_343_cast_fp16")]; + string attn_output_65_pad_type_0 = const()[name = string("attn_output_65_pad_type_0"), val = string("valid")]; + tensor attn_output_65_strides_0 = const()[name = string("attn_output_65_strides_0"), val = tensor([1, 1])]; + tensor attn_output_65_pad_0 = const()[name = string("attn_output_65_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_65_dilations_0 = const()[name = string("attn_output_65_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_65_groups_0 = const()[name = string("attn_output_65_groups_0"), val = int32(1)]; + tensor attn_output_65_cast_fp16 = conv(dilations = attn_output_65_dilations_0, groups = attn_output_65_groups_0, pad = attn_output_65_pad_0, pad_type = attn_output_65_pad_type_0, strides = attn_output_65_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_343_cast_fp16)[name = string("attn_output_65_cast_fp16")]; + tensor x_249_cast_fp16 = add(x = x_243_cast_fp16, y = attn_output_65_cast_fp16)[name = string("x_249_cast_fp16")]; + tensor var_8168_cast_fp16 = mul(x = x_249_cast_fp16, y = x_249_cast_fp16)[name = string("op_8168_cast_fp16")]; + tensor variance_273_axes_0 = const()[name = string("variance_273_axes_0"), val = tensor([1])]; + bool variance_273_keep_dims_0 = const()[name = string("variance_273_keep_dims_0"), val = bool(true)]; + tensor variance_273_cast_fp16 = reduce_mean(axes = variance_273_axes_0, keep_dims = variance_273_keep_dims_0, x = var_8168_cast_fp16)[name = string("variance_273_cast_fp16")]; + fp16 var_8171_to_fp16 = const()[name = string("op_8171_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8172_cast_fp16 = add(x = variance_273_cast_fp16, y = var_8171_to_fp16)[name = string("op_8172_cast_fp16")]; + fp32 var_8173_epsilon_0 = const()[name = string("op_8173_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8173_cast_fp16 = rsqrt(epsilon = var_8173_epsilon_0, x = var_8172_cast_fp16)[name = string("op_8173_cast_fp16")]; + tensor var_8174_cast_fp16 = mul(x = x_249_cast_fp16, y = var_8173_cast_fp16)[name = string("op_8174_cast_fp16")]; + tensor input_345_cast_fp16 = mul(x = var_8174_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_345_cast_fp16")]; + string input_347_pad_type_0 = const()[name = string("input_347_pad_type_0"), val = string("valid")]; + tensor input_347_strides_0 = const()[name = string("input_347_strides_0"), val = tensor([1, 1])]; + tensor input_347_pad_0 = const()[name = string("input_347_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_347_dilations_0 = const()[name = string("input_347_dilations_0"), val = tensor([1, 1])]; + int32 input_347_groups_0 = const()[name = string("input_347_groups_0"), val = int32(1)]; + tensor input_347_cast_fp16 = conv(dilations = input_347_dilations_0, groups = input_347_groups_0, pad = input_347_pad_0, pad_type = input_347_pad_type_0, strides = input_347_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_345_cast_fp16)[name = string("input_347_cast_fp16")]; + tensor var_8182_cast_fp16 = silu(x = input_347_cast_fp16)[name = string("op_8182_cast_fp16")]; + string var_8188_pad_type_0 = const()[name = string("op_8188_pad_type_0"), val = string("valid")]; + tensor var_8188_strides_0 = const()[name = string("op_8188_strides_0"), val = tensor([1, 1])]; + tensor var_8188_pad_0 = const()[name = string("op_8188_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8188_dilations_0 = const()[name = string("op_8188_dilations_0"), val = tensor([1, 1])]; + int32 var_8188_groups_0 = const()[name = string("op_8188_groups_0"), val = int32(1)]; + tensor var_8188_cast_fp16 = conv(dilations = var_8188_dilations_0, groups = var_8188_groups_0, pad = var_8188_pad_0, pad_type = var_8188_pad_type_0, strides = var_8188_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_345_cast_fp16)[name = string("op_8188_cast_fp16")]; + tensor input_349_cast_fp16 = mul(x = var_8182_cast_fp16, y = var_8188_cast_fp16)[name = string("input_349_cast_fp16")]; + string h_65_pad_type_0 = const()[name = string("h_65_pad_type_0"), val = string("valid")]; + tensor h_65_strides_0 = const()[name = string("h_65_strides_0"), val = tensor([1, 1])]; + tensor h_65_pad_0 = const()[name = string("h_65_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_65_dilations_0 = const()[name = string("h_65_dilations_0"), val = tensor([1, 1])]; + int32 h_65_groups_0 = const()[name = string("h_65_groups_0"), val = int32(1)]; + tensor h_65_cast_fp16 = conv(dilations = h_65_dilations_0, groups = h_65_groups_0, pad = h_65_pad_0, pad_type = h_65_pad_type_0, strides = h_65_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_349_cast_fp16)[name = string("h_65_cast_fp16")]; + tensor x_251_cast_fp16 = add(x = x_249_cast_fp16, y = h_65_cast_fp16)[name = string("x_251_cast_fp16")]; + tensor key_cache_67_begin_0 = const()[name = string("key_cache_67_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_67_end_0 = const()[name = string("key_cache_67_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_67_end_mask_0 = const()[name = string("key_cache_67_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_67_cast_fp16 = slice_by_index(begin = key_cache_67_begin_0, end = key_cache_67_end_0, end_mask = key_cache_67_end_mask_0, x = layer_key_caches_13_cast_fp16)[name = string("key_cache_67_cast_fp16")]; + tensor value_cache_67_begin_0 = const()[name = string("value_cache_67_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_67_end_0 = const()[name = string("value_cache_67_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_67_end_mask_0 = const()[name = string("value_cache_67_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_67_cast_fp16 = slice_by_index(begin = value_cache_67_begin_0, end = value_cache_67_end_0, end_mask = value_cache_67_end_mask_0, x = layer_value_caches_13_cast_fp16)[name = string("value_cache_67_cast_fp16")]; + int32 var_8241 = const()[name = string("op_8241"), val = int32(2)]; + int32 var_8245 = const()[name = string("op_8245"), val = int32(3)]; + tensor var_8260_cast_fp16 = mul(x = x_251_cast_fp16, y = x_251_cast_fp16)[name = string("op_8260_cast_fp16")]; + tensor variance_275_axes_0 = const()[name = string("variance_275_axes_0"), val = tensor([1])]; + bool variance_275_keep_dims_0 = const()[name = string("variance_275_keep_dims_0"), val = bool(true)]; + tensor variance_275_cast_fp16 = reduce_mean(axes = variance_275_axes_0, keep_dims = variance_275_keep_dims_0, x = var_8260_cast_fp16)[name = string("variance_275_cast_fp16")]; + fp16 var_8263_to_fp16 = const()[name = string("op_8263_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8264_cast_fp16 = add(x = variance_275_cast_fp16, y = var_8263_to_fp16)[name = string("op_8264_cast_fp16")]; + fp32 var_8265_epsilon_0 = const()[name = string("op_8265_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8265_cast_fp16 = rsqrt(epsilon = var_8265_epsilon_0, x = var_8264_cast_fp16)[name = string("op_8265_cast_fp16")]; + tensor var_8266_cast_fp16 = mul(x = x_251_cast_fp16, y = var_8265_cast_fp16)[name = string("op_8266_cast_fp16")]; + tensor input_351_cast_fp16 = mul(x = var_8266_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_351_cast_fp16")]; + string q_199_pad_type_0 = const()[name = string("q_199_pad_type_0"), val = string("valid")]; + tensor q_199_strides_0 = const()[name = string("q_199_strides_0"), val = tensor([1, 1])]; + tensor q_199_pad_0 = const()[name = string("q_199_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_199_dilations_0 = const()[name = string("q_199_dilations_0"), val = tensor([1, 1])]; + int32 q_199_groups_0 = const()[name = string("q_199_groups_0"), val = int32(1)]; + tensor q_199_cast_fp16 = conv(dilations = q_199_dilations_0, groups = q_199_groups_0, pad = q_199_pad_0, pad_type = q_199_pad_type_0, strides = q_199_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_351_cast_fp16)[name = string("q_199_cast_fp16")]; + string k_199_pad_type_0 = const()[name = string("k_199_pad_type_0"), val = string("valid")]; + tensor k_199_strides_0 = const()[name = string("k_199_strides_0"), val = tensor([1, 1])]; + tensor k_199_pad_0 = const()[name = string("k_199_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_199_dilations_0 = const()[name = string("k_199_dilations_0"), val = tensor([1, 1])]; + int32 k_199_groups_0 = const()[name = string("k_199_groups_0"), val = int32(1)]; + tensor k_199_cast_fp16 = conv(dilations = k_199_dilations_0, groups = k_199_groups_0, pad = k_199_pad_0, pad_type = k_199_pad_type_0, strides = k_199_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_351_cast_fp16)[name = string("k_199_cast_fp16")]; + string v_67_pad_type_0 = const()[name = string("v_67_pad_type_0"), val = string("valid")]; + tensor v_67_strides_0 = const()[name = string("v_67_strides_0"), val = tensor([1, 1])]; + tensor v_67_pad_0 = const()[name = string("v_67_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_67_dilations_0 = const()[name = string("v_67_dilations_0"), val = tensor([1, 1])]; + int32 v_67_groups_0 = const()[name = string("v_67_groups_0"), val = int32(1)]; + tensor v_67_cast_fp16 = conv(dilations = v_67_dilations_0, groups = v_67_groups_0, pad = v_67_pad_0, pad_type = v_67_pad_type_0, strides = v_67_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_351_cast_fp16)[name = string("v_67_cast_fp16")]; + tensor var_8300 = const()[name = string("op_8300"), val = tensor([16, 128, 1, 1])]; + tensor x_253_cast_fp16 = reshape(shape = var_8300, x = q_199_cast_fp16)[name = string("x_253_cast_fp16")]; + tensor var_8303_cast_fp16 = mul(x = x_253_cast_fp16, y = x_253_cast_fp16)[name = string("op_8303_cast_fp16")]; + tensor variance_277_axes_0 = const()[name = string("variance_277_axes_0"), val = tensor([1])]; + bool variance_277_keep_dims_0 = const()[name = string("variance_277_keep_dims_0"), val = bool(true)]; + tensor variance_277_cast_fp16 = reduce_mean(axes = variance_277_axes_0, keep_dims = variance_277_keep_dims_0, x = var_8303_cast_fp16)[name = string("variance_277_cast_fp16")]; + fp16 var_8306_to_fp16 = const()[name = string("op_8306_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8307_cast_fp16 = add(x = variance_277_cast_fp16, y = var_8306_to_fp16)[name = string("op_8307_cast_fp16")]; + fp32 var_8308_epsilon_0 = const()[name = string("op_8308_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8308_cast_fp16 = rsqrt(epsilon = var_8308_epsilon_0, x = var_8307_cast_fp16)[name = string("op_8308_cast_fp16")]; + tensor var_8309_cast_fp16 = mul(x = x_253_cast_fp16, y = var_8308_cast_fp16)[name = string("op_8309_cast_fp16")]; + tensor q_201_cast_fp16 = mul(x = var_8309_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_201_cast_fp16")]; + tensor var_8311 = const()[name = string("op_8311"), val = tensor([8, 128, 1, 1])]; + tensor x_255_cast_fp16 = reshape(shape = var_8311, x = k_199_cast_fp16)[name = string("x_255_cast_fp16")]; + tensor var_8314_cast_fp16 = mul(x = x_255_cast_fp16, y = x_255_cast_fp16)[name = string("op_8314_cast_fp16")]; + tensor variance_279_axes_0 = const()[name = string("variance_279_axes_0"), val = tensor([1])]; + bool variance_279_keep_dims_0 = const()[name = string("variance_279_keep_dims_0"), val = bool(true)]; + tensor variance_279_cast_fp16 = reduce_mean(axes = variance_279_axes_0, keep_dims = variance_279_keep_dims_0, x = var_8314_cast_fp16)[name = string("variance_279_cast_fp16")]; + fp16 var_8317_to_fp16 = const()[name = string("op_8317_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8318_cast_fp16 = add(x = variance_279_cast_fp16, y = var_8317_to_fp16)[name = string("op_8318_cast_fp16")]; + fp32 var_8319_epsilon_0 = const()[name = string("op_8319_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8319_cast_fp16 = rsqrt(epsilon = var_8319_epsilon_0, x = var_8318_cast_fp16)[name = string("op_8319_cast_fp16")]; + tensor var_8320_cast_fp16 = mul(x = x_255_cast_fp16, y = var_8319_cast_fp16)[name = string("op_8320_cast_fp16")]; + tensor k_201_cast_fp16 = mul(x = var_8320_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_201_cast_fp16")]; + tensor var_8322 = const()[name = string("op_8322"), val = tensor([1, 16, 128, 1])]; + tensor z_133_cast_fp16 = reshape(shape = var_8322, x = q_201_cast_fp16)[name = string("z_133_cast_fp16")]; + tensor var_8324 = const()[name = string("op_8324"), val = tensor([1, 8, 128, 1])]; + tensor z_135_cast_fp16 = reshape(shape = var_8324, x = k_201_cast_fp16)[name = string("z_135_cast_fp16")]; + tensor z1_133_begin_0 = const()[name = string("z1_133_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_133_end_0 = const()[name = string("z1_133_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_133_end_mask_0 = const()[name = string("z1_133_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_133_cast_fp16 = slice_by_index(begin = z1_133_begin_0, end = z1_133_end_0, end_mask = z1_133_end_mask_0, x = z_133_cast_fp16)[name = string("z1_133_cast_fp16")]; + tensor z2_133_begin_0 = const()[name = string("z2_133_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_133_end_0 = const()[name = string("z2_133_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_133_end_mask_0 = const()[name = string("z2_133_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_133_cast_fp16 = slice_by_index(begin = z2_133_begin_0, end = z2_133_end_0, end_mask = z2_133_end_mask_0, x = z_133_cast_fp16)[name = string("z2_133_cast_fp16")]; + tensor var_8332_cast_fp16 = mul(x = z_133_cast_fp16, y = cos_61_to_fp16)[name = string("op_8332_cast_fp16")]; + fp16 const_73_promoted_to_fp16 = const()[name = string("const_73_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8333_cast_fp16 = mul(x = z2_133_cast_fp16, y = const_73_promoted_to_fp16)[name = string("op_8333_cast_fp16")]; + bool var_8335_interleave_0 = const()[name = string("op_8335_interleave_0"), val = bool(false)]; + tensor var_8335_cast_fp16 = concat(axis = var_8241, interleave = var_8335_interleave_0, values = (var_8333_cast_fp16, z1_133_cast_fp16))[name = string("op_8335_cast_fp16")]; + tensor var_8336_cast_fp16 = mul(x = var_8335_cast_fp16, y = sin_61_to_fp16)[name = string("op_8336_cast_fp16")]; + tensor q_203_cast_fp16 = add(x = var_8332_cast_fp16, y = var_8336_cast_fp16)[name = string("q_203_cast_fp16")]; + tensor z1_135_begin_0 = const()[name = string("z1_135_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_135_end_0 = const()[name = string("z1_135_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_135_end_mask_0 = const()[name = string("z1_135_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_135_cast_fp16 = slice_by_index(begin = z1_135_begin_0, end = z1_135_end_0, end_mask = z1_135_end_mask_0, x = z_135_cast_fp16)[name = string("z1_135_cast_fp16")]; + tensor z2_135_begin_0 = const()[name = string("z2_135_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_135_end_0 = const()[name = string("z2_135_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_135_end_mask_0 = const()[name = string("z2_135_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_135_cast_fp16 = slice_by_index(begin = z2_135_begin_0, end = z2_135_end_0, end_mask = z2_135_end_mask_0, x = z_135_cast_fp16)[name = string("z2_135_cast_fp16")]; + tensor var_8344_cast_fp16 = mul(x = z_135_cast_fp16, y = cos_61_to_fp16)[name = string("op_8344_cast_fp16")]; + fp16 const_74_promoted_to_fp16 = const()[name = string("const_74_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8345_cast_fp16 = mul(x = z2_135_cast_fp16, y = const_74_promoted_to_fp16)[name = string("op_8345_cast_fp16")]; + bool var_8347_interleave_0 = const()[name = string("op_8347_interleave_0"), val = bool(false)]; + tensor var_8347_cast_fp16 = concat(axis = var_8241, interleave = var_8347_interleave_0, values = (var_8345_cast_fp16, z1_135_cast_fp16))[name = string("op_8347_cast_fp16")]; + tensor var_8348_cast_fp16 = mul(x = var_8347_cast_fp16, y = sin_61_to_fp16)[name = string("op_8348_cast_fp16")]; + tensor k_203_cast_fp16 = add(x = var_8344_cast_fp16, y = var_8348_cast_fp16)[name = string("k_203_cast_fp16")]; + tensor var_8350 = const()[name = string("op_8350"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_67_cast_fp16 = reshape(shape = var_8350, x = k_203_cast_fp16)[name = string("cur_key_67_cast_fp16")]; + tensor var_8352_to_fp16 = const()[name = string("op_8352_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638144)))]; + tensor var_8353_cast_fp16 = mul(x = key_cache_67_cast_fp16, y = var_8352_to_fp16)[name = string("op_8353_cast_fp16")]; + tensor upd_67_to_fp16 = const()[name = string("upd_67_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638272)))]; + tensor var_8354_cast_fp16 = mul(x = cur_key_67_cast_fp16, y = upd_67_to_fp16)[name = string("op_8354_cast_fp16")]; + tensor key_67_cast_fp16 = add(x = var_8353_cast_fp16, y = var_8354_cast_fp16)[name = string("key_67_cast_fp16")]; + tensor var_8356_to_fp16 = const()[name = string("op_8356_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638144)))]; + tensor var_8357_cast_fp16 = mul(x = value_cache_67_cast_fp16, y = var_8356_to_fp16)[name = string("op_8357_cast_fp16")]; + tensor var_8358_cast_fp16 = mul(x = v_67_cast_fp16, y = upd_67_to_fp16)[name = string("op_8358_cast_fp16")]; + tensor value_67_cast_fp16 = add(x = var_8357_cast_fp16, y = var_8358_cast_fp16)[name = string("value_67_cast_fp16")]; + tensor var_8360 = const()[name = string("op_8360"), val = tensor([1, 8, 128, 16])]; + tensor kh_133_cast_fp16 = reshape(shape = var_8360, x = key_67_cast_fp16)[name = string("kh_133_cast_fp16")]; + tensor var_8362 = const()[name = string("op_8362"), val = tensor([1, 8, 128, 16])]; + tensor vh_133_cast_fp16 = reshape(shape = var_8362, x = value_67_cast_fp16)[name = string("vh_133_cast_fp16")]; + tensor transpose_132_perm_0 = const()[name = string("transpose_132_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_66_reps_0 = const()[name = string("tile_66_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_132_cast_fp16 = transpose(perm = transpose_132_perm_0, x = kh_133_cast_fp16)[name = string("transpose_281")]; + tensor tile_66_cast_fp16 = tile(reps = tile_66_reps_0, x = transpose_132_cast_fp16)[name = string("tile_66_cast_fp16")]; + tensor concat_165 = const()[name = string("concat_165"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_132_cast_fp16 = reshape(shape = concat_165, x = tile_66_cast_fp16)[name = string("reshape_132_cast_fp16")]; + tensor transpose_133_perm_0 = const()[name = string("transpose_133_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_166 = const()[name = string("concat_166"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_133_cast_fp16 = transpose(perm = transpose_133_perm_0, x = reshape_132_cast_fp16)[name = string("transpose_280")]; + tensor reshape_133_cast_fp16 = reshape(shape = concat_166, x = transpose_133_cast_fp16)[name = string("reshape_133_cast_fp16")]; + tensor transpose_134_perm_0 = const()[name = string("transpose_134_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_67_reps_0 = const()[name = string("tile_67_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_134_cast_fp16 = transpose(perm = transpose_134_perm_0, x = vh_133_cast_fp16)[name = string("transpose_279")]; + tensor tile_67_cast_fp16 = tile(reps = tile_67_reps_0, x = transpose_134_cast_fp16)[name = string("tile_67_cast_fp16")]; + tensor concat_167 = const()[name = string("concat_167"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_134_cast_fp16 = reshape(shape = concat_167, x = tile_67_cast_fp16)[name = string("reshape_134_cast_fp16")]; + tensor transpose_135_perm_0 = const()[name = string("transpose_135_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_168 = const()[name = string("concat_168"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_135_cast_fp16 = transpose(perm = transpose_135_perm_0, x = reshape_134_cast_fp16)[name = string("transpose_278")]; + tensor reshape_135_cast_fp16 = reshape(shape = concat_168, x = transpose_135_cast_fp16)[name = string("reshape_135_cast_fp16")]; + fp16 var_8366_to_fp16 = const()[name = string("op_8366_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_8367_cast_fp16 = mul(x = q_203_cast_fp16, y = var_8366_to_fp16)[name = string("op_8367_cast_fp16")]; + tensor transpose_449_perm_0 = const()[name = string("transpose_449_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_143_transpose_x_1 = const()[name = string("w_143_transpose_x_1"), val = bool(true)]; + bool w_143_transpose_y_1 = const()[name = string("w_143_transpose_y_1"), val = bool(false)]; + tensor transpose_449_cast_fp16 = transpose(perm = transpose_449_perm_0, x = reshape_133_cast_fp16)[name = string("transpose_277")]; + tensor w_143_cast_fp16 = matmul(transpose_x = w_143_transpose_x_1, transpose_y = w_143_transpose_y_1, x = var_8367_cast_fp16, y = transpose_449_cast_fp16)[name = string("w_143_cast_fp16")]; + tensor pad_67_to_fp16 = const()[name = string("pad_67_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638400)))]; + tensor var_8370_cast_fp16 = add(x = w_143_cast_fp16, y = pad_67_to_fp16)[name = string("op_8370_cast_fp16")]; + tensor w_145_cast_fp16 = softmax(axis = var_8245, x = var_8370_cast_fp16)[name = string("w_145_cast_fp16")]; + tensor transpose_450_perm_0 = const()[name = string("transpose_450_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_67_transpose_x_1 = const()[name = string("attn_67_transpose_x_1"), val = bool(false)]; + bool attn_67_transpose_y_1 = const()[name = string("attn_67_transpose_y_1"), val = bool(true)]; + tensor transpose_450_cast_fp16 = transpose(perm = transpose_450_perm_0, x = reshape_135_cast_fp16)[name = string("transpose_276")]; + tensor attn_67_cast_fp16 = matmul(transpose_x = attn_67_transpose_x_1, transpose_y = attn_67_transpose_y_1, x = transpose_450_cast_fp16, y = w_145_cast_fp16)[name = string("attn_67_cast_fp16")]; + tensor var_8374 = const()[name = string("op_8374"), val = tensor([1, 2048, 1, 1])]; + tensor input_353_cast_fp16 = reshape(shape = var_8374, x = attn_67_cast_fp16)[name = string("input_353_cast_fp16")]; + string attn_output_67_pad_type_0 = const()[name = string("attn_output_67_pad_type_0"), val = string("valid")]; + tensor attn_output_67_strides_0 = const()[name = string("attn_output_67_strides_0"), val = tensor([1, 1])]; + tensor attn_output_67_pad_0 = const()[name = string("attn_output_67_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_67_dilations_0 = const()[name = string("attn_output_67_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_67_groups_0 = const()[name = string("attn_output_67_groups_0"), val = int32(1)]; + tensor attn_output_67_cast_fp16 = conv(dilations = attn_output_67_dilations_0, groups = attn_output_67_groups_0, pad = attn_output_67_pad_0, pad_type = attn_output_67_pad_type_0, strides = attn_output_67_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_353_cast_fp16)[name = string("attn_output_67_cast_fp16")]; + tensor x_257_cast_fp16 = add(x = x_251_cast_fp16, y = attn_output_67_cast_fp16)[name = string("x_257_cast_fp16")]; + tensor var_8388_cast_fp16 = mul(x = x_257_cast_fp16, y = x_257_cast_fp16)[name = string("op_8388_cast_fp16")]; + tensor variance_281_axes_0 = const()[name = string("variance_281_axes_0"), val = tensor([1])]; + bool variance_281_keep_dims_0 = const()[name = string("variance_281_keep_dims_0"), val = bool(true)]; + tensor variance_281_cast_fp16 = reduce_mean(axes = variance_281_axes_0, keep_dims = variance_281_keep_dims_0, x = var_8388_cast_fp16)[name = string("variance_281_cast_fp16")]; + fp16 var_8391_to_fp16 = const()[name = string("op_8391_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8392_cast_fp16 = add(x = variance_281_cast_fp16, y = var_8391_to_fp16)[name = string("op_8392_cast_fp16")]; + fp32 var_8393_epsilon_0 = const()[name = string("op_8393_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8393_cast_fp16 = rsqrt(epsilon = var_8393_epsilon_0, x = var_8392_cast_fp16)[name = string("op_8393_cast_fp16")]; + tensor var_8394_cast_fp16 = mul(x = x_257_cast_fp16, y = var_8393_cast_fp16)[name = string("op_8394_cast_fp16")]; + tensor input_355_cast_fp16 = mul(x = var_8394_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_355_cast_fp16")]; + string input_357_pad_type_0 = const()[name = string("input_357_pad_type_0"), val = string("valid")]; + tensor input_357_strides_0 = const()[name = string("input_357_strides_0"), val = tensor([1, 1])]; + tensor input_357_pad_0 = const()[name = string("input_357_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_357_dilations_0 = const()[name = string("input_357_dilations_0"), val = tensor([1, 1])]; + int32 input_357_groups_0 = const()[name = string("input_357_groups_0"), val = int32(1)]; + tensor input_357_cast_fp16 = conv(dilations = input_357_dilations_0, groups = input_357_groups_0, pad = input_357_pad_0, pad_type = input_357_pad_type_0, strides = input_357_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_355_cast_fp16)[name = string("input_357_cast_fp16")]; + tensor var_8402_cast_fp16 = silu(x = input_357_cast_fp16)[name = string("op_8402_cast_fp16")]; + string var_8408_pad_type_0 = const()[name = string("op_8408_pad_type_0"), val = string("valid")]; + tensor var_8408_strides_0 = const()[name = string("op_8408_strides_0"), val = tensor([1, 1])]; + tensor var_8408_pad_0 = const()[name = string("op_8408_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8408_dilations_0 = const()[name = string("op_8408_dilations_0"), val = tensor([1, 1])]; + int32 var_8408_groups_0 = const()[name = string("op_8408_groups_0"), val = int32(1)]; + tensor var_8408_cast_fp16 = conv(dilations = var_8408_dilations_0, groups = var_8408_groups_0, pad = var_8408_pad_0, pad_type = var_8408_pad_type_0, strides = var_8408_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_355_cast_fp16)[name = string("op_8408_cast_fp16")]; + tensor input_359_cast_fp16 = mul(x = var_8402_cast_fp16, y = var_8408_cast_fp16)[name = string("input_359_cast_fp16")]; + string h_67_pad_type_0 = const()[name = string("h_67_pad_type_0"), val = string("valid")]; + tensor h_67_strides_0 = const()[name = string("h_67_strides_0"), val = tensor([1, 1])]; + tensor h_67_pad_0 = const()[name = string("h_67_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_67_dilations_0 = const()[name = string("h_67_dilations_0"), val = tensor([1, 1])]; + int32 h_67_groups_0 = const()[name = string("h_67_groups_0"), val = int32(1)]; + tensor h_67_cast_fp16 = conv(dilations = h_67_dilations_0, groups = h_67_groups_0, pad = h_67_pad_0, pad_type = h_67_pad_type_0, strides = h_67_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_359_cast_fp16)[name = string("h_67_cast_fp16")]; + tensor x_259_cast_fp16 = add(x = x_257_cast_fp16, y = h_67_cast_fp16)[name = string("x_259_cast_fp16")]; + tensor key_cache_69_begin_0 = const()[name = string("key_cache_69_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_69_end_0 = const()[name = string("key_cache_69_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_69_end_mask_0 = const()[name = string("key_cache_69_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_69_cast_fp16 = slice_by_index(begin = key_cache_69_begin_0, end = key_cache_69_end_0, end_mask = key_cache_69_end_mask_0, x = layer_key_caches_13_cast_fp16)[name = string("key_cache_69_cast_fp16")]; + tensor value_cache_69_begin_0 = const()[name = string("value_cache_69_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_69_end_0 = const()[name = string("value_cache_69_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_69_end_mask_0 = const()[name = string("value_cache_69_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_69_cast_fp16 = slice_by_index(begin = value_cache_69_begin_0, end = value_cache_69_end_0, end_mask = value_cache_69_end_mask_0, x = layer_value_caches_13_cast_fp16)[name = string("value_cache_69_cast_fp16")]; + int32 var_8461 = const()[name = string("op_8461"), val = int32(2)]; + int32 var_8465 = const()[name = string("op_8465"), val = int32(3)]; + tensor var_8480_cast_fp16 = mul(x = x_259_cast_fp16, y = x_259_cast_fp16)[name = string("op_8480_cast_fp16")]; + tensor variance_283_axes_0 = const()[name = string("variance_283_axes_0"), val = tensor([1])]; + bool variance_283_keep_dims_0 = const()[name = string("variance_283_keep_dims_0"), val = bool(true)]; + tensor variance_283_cast_fp16 = reduce_mean(axes = variance_283_axes_0, keep_dims = variance_283_keep_dims_0, x = var_8480_cast_fp16)[name = string("variance_283_cast_fp16")]; + fp16 var_8483_to_fp16 = const()[name = string("op_8483_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8484_cast_fp16 = add(x = variance_283_cast_fp16, y = var_8483_to_fp16)[name = string("op_8484_cast_fp16")]; + fp32 var_8485_epsilon_0 = const()[name = string("op_8485_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8485_cast_fp16 = rsqrt(epsilon = var_8485_epsilon_0, x = var_8484_cast_fp16)[name = string("op_8485_cast_fp16")]; + tensor var_8486_cast_fp16 = mul(x = x_259_cast_fp16, y = var_8485_cast_fp16)[name = string("op_8486_cast_fp16")]; + tensor input_361_cast_fp16 = mul(x = var_8486_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_361_cast_fp16")]; + string q_205_pad_type_0 = const()[name = string("q_205_pad_type_0"), val = string("valid")]; + tensor q_205_strides_0 = const()[name = string("q_205_strides_0"), val = tensor([1, 1])]; + tensor q_205_pad_0 = const()[name = string("q_205_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_205_dilations_0 = const()[name = string("q_205_dilations_0"), val = tensor([1, 1])]; + int32 q_205_groups_0 = const()[name = string("q_205_groups_0"), val = int32(1)]; + tensor q_205_cast_fp16 = conv(dilations = q_205_dilations_0, groups = q_205_groups_0, pad = q_205_pad_0, pad_type = q_205_pad_type_0, strides = q_205_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_361_cast_fp16)[name = string("q_205_cast_fp16")]; + string k_205_pad_type_0 = const()[name = string("k_205_pad_type_0"), val = string("valid")]; + tensor k_205_strides_0 = const()[name = string("k_205_strides_0"), val = tensor([1, 1])]; + tensor k_205_pad_0 = const()[name = string("k_205_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_205_dilations_0 = const()[name = string("k_205_dilations_0"), val = tensor([1, 1])]; + int32 k_205_groups_0 = const()[name = string("k_205_groups_0"), val = int32(1)]; + tensor k_205_cast_fp16 = conv(dilations = k_205_dilations_0, groups = k_205_groups_0, pad = k_205_pad_0, pad_type = k_205_pad_type_0, strides = k_205_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_361_cast_fp16)[name = string("k_205_cast_fp16")]; + string v_69_pad_type_0 = const()[name = string("v_69_pad_type_0"), val = string("valid")]; + tensor v_69_strides_0 = const()[name = string("v_69_strides_0"), val = tensor([1, 1])]; + tensor v_69_pad_0 = const()[name = string("v_69_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_69_dilations_0 = const()[name = string("v_69_dilations_0"), val = tensor([1, 1])]; + int32 v_69_groups_0 = const()[name = string("v_69_groups_0"), val = int32(1)]; + tensor v_69_cast_fp16 = conv(dilations = v_69_dilations_0, groups = v_69_groups_0, pad = v_69_pad_0, pad_type = v_69_pad_type_0, strides = v_69_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_361_cast_fp16)[name = string("v_69_cast_fp16")]; + tensor var_8520 = const()[name = string("op_8520"), val = tensor([16, 128, 1, 1])]; + tensor x_261_cast_fp16 = reshape(shape = var_8520, x = q_205_cast_fp16)[name = string("x_261_cast_fp16")]; + tensor var_8523_cast_fp16 = mul(x = x_261_cast_fp16, y = x_261_cast_fp16)[name = string("op_8523_cast_fp16")]; + tensor variance_285_axes_0 = const()[name = string("variance_285_axes_0"), val = tensor([1])]; + bool variance_285_keep_dims_0 = const()[name = string("variance_285_keep_dims_0"), val = bool(true)]; + tensor variance_285_cast_fp16 = reduce_mean(axes = variance_285_axes_0, keep_dims = variance_285_keep_dims_0, x = var_8523_cast_fp16)[name = string("variance_285_cast_fp16")]; + fp16 var_8526_to_fp16 = const()[name = string("op_8526_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8527_cast_fp16 = add(x = variance_285_cast_fp16, y = var_8526_to_fp16)[name = string("op_8527_cast_fp16")]; + fp32 var_8528_epsilon_0 = const()[name = string("op_8528_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8528_cast_fp16 = rsqrt(epsilon = var_8528_epsilon_0, x = var_8527_cast_fp16)[name = string("op_8528_cast_fp16")]; + tensor var_8529_cast_fp16 = mul(x = x_261_cast_fp16, y = var_8528_cast_fp16)[name = string("op_8529_cast_fp16")]; + tensor q_207_cast_fp16 = mul(x = var_8529_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_207_cast_fp16")]; + tensor var_8531 = const()[name = string("op_8531"), val = tensor([8, 128, 1, 1])]; + tensor x_263_cast_fp16 = reshape(shape = var_8531, x = k_205_cast_fp16)[name = string("x_263_cast_fp16")]; + tensor var_8534_cast_fp16 = mul(x = x_263_cast_fp16, y = x_263_cast_fp16)[name = string("op_8534_cast_fp16")]; + tensor variance_287_axes_0 = const()[name = string("variance_287_axes_0"), val = tensor([1])]; + bool variance_287_keep_dims_0 = const()[name = string("variance_287_keep_dims_0"), val = bool(true)]; + tensor variance_287_cast_fp16 = reduce_mean(axes = variance_287_axes_0, keep_dims = variance_287_keep_dims_0, x = var_8534_cast_fp16)[name = string("variance_287_cast_fp16")]; + fp16 var_8537_to_fp16 = const()[name = string("op_8537_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8538_cast_fp16 = add(x = variance_287_cast_fp16, y = var_8537_to_fp16)[name = string("op_8538_cast_fp16")]; + fp32 var_8539_epsilon_0 = const()[name = string("op_8539_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8539_cast_fp16 = rsqrt(epsilon = var_8539_epsilon_0, x = var_8538_cast_fp16)[name = string("op_8539_cast_fp16")]; + tensor var_8540_cast_fp16 = mul(x = x_263_cast_fp16, y = var_8539_cast_fp16)[name = string("op_8540_cast_fp16")]; + tensor k_207_cast_fp16 = mul(x = var_8540_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_207_cast_fp16")]; + tensor var_8542 = const()[name = string("op_8542"), val = tensor([1, 16, 128, 1])]; + tensor z_137_cast_fp16 = reshape(shape = var_8542, x = q_207_cast_fp16)[name = string("z_137_cast_fp16")]; + tensor var_8544 = const()[name = string("op_8544"), val = tensor([1, 8, 128, 1])]; + tensor z_139_cast_fp16 = reshape(shape = var_8544, x = k_207_cast_fp16)[name = string("z_139_cast_fp16")]; + tensor z1_137_begin_0 = const()[name = string("z1_137_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_137_end_0 = const()[name = string("z1_137_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_137_end_mask_0 = const()[name = string("z1_137_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_137_cast_fp16 = slice_by_index(begin = z1_137_begin_0, end = z1_137_end_0, end_mask = z1_137_end_mask_0, x = z_137_cast_fp16)[name = string("z1_137_cast_fp16")]; + tensor z2_137_begin_0 = const()[name = string("z2_137_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_137_end_0 = const()[name = string("z2_137_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_137_end_mask_0 = const()[name = string("z2_137_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_137_cast_fp16 = slice_by_index(begin = z2_137_begin_0, end = z2_137_end_0, end_mask = z2_137_end_mask_0, x = z_137_cast_fp16)[name = string("z2_137_cast_fp16")]; + tensor var_8552_cast_fp16 = mul(x = z_137_cast_fp16, y = cos_61_to_fp16)[name = string("op_8552_cast_fp16")]; + fp16 const_75_promoted_to_fp16 = const()[name = string("const_75_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8553_cast_fp16 = mul(x = z2_137_cast_fp16, y = const_75_promoted_to_fp16)[name = string("op_8553_cast_fp16")]; + bool var_8555_interleave_0 = const()[name = string("op_8555_interleave_0"), val = bool(false)]; + tensor var_8555_cast_fp16 = concat(axis = var_8461, interleave = var_8555_interleave_0, values = (var_8553_cast_fp16, z1_137_cast_fp16))[name = string("op_8555_cast_fp16")]; + tensor var_8556_cast_fp16 = mul(x = var_8555_cast_fp16, y = sin_61_to_fp16)[name = string("op_8556_cast_fp16")]; + tensor q_209_cast_fp16 = add(x = var_8552_cast_fp16, y = var_8556_cast_fp16)[name = string("q_209_cast_fp16")]; + tensor z1_139_begin_0 = const()[name = string("z1_139_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_139_end_0 = const()[name = string("z1_139_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_139_end_mask_0 = const()[name = string("z1_139_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_139_cast_fp16 = slice_by_index(begin = z1_139_begin_0, end = z1_139_end_0, end_mask = z1_139_end_mask_0, x = z_139_cast_fp16)[name = string("z1_139_cast_fp16")]; + tensor z2_139_begin_0 = const()[name = string("z2_139_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_139_end_0 = const()[name = string("z2_139_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_139_end_mask_0 = const()[name = string("z2_139_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_139_cast_fp16 = slice_by_index(begin = z2_139_begin_0, end = z2_139_end_0, end_mask = z2_139_end_mask_0, x = z_139_cast_fp16)[name = string("z2_139_cast_fp16")]; + tensor var_8564_cast_fp16 = mul(x = z_139_cast_fp16, y = cos_61_to_fp16)[name = string("op_8564_cast_fp16")]; + fp16 const_76_promoted_to_fp16 = const()[name = string("const_76_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8565_cast_fp16 = mul(x = z2_139_cast_fp16, y = const_76_promoted_to_fp16)[name = string("op_8565_cast_fp16")]; + bool var_8567_interleave_0 = const()[name = string("op_8567_interleave_0"), val = bool(false)]; + tensor var_8567_cast_fp16 = concat(axis = var_8461, interleave = var_8567_interleave_0, values = (var_8565_cast_fp16, z1_139_cast_fp16))[name = string("op_8567_cast_fp16")]; + tensor var_8568_cast_fp16 = mul(x = var_8567_cast_fp16, y = sin_61_to_fp16)[name = string("op_8568_cast_fp16")]; + tensor k_209_cast_fp16 = add(x = var_8564_cast_fp16, y = var_8568_cast_fp16)[name = string("k_209_cast_fp16")]; + tensor var_8570 = const()[name = string("op_8570"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_69_cast_fp16 = reshape(shape = var_8570, x = k_209_cast_fp16)[name = string("cur_key_69_cast_fp16")]; + tensor var_8572_to_fp16 = const()[name = string("op_8572_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638144)))]; + tensor var_8573_cast_fp16 = mul(x = key_cache_69_cast_fp16, y = var_8572_to_fp16)[name = string("op_8573_cast_fp16")]; + tensor upd_69_to_fp16 = const()[name = string("upd_69_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638272)))]; + tensor var_8574_cast_fp16 = mul(x = cur_key_69_cast_fp16, y = upd_69_to_fp16)[name = string("op_8574_cast_fp16")]; + tensor key_69_cast_fp16 = add(x = var_8573_cast_fp16, y = var_8574_cast_fp16)[name = string("key_69_cast_fp16")]; + tensor var_8576_to_fp16 = const()[name = string("op_8576_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638144)))]; + tensor var_8577_cast_fp16 = mul(x = value_cache_69_cast_fp16, y = var_8576_to_fp16)[name = string("op_8577_cast_fp16")]; + tensor var_8578_cast_fp16 = mul(x = v_69_cast_fp16, y = upd_69_to_fp16)[name = string("op_8578_cast_fp16")]; + tensor value_69_cast_fp16 = add(x = var_8577_cast_fp16, y = var_8578_cast_fp16)[name = string("value_69_cast_fp16")]; + tensor var_8580 = const()[name = string("op_8580"), val = tensor([1, 8, 128, 16])]; + tensor kh_137_cast_fp16 = reshape(shape = var_8580, x = key_69_cast_fp16)[name = string("kh_137_cast_fp16")]; + tensor var_8582 = const()[name = string("op_8582"), val = tensor([1, 8, 128, 16])]; + tensor vh_137_cast_fp16 = reshape(shape = var_8582, x = value_69_cast_fp16)[name = string("vh_137_cast_fp16")]; + tensor transpose_136_perm_0 = const()[name = string("transpose_136_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_68_reps_0 = const()[name = string("tile_68_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_136_cast_fp16 = transpose(perm = transpose_136_perm_0, x = kh_137_cast_fp16)[name = string("transpose_275")]; + tensor tile_68_cast_fp16 = tile(reps = tile_68_reps_0, x = transpose_136_cast_fp16)[name = string("tile_68_cast_fp16")]; + tensor concat_169 = const()[name = string("concat_169"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_136_cast_fp16 = reshape(shape = concat_169, x = tile_68_cast_fp16)[name = string("reshape_136_cast_fp16")]; + tensor transpose_137_perm_0 = const()[name = string("transpose_137_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_170 = const()[name = string("concat_170"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_137_cast_fp16 = transpose(perm = transpose_137_perm_0, x = reshape_136_cast_fp16)[name = string("transpose_274")]; + tensor reshape_137_cast_fp16 = reshape(shape = concat_170, x = transpose_137_cast_fp16)[name = string("reshape_137_cast_fp16")]; + tensor transpose_138_perm_0 = const()[name = string("transpose_138_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_69_reps_0 = const()[name = string("tile_69_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_138_cast_fp16 = transpose(perm = transpose_138_perm_0, x = vh_137_cast_fp16)[name = string("transpose_273")]; + tensor tile_69_cast_fp16 = tile(reps = tile_69_reps_0, x = transpose_138_cast_fp16)[name = string("tile_69_cast_fp16")]; + tensor concat_171 = const()[name = string("concat_171"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_138_cast_fp16 = reshape(shape = concat_171, x = tile_69_cast_fp16)[name = string("reshape_138_cast_fp16")]; + tensor transpose_139_perm_0 = const()[name = string("transpose_139_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_172 = const()[name = string("concat_172"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_139_cast_fp16 = transpose(perm = transpose_139_perm_0, x = reshape_138_cast_fp16)[name = string("transpose_272")]; + tensor reshape_139_cast_fp16 = reshape(shape = concat_172, x = transpose_139_cast_fp16)[name = string("reshape_139_cast_fp16")]; + fp16 var_8586_to_fp16 = const()[name = string("op_8586_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_8587_cast_fp16 = mul(x = q_209_cast_fp16, y = var_8586_to_fp16)[name = string("op_8587_cast_fp16")]; + tensor transpose_453_perm_0 = const()[name = string("transpose_453_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_147_transpose_x_1 = const()[name = string("w_147_transpose_x_1"), val = bool(true)]; + bool w_147_transpose_y_1 = const()[name = string("w_147_transpose_y_1"), val = bool(false)]; + tensor transpose_453_cast_fp16 = transpose(perm = transpose_453_perm_0, x = reshape_137_cast_fp16)[name = string("transpose_271")]; + tensor w_147_cast_fp16 = matmul(transpose_x = w_147_transpose_x_1, transpose_y = w_147_transpose_y_1, x = var_8587_cast_fp16, y = transpose_453_cast_fp16)[name = string("w_147_cast_fp16")]; + tensor pad_69_to_fp16 = const()[name = string("pad_69_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638400)))]; + tensor var_8590_cast_fp16 = add(x = w_147_cast_fp16, y = pad_69_to_fp16)[name = string("op_8590_cast_fp16")]; + tensor w_149_cast_fp16 = softmax(axis = var_8465, x = var_8590_cast_fp16)[name = string("w_149_cast_fp16")]; + tensor transpose_454_perm_0 = const()[name = string("transpose_454_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_69_transpose_x_1 = const()[name = string("attn_69_transpose_x_1"), val = bool(false)]; + bool attn_69_transpose_y_1 = const()[name = string("attn_69_transpose_y_1"), val = bool(true)]; + tensor transpose_454_cast_fp16 = transpose(perm = transpose_454_perm_0, x = reshape_139_cast_fp16)[name = string("transpose_270")]; + tensor attn_69_cast_fp16 = matmul(transpose_x = attn_69_transpose_x_1, transpose_y = attn_69_transpose_y_1, x = transpose_454_cast_fp16, y = w_149_cast_fp16)[name = string("attn_69_cast_fp16")]; + tensor var_8594 = const()[name = string("op_8594"), val = tensor([1, 2048, 1, 1])]; + tensor input_363_cast_fp16 = reshape(shape = var_8594, x = attn_69_cast_fp16)[name = string("input_363_cast_fp16")]; + string attn_output_69_pad_type_0 = const()[name = string("attn_output_69_pad_type_0"), val = string("valid")]; + tensor attn_output_69_strides_0 = const()[name = string("attn_output_69_strides_0"), val = tensor([1, 1])]; + tensor attn_output_69_pad_0 = const()[name = string("attn_output_69_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_69_dilations_0 = const()[name = string("attn_output_69_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_69_groups_0 = const()[name = string("attn_output_69_groups_0"), val = int32(1)]; + tensor attn_output_69_cast_fp16 = conv(dilations = attn_output_69_dilations_0, groups = attn_output_69_groups_0, pad = attn_output_69_pad_0, pad_type = attn_output_69_pad_type_0, strides = attn_output_69_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_363_cast_fp16)[name = string("attn_output_69_cast_fp16")]; + tensor x_265_cast_fp16 = add(x = x_259_cast_fp16, y = attn_output_69_cast_fp16)[name = string("x_265_cast_fp16")]; + tensor var_8608_cast_fp16 = mul(x = x_265_cast_fp16, y = x_265_cast_fp16)[name = string("op_8608_cast_fp16")]; + tensor variance_289_axes_0 = const()[name = string("variance_289_axes_0"), val = tensor([1])]; + bool variance_289_keep_dims_0 = const()[name = string("variance_289_keep_dims_0"), val = bool(true)]; + tensor variance_289_cast_fp16 = reduce_mean(axes = variance_289_axes_0, keep_dims = variance_289_keep_dims_0, x = var_8608_cast_fp16)[name = string("variance_289_cast_fp16")]; + fp16 var_8611_to_fp16 = const()[name = string("op_8611_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8612_cast_fp16 = add(x = variance_289_cast_fp16, y = var_8611_to_fp16)[name = string("op_8612_cast_fp16")]; + fp32 var_8613_epsilon_0 = const()[name = string("op_8613_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8613_cast_fp16 = rsqrt(epsilon = var_8613_epsilon_0, x = var_8612_cast_fp16)[name = string("op_8613_cast_fp16")]; + tensor var_8614_cast_fp16 = mul(x = x_265_cast_fp16, y = var_8613_cast_fp16)[name = string("op_8614_cast_fp16")]; + tensor input_365_cast_fp16 = mul(x = var_8614_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_365_cast_fp16")]; + string input_367_pad_type_0 = const()[name = string("input_367_pad_type_0"), val = string("valid")]; + tensor input_367_strides_0 = const()[name = string("input_367_strides_0"), val = tensor([1, 1])]; + tensor input_367_pad_0 = const()[name = string("input_367_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_367_dilations_0 = const()[name = string("input_367_dilations_0"), val = tensor([1, 1])]; + int32 input_367_groups_0 = const()[name = string("input_367_groups_0"), val = int32(1)]; + tensor input_367_cast_fp16 = conv(dilations = input_367_dilations_0, groups = input_367_groups_0, pad = input_367_pad_0, pad_type = input_367_pad_type_0, strides = input_367_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_365_cast_fp16)[name = string("input_367_cast_fp16")]; + tensor var_8622_cast_fp16 = silu(x = input_367_cast_fp16)[name = string("op_8622_cast_fp16")]; + string var_8628_pad_type_0 = const()[name = string("op_8628_pad_type_0"), val = string("valid")]; + tensor var_8628_strides_0 = const()[name = string("op_8628_strides_0"), val = tensor([1, 1])]; + tensor var_8628_pad_0 = const()[name = string("op_8628_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8628_dilations_0 = const()[name = string("op_8628_dilations_0"), val = tensor([1, 1])]; + int32 var_8628_groups_0 = const()[name = string("op_8628_groups_0"), val = int32(1)]; + tensor var_8628_cast_fp16 = conv(dilations = var_8628_dilations_0, groups = var_8628_groups_0, pad = var_8628_pad_0, pad_type = var_8628_pad_type_0, strides = var_8628_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_365_cast_fp16)[name = string("op_8628_cast_fp16")]; + tensor input_369_cast_fp16 = mul(x = var_8622_cast_fp16, y = var_8628_cast_fp16)[name = string("input_369_cast_fp16")]; + string h_69_pad_type_0 = const()[name = string("h_69_pad_type_0"), val = string("valid")]; + tensor h_69_strides_0 = const()[name = string("h_69_strides_0"), val = tensor([1, 1])]; + tensor h_69_pad_0 = const()[name = string("h_69_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_69_dilations_0 = const()[name = string("h_69_dilations_0"), val = tensor([1, 1])]; + int32 h_69_groups_0 = const()[name = string("h_69_groups_0"), val = int32(1)]; + tensor h_69_cast_fp16 = conv(dilations = h_69_dilations_0, groups = h_69_groups_0, pad = h_69_pad_0, pad_type = h_69_pad_type_0, strides = h_69_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_369_cast_fp16)[name = string("h_69_cast_fp16")]; + tensor inputs_11_cast_fp16 = add(x = x_265_cast_fp16, y = h_69_cast_fp16)[name = string("inputs_11_cast_fp16")]; + int32 var_8656 = const()[name = string("op_8656"), val = int32(1)]; + bool layer_key_caches_15_interleave_0 = const()[name = string("layer_key_caches_15_interleave_0"), val = bool(false)]; + tensor layer_key_caches_15_cast_fp16 = concat(axis = var_8656, interleave = layer_key_caches_15_interleave_0, values = (key_61_cast_fp16, key_63_cast_fp16, key_65_cast_fp16, key_67_cast_fp16, key_69_cast_fp16))[name = string("layer_key_caches_15_cast_fp16")]; + int32 var_8659 = const()[name = string("op_8659"), val = int32(1)]; + bool layer_value_caches_15_interleave_0 = const()[name = string("layer_value_caches_15_interleave_0"), val = bool(false)]; + tensor layer_value_caches_15_cast_fp16 = concat(axis = var_8659, interleave = layer_value_caches_15_interleave_0, values = (value_61_cast_fp16, value_63_cast_fp16, value_65_cast_fp16, value_67_cast_fp16, value_69_cast_fp16))[name = string("layer_value_caches_15_cast_fp16")]; + tensor inputs_sq_11_cast_fp16 = mul(x = inputs_11_cast_fp16, y = inputs_11_cast_fp16)[name = string("inputs_sq_11_cast_fp16")]; + tensor variance_291_axes_0 = const()[name = string("variance_291_axes_0"), val = tensor([1])]; + bool variance_291_keep_dims_0 = const()[name = string("variance_291_keep_dims_0"), val = bool(true)]; + tensor variance_291_cast_fp16 = reduce_mean(axes = variance_291_axes_0, keep_dims = variance_291_keep_dims_0, x = inputs_sq_11_cast_fp16)[name = string("variance_291_cast_fp16")]; + fp16 var_8669_to_fp16 = const()[name = string("op_8669_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8670_cast_fp16 = add(x = variance_291_cast_fp16, y = var_8669_to_fp16)[name = string("op_8670_cast_fp16")]; + fp32 var_8671_epsilon_0 = const()[name = string("op_8671_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8671_cast_fp16 = rsqrt(epsilon = var_8671_epsilon_0, x = var_8670_cast_fp16)[name = string("op_8671_cast_fp16")]; + tensor hidden_states_11_cast_fp16 = mul(x = inputs_11_cast_fp16, y = var_8671_cast_fp16)[name = string("hidden_states_11_cast_fp16")]; + tensor input_371_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_11_cast_fp16)[name = string("input_371_cast_fp16")]; + string logits_21_pad_type_0 = const()[name = string("logits_21_pad_type_0"), val = string("valid")]; + tensor logits_21_strides_0 = const()[name = string("logits_21_strides_0"), val = tensor([1, 1])]; + tensor logits_21_pad_0 = const()[name = string("logits_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_21_dilations_0 = const()[name = string("logits_21_dilations_0"), val = tensor([1, 1])]; + int32 logits_21_groups_0 = const()[name = string("logits_21_groups_0"), val = int32(1)]; + tensor lm_heads_5_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89195648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91292864))))[name = string("lm_heads_5_weight_to_fp16_palettized")]; + tensor logits_21_cast_fp16 = conv(dilations = logits_21_dilations_0, groups = logits_21_groups_0, pad = logits_21_pad_0, pad_type = logits_21_pad_type_0, strides = logits_21_strides_0, weight = lm_heads_5_weight_to_fp16_palettized, x = input_371_cast_fp16)[name = string("logits_21_cast_fp16")]; + tensor var_8689 = const()[name = string("op_8689"), val = tensor([1, 2048])]; + tensor logits_23_cast_fp16 = reshape(shape = var_8689, x = logits_21_cast_fp16)[name = string("logits_23_cast_fp16")]; + tensor scaled_logits_11_cast_fp16 = real_div(x = logits_23_cast_fp16, y = temperature)[name = string("scaled_logits_11_cast_fp16")]; + int32 var_8699 = const()[name = string("op_8699"), val = int32(100)]; + int32 top_values_11_axis_0 = const()[name = string("top_values_11_axis_0"), val = int32(1)]; + bool top_values_11_ascending_0 = const()[name = string("top_values_11_ascending_0"), val = bool(false)]; + bool top_values_11_sort_0 = const()[name = string("top_values_11_sort_0"), val = bool(true)]; + bool top_values_11_return_indices_0 = const()[name = string("top_values_11_return_indices_0"), val = bool(true)]; + string top_values_11_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_11_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_11_cast_fp16_cast_uint16_0, tensor top_values_11_cast_fp16_cast_uint16_1 = topk(ascending = top_values_11_ascending_0, axis = top_values_11_axis_0, k = var_8699, output_indices_dtype = top_values_11_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_11_return_indices_0, sort = top_values_11_sort_0, x = scaled_logits_11_cast_fp16)[name = string("top_values_11_cast_fp16_cast_uint16")]; + tensor var_8705_cast_fp16 = mul(x = top_values_11_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_8705_cast_fp16")]; + tensor var_8709_cast_fp16 = add(x = var_8705_cast_fp16, y = var_2438_cast_fp16)[name = string("op_8709_cast_fp16")]; + tensor reduce_min_5_axes_0 = const()[name = string("reduce_min_5_axes_0"), val = tensor([1])]; + bool reduce_min_5_keep_dims_0 = const()[name = string("reduce_min_5_keep_dims_0"), val = bool(true)]; + tensor reduce_min_5_cast_fp16 = reduce_min(axes = reduce_min_5_axes_0, keep_dims = reduce_min_5_keep_dims_0, x = var_8709_cast_fp16)[name = string("reduce_min_5_cast_fp16")]; + tensor var_8712_cast_fp16 = greater_equal(x = scaled_logits_11_cast_fp16, y = reduce_min_5_cast_fp16)[name = string("op_8712_cast_fp16")]; + fp16 var_8713_value_0_to_fp16 = const()[name = string("op_8713_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_8713_cast_fp16 = fill_like(ref_tensor = scaled_logits_11_cast_fp16, value = var_8713_value_0_to_fp16)[name = string("op_8713_cast_fp16")]; + tensor masked_logits_11_cast_fp16 = select(a = scaled_logits_11_cast_fp16, b = var_8713_cast_fp16, cond = var_8712_cast_fp16)[name = string("masked_logits_11_cast_fp16")]; + tensor var_8717_begin_0 = const()[name = string("op_8717_begin_0"), val = tensor([5, 0])]; + tensor var_8717_end_0 = const()[name = string("op_8717_end_0"), val = tensor([6, 2048])]; + tensor var_8717_end_mask_0 = const()[name = string("op_8717_end_mask_0"), val = tensor([false, true])]; + tensor var_8717_squeeze_mask_0 = const()[name = string("op_8717_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_8717_cast_fp16 = slice_by_index(begin = var_8717_begin_0, end = var_8717_end_0, end_mask = var_8717_end_mask_0, squeeze_mask = var_8717_squeeze_mask_0, x = gumbel)[name = string("op_8717_cast_fp16")]; + tensor var_8720 = const()[name = string("op_8720"), val = tensor([1, 2048])]; + tensor var_8721_cast_fp16 = reshape(shape = var_8720, x = var_8717_cast_fp16)[name = string("op_8721_cast_fp16")]; + tensor noisy_logits_11_cast_fp16 = add(x = masked_logits_11_cast_fp16, y = var_8721_cast_fp16)[name = string("noisy_logits_11_cast_fp16")]; + int32 code_11_axis_0 = const()[name = string("code_11_axis_0"), val = int32(1)]; + bool code_11_keep_dims_0 = const()[name = string("code_11_keep_dims_0"), val = bool(false)]; + string code_11_output_dtype_0 = const()[name = string("code_11_output_dtype_0"), val = string("int32")]; + tensor code_11_cast_fp16 = reduce_argmax(axis = code_11_axis_0, keep_dims = code_11_keep_dims_0, output_dtype = code_11_output_dtype_0, x = noisy_logits_11_cast_fp16)[name = string("code_11_cast_fp16")]; + int32 var_8732 = const()[name = string("op_8732"), val = int32(10240)]; + tensor input_373 = add(x = code_11_cast_fp16, y = var_8732)[name = string("input_373")]; + int32 code_embed_21_axis_0 = const()[name = string("code_embed_21_axis_0"), val = int32(0)]; + int32 code_embed_21_batch_dims_0 = const()[name = string("code_embed_21_batch_dims_0"), val = int32(0)]; + bool code_embed_21_validate_indices_0 = const()[name = string("code_embed_21_validate_indices_0"), val = bool(false)]; + string input_373_to_uint16_dtype_0 = const()[name = string("input_373_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_373_to_uint16 = cast(dtype = input_373_to_uint16_dtype_0, x = input_373)[name = string("cast_9")]; + tensor code_embed_21_cast_fp16_cast_uint16 = gather(axis = code_embed_21_axis_0, batch_dims = code_embed_21_batch_dims_0, indices = input_373_to_uint16, validate_indices = code_embed_21_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_21_cast_fp16_cast_uint16")]; + tensor var_8736 = const()[name = string("op_8736"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_23_cast_fp16 = reshape(shape = var_8736, x = code_embed_21_cast_fp16_cast_uint16)[name = string("code_embed_23_cast_fp16")]; + tensor embed_sum_13_cast_fp16 = add(x = embed_sum_11_cast_fp16, y = code_embed_23_cast_fp16)[name = string("embed_sum_13_cast_fp16")]; + tensor key_cache_71_begin_0 = const()[name = string("key_cache_71_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_71_end_0 = const()[name = string("key_cache_71_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_71_end_mask_0 = const()[name = string("key_cache_71_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_71_cast_fp16 = slice_by_index(begin = key_cache_71_begin_0, end = key_cache_71_end_0, end_mask = key_cache_71_end_mask_0, x = layer_key_caches_15_cast_fp16)[name = string("key_cache_71_cast_fp16")]; + tensor value_cache_71_begin_0 = const()[name = string("value_cache_71_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_71_end_0 = const()[name = string("value_cache_71_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_71_end_mask_0 = const()[name = string("value_cache_71_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_71_cast_fp16 = slice_by_index(begin = value_cache_71_begin_0, end = value_cache_71_end_0, end_mask = value_cache_71_end_mask_0, x = layer_value_caches_15_cast_fp16)[name = string("value_cache_71_cast_fp16")]; + int32 var_8835 = const()[name = string("op_8835"), val = int32(2)]; + int32 var_8839 = const()[name = string("op_8839"), val = int32(3)]; + tensor var_8854_cast_fp16 = mul(x = code_embed_23_cast_fp16, y = code_embed_23_cast_fp16)[name = string("op_8854_cast_fp16")]; + tensor variance_293_axes_0 = const()[name = string("variance_293_axes_0"), val = tensor([1])]; + bool variance_293_keep_dims_0 = const()[name = string("variance_293_keep_dims_0"), val = bool(true)]; + tensor variance_293_cast_fp16 = reduce_mean(axes = variance_293_axes_0, keep_dims = variance_293_keep_dims_0, x = var_8854_cast_fp16)[name = string("variance_293_cast_fp16")]; + fp16 var_8857_to_fp16 = const()[name = string("op_8857_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8858_cast_fp16 = add(x = variance_293_cast_fp16, y = var_8857_to_fp16)[name = string("op_8858_cast_fp16")]; + fp32 var_8859_epsilon_0 = const()[name = string("op_8859_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8859_cast_fp16 = rsqrt(epsilon = var_8859_epsilon_0, x = var_8858_cast_fp16)[name = string("op_8859_cast_fp16")]; + tensor var_8860_cast_fp16 = mul(x = code_embed_23_cast_fp16, y = var_8859_cast_fp16)[name = string("op_8860_cast_fp16")]; + tensor input_375_cast_fp16 = mul(x = var_8860_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_375_cast_fp16")]; + string q_211_pad_type_0 = const()[name = string("q_211_pad_type_0"), val = string("valid")]; + tensor q_211_strides_0 = const()[name = string("q_211_strides_0"), val = tensor([1, 1])]; + tensor q_211_pad_0 = const()[name = string("q_211_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_211_dilations_0 = const()[name = string("q_211_dilations_0"), val = tensor([1, 1])]; + int32 q_211_groups_0 = const()[name = string("q_211_groups_0"), val = int32(1)]; + tensor q_211_cast_fp16 = conv(dilations = q_211_dilations_0, groups = q_211_groups_0, pad = q_211_pad_0, pad_type = q_211_pad_type_0, strides = q_211_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_375_cast_fp16)[name = string("q_211_cast_fp16")]; + string k_211_pad_type_0 = const()[name = string("k_211_pad_type_0"), val = string("valid")]; + tensor k_211_strides_0 = const()[name = string("k_211_strides_0"), val = tensor([1, 1])]; + tensor k_211_pad_0 = const()[name = string("k_211_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_211_dilations_0 = const()[name = string("k_211_dilations_0"), val = tensor([1, 1])]; + int32 k_211_groups_0 = const()[name = string("k_211_groups_0"), val = int32(1)]; + tensor k_211_cast_fp16 = conv(dilations = k_211_dilations_0, groups = k_211_groups_0, pad = k_211_pad_0, pad_type = k_211_pad_type_0, strides = k_211_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_375_cast_fp16)[name = string("k_211_cast_fp16")]; + string v_71_pad_type_0 = const()[name = string("v_71_pad_type_0"), val = string("valid")]; + tensor v_71_strides_0 = const()[name = string("v_71_strides_0"), val = tensor([1, 1])]; + tensor v_71_pad_0 = const()[name = string("v_71_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_71_dilations_0 = const()[name = string("v_71_dilations_0"), val = tensor([1, 1])]; + int32 v_71_groups_0 = const()[name = string("v_71_groups_0"), val = int32(1)]; + tensor v_71_cast_fp16 = conv(dilations = v_71_dilations_0, groups = v_71_groups_0, pad = v_71_pad_0, pad_type = v_71_pad_type_0, strides = v_71_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_375_cast_fp16)[name = string("v_71_cast_fp16")]; + tensor var_8894 = const()[name = string("op_8894"), val = tensor([16, 128, 1, 1])]; + tensor x_267_cast_fp16 = reshape(shape = var_8894, x = q_211_cast_fp16)[name = string("x_267_cast_fp16")]; + tensor var_8897_cast_fp16 = mul(x = x_267_cast_fp16, y = x_267_cast_fp16)[name = string("op_8897_cast_fp16")]; + tensor variance_295_axes_0 = const()[name = string("variance_295_axes_0"), val = tensor([1])]; + bool variance_295_keep_dims_0 = const()[name = string("variance_295_keep_dims_0"), val = bool(true)]; + tensor variance_295_cast_fp16 = reduce_mean(axes = variance_295_axes_0, keep_dims = variance_295_keep_dims_0, x = var_8897_cast_fp16)[name = string("variance_295_cast_fp16")]; + fp16 var_8900_to_fp16 = const()[name = string("op_8900_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8901_cast_fp16 = add(x = variance_295_cast_fp16, y = var_8900_to_fp16)[name = string("op_8901_cast_fp16")]; + fp32 var_8902_epsilon_0 = const()[name = string("op_8902_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8902_cast_fp16 = rsqrt(epsilon = var_8902_epsilon_0, x = var_8901_cast_fp16)[name = string("op_8902_cast_fp16")]; + tensor var_8903_cast_fp16 = mul(x = x_267_cast_fp16, y = var_8902_cast_fp16)[name = string("op_8903_cast_fp16")]; + tensor q_213_cast_fp16 = mul(x = var_8903_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_213_cast_fp16")]; + tensor var_8905 = const()[name = string("op_8905"), val = tensor([8, 128, 1, 1])]; + tensor x_269_cast_fp16 = reshape(shape = var_8905, x = k_211_cast_fp16)[name = string("x_269_cast_fp16")]; + tensor var_8908_cast_fp16 = mul(x = x_269_cast_fp16, y = x_269_cast_fp16)[name = string("op_8908_cast_fp16")]; + tensor variance_297_axes_0 = const()[name = string("variance_297_axes_0"), val = tensor([1])]; + bool variance_297_keep_dims_0 = const()[name = string("variance_297_keep_dims_0"), val = bool(true)]; + tensor variance_297_cast_fp16 = reduce_mean(axes = variance_297_axes_0, keep_dims = variance_297_keep_dims_0, x = var_8908_cast_fp16)[name = string("variance_297_cast_fp16")]; + fp16 var_8911_to_fp16 = const()[name = string("op_8911_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8912_cast_fp16 = add(x = variance_297_cast_fp16, y = var_8911_to_fp16)[name = string("op_8912_cast_fp16")]; + fp32 var_8913_epsilon_0 = const()[name = string("op_8913_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8913_cast_fp16 = rsqrt(epsilon = var_8913_epsilon_0, x = var_8912_cast_fp16)[name = string("op_8913_cast_fp16")]; + tensor var_8914_cast_fp16 = mul(x = x_269_cast_fp16, y = var_8913_cast_fp16)[name = string("op_8914_cast_fp16")]; + tensor k_213_cast_fp16 = mul(x = var_8914_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_213_cast_fp16")]; + tensor var_8916 = const()[name = string("op_8916"), val = tensor([1, 16, 128, 1])]; + tensor z_141_cast_fp16 = reshape(shape = var_8916, x = q_213_cast_fp16)[name = string("z_141_cast_fp16")]; + tensor var_8918 = const()[name = string("op_8918"), val = tensor([1, 8, 128, 1])]; + tensor z_143_cast_fp16 = reshape(shape = var_8918, x = k_213_cast_fp16)[name = string("z_143_cast_fp16")]; + tensor z1_141_begin_0 = const()[name = string("z1_141_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_141_end_0 = const()[name = string("z1_141_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_141_end_mask_0 = const()[name = string("z1_141_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_141_cast_fp16 = slice_by_index(begin = z1_141_begin_0, end = z1_141_end_0, end_mask = z1_141_end_mask_0, x = z_141_cast_fp16)[name = string("z1_141_cast_fp16")]; + tensor z2_141_begin_0 = const()[name = string("z2_141_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_141_end_0 = const()[name = string("z2_141_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_141_end_mask_0 = const()[name = string("z2_141_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_141_cast_fp16 = slice_by_index(begin = z2_141_begin_0, end = z2_141_end_0, end_mask = z2_141_end_mask_0, x = z_141_cast_fp16)[name = string("z2_141_cast_fp16")]; + tensor cos_71_to_fp16 = const()[name = string("cos_71_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638528)))]; + tensor var_8926_cast_fp16 = mul(x = z_141_cast_fp16, y = cos_71_to_fp16)[name = string("op_8926_cast_fp16")]; + fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8927_cast_fp16 = mul(x = z2_141_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_8927_cast_fp16")]; + bool var_8929_interleave_0 = const()[name = string("op_8929_interleave_0"), val = bool(false)]; + tensor var_8929_cast_fp16 = concat(axis = var_8835, interleave = var_8929_interleave_0, values = (var_8927_cast_fp16, z1_141_cast_fp16))[name = string("op_8929_cast_fp16")]; + tensor sin_71_to_fp16 = const()[name = string("sin_71_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141638848)))]; + tensor var_8930_cast_fp16 = mul(x = var_8929_cast_fp16, y = sin_71_to_fp16)[name = string("op_8930_cast_fp16")]; + tensor q_215_cast_fp16 = add(x = var_8926_cast_fp16, y = var_8930_cast_fp16)[name = string("q_215_cast_fp16")]; + tensor z1_143_begin_0 = const()[name = string("z1_143_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_143_end_0 = const()[name = string("z1_143_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_143_end_mask_0 = const()[name = string("z1_143_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_143_cast_fp16 = slice_by_index(begin = z1_143_begin_0, end = z1_143_end_0, end_mask = z1_143_end_mask_0, x = z_143_cast_fp16)[name = string("z1_143_cast_fp16")]; + tensor z2_143_begin_0 = const()[name = string("z2_143_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_143_end_0 = const()[name = string("z2_143_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_143_end_mask_0 = const()[name = string("z2_143_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_143_cast_fp16 = slice_by_index(begin = z2_143_begin_0, end = z2_143_end_0, end_mask = z2_143_end_mask_0, x = z_143_cast_fp16)[name = string("z2_143_cast_fp16")]; + tensor var_8938_cast_fp16 = mul(x = z_143_cast_fp16, y = cos_71_to_fp16)[name = string("op_8938_cast_fp16")]; + fp16 const_79_promoted_to_fp16 = const()[name = string("const_79_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8939_cast_fp16 = mul(x = z2_143_cast_fp16, y = const_79_promoted_to_fp16)[name = string("op_8939_cast_fp16")]; + bool var_8941_interleave_0 = const()[name = string("op_8941_interleave_0"), val = bool(false)]; + tensor var_8941_cast_fp16 = concat(axis = var_8835, interleave = var_8941_interleave_0, values = (var_8939_cast_fp16, z1_143_cast_fp16))[name = string("op_8941_cast_fp16")]; + tensor var_8942_cast_fp16 = mul(x = var_8941_cast_fp16, y = sin_71_to_fp16)[name = string("op_8942_cast_fp16")]; + tensor k_215_cast_fp16 = add(x = var_8938_cast_fp16, y = var_8942_cast_fp16)[name = string("k_215_cast_fp16")]; + tensor var_8944 = const()[name = string("op_8944"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_71_cast_fp16 = reshape(shape = var_8944, x = k_215_cast_fp16)[name = string("cur_key_71_cast_fp16")]; + tensor var_8946_to_fp16 = const()[name = string("op_8946_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639168)))]; + tensor var_8947_cast_fp16 = mul(x = key_cache_71_cast_fp16, y = var_8946_to_fp16)[name = string("op_8947_cast_fp16")]; + tensor upd_71_to_fp16 = const()[name = string("upd_71_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639296)))]; + tensor var_8948_cast_fp16 = mul(x = cur_key_71_cast_fp16, y = upd_71_to_fp16)[name = string("op_8948_cast_fp16")]; + tensor key_71_cast_fp16 = add(x = var_8947_cast_fp16, y = var_8948_cast_fp16)[name = string("key_71_cast_fp16")]; + tensor var_8950_to_fp16 = const()[name = string("op_8950_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639168)))]; + tensor var_8951_cast_fp16 = mul(x = value_cache_71_cast_fp16, y = var_8950_to_fp16)[name = string("op_8951_cast_fp16")]; + tensor var_8952_cast_fp16 = mul(x = v_71_cast_fp16, y = upd_71_to_fp16)[name = string("op_8952_cast_fp16")]; + tensor value_71_cast_fp16 = add(x = var_8951_cast_fp16, y = var_8952_cast_fp16)[name = string("value_71_cast_fp16")]; + tensor var_8954 = const()[name = string("op_8954"), val = tensor([1, 8, 128, 16])]; + tensor kh_141_cast_fp16 = reshape(shape = var_8954, x = key_71_cast_fp16)[name = string("kh_141_cast_fp16")]; + tensor var_8956 = const()[name = string("op_8956"), val = tensor([1, 8, 128, 16])]; + tensor vh_141_cast_fp16 = reshape(shape = var_8956, x = value_71_cast_fp16)[name = string("vh_141_cast_fp16")]; + tensor transpose_140_perm_0 = const()[name = string("transpose_140_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_70_reps_0 = const()[name = string("tile_70_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_140_cast_fp16 = transpose(perm = transpose_140_perm_0, x = kh_141_cast_fp16)[name = string("transpose_269")]; + tensor tile_70_cast_fp16 = tile(reps = tile_70_reps_0, x = transpose_140_cast_fp16)[name = string("tile_70_cast_fp16")]; + tensor concat_178 = const()[name = string("concat_178"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_140_cast_fp16 = reshape(shape = concat_178, x = tile_70_cast_fp16)[name = string("reshape_140_cast_fp16")]; + tensor transpose_141_perm_0 = const()[name = string("transpose_141_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_179 = const()[name = string("concat_179"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_141_cast_fp16 = transpose(perm = transpose_141_perm_0, x = reshape_140_cast_fp16)[name = string("transpose_268")]; + tensor reshape_141_cast_fp16 = reshape(shape = concat_179, x = transpose_141_cast_fp16)[name = string("reshape_141_cast_fp16")]; + tensor transpose_142_perm_0 = const()[name = string("transpose_142_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_71_reps_0 = const()[name = string("tile_71_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_142_cast_fp16 = transpose(perm = transpose_142_perm_0, x = vh_141_cast_fp16)[name = string("transpose_267")]; + tensor tile_71_cast_fp16 = tile(reps = tile_71_reps_0, x = transpose_142_cast_fp16)[name = string("tile_71_cast_fp16")]; + tensor concat_180 = const()[name = string("concat_180"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_142_cast_fp16 = reshape(shape = concat_180, x = tile_71_cast_fp16)[name = string("reshape_142_cast_fp16")]; + tensor transpose_143_perm_0 = const()[name = string("transpose_143_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_181 = const()[name = string("concat_181"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_143_cast_fp16 = transpose(perm = transpose_143_perm_0, x = reshape_142_cast_fp16)[name = string("transpose_266")]; + tensor reshape_143_cast_fp16 = reshape(shape = concat_181, x = transpose_143_cast_fp16)[name = string("reshape_143_cast_fp16")]; + fp16 var_8960_to_fp16 = const()[name = string("op_8960_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_8961_cast_fp16 = mul(x = q_215_cast_fp16, y = var_8960_to_fp16)[name = string("op_8961_cast_fp16")]; + tensor transpose_457_perm_0 = const()[name = string("transpose_457_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_153_transpose_x_1 = const()[name = string("w_153_transpose_x_1"), val = bool(true)]; + bool w_153_transpose_y_1 = const()[name = string("w_153_transpose_y_1"), val = bool(false)]; + tensor transpose_457_cast_fp16 = transpose(perm = transpose_457_perm_0, x = reshape_141_cast_fp16)[name = string("transpose_265")]; + tensor w_153_cast_fp16 = matmul(transpose_x = w_153_transpose_x_1, transpose_y = w_153_transpose_y_1, x = var_8961_cast_fp16, y = transpose_457_cast_fp16)[name = string("w_153_cast_fp16")]; + tensor pad_71_to_fp16 = const()[name = string("pad_71_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639424)))]; + tensor var_8964_cast_fp16 = add(x = w_153_cast_fp16, y = pad_71_to_fp16)[name = string("op_8964_cast_fp16")]; + tensor w_155_cast_fp16 = softmax(axis = var_8839, x = var_8964_cast_fp16)[name = string("w_155_cast_fp16")]; + tensor transpose_458_perm_0 = const()[name = string("transpose_458_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_71_transpose_x_1 = const()[name = string("attn_71_transpose_x_1"), val = bool(false)]; + bool attn_71_transpose_y_1 = const()[name = string("attn_71_transpose_y_1"), val = bool(true)]; + tensor transpose_458_cast_fp16 = transpose(perm = transpose_458_perm_0, x = reshape_143_cast_fp16)[name = string("transpose_264")]; + tensor attn_71_cast_fp16 = matmul(transpose_x = attn_71_transpose_x_1, transpose_y = attn_71_transpose_y_1, x = transpose_458_cast_fp16, y = w_155_cast_fp16)[name = string("attn_71_cast_fp16")]; + tensor var_8968 = const()[name = string("op_8968"), val = tensor([1, 2048, 1, 1])]; + tensor input_377_cast_fp16 = reshape(shape = var_8968, x = attn_71_cast_fp16)[name = string("input_377_cast_fp16")]; + string attn_output_71_pad_type_0 = const()[name = string("attn_output_71_pad_type_0"), val = string("valid")]; + tensor attn_output_71_strides_0 = const()[name = string("attn_output_71_strides_0"), val = tensor([1, 1])]; + tensor attn_output_71_pad_0 = const()[name = string("attn_output_71_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_71_dilations_0 = const()[name = string("attn_output_71_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_71_groups_0 = const()[name = string("attn_output_71_groups_0"), val = int32(1)]; + tensor attn_output_71_cast_fp16 = conv(dilations = attn_output_71_dilations_0, groups = attn_output_71_groups_0, pad = attn_output_71_pad_0, pad_type = attn_output_71_pad_type_0, strides = attn_output_71_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_377_cast_fp16)[name = string("attn_output_71_cast_fp16")]; + tensor x_271_cast_fp16 = add(x = code_embed_23_cast_fp16, y = attn_output_71_cast_fp16)[name = string("x_271_cast_fp16")]; + tensor var_8982_cast_fp16 = mul(x = x_271_cast_fp16, y = x_271_cast_fp16)[name = string("op_8982_cast_fp16")]; + tensor variance_299_axes_0 = const()[name = string("variance_299_axes_0"), val = tensor([1])]; + bool variance_299_keep_dims_0 = const()[name = string("variance_299_keep_dims_0"), val = bool(true)]; + tensor variance_299_cast_fp16 = reduce_mean(axes = variance_299_axes_0, keep_dims = variance_299_keep_dims_0, x = var_8982_cast_fp16)[name = string("variance_299_cast_fp16")]; + fp16 var_8985_to_fp16 = const()[name = string("op_8985_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8986_cast_fp16 = add(x = variance_299_cast_fp16, y = var_8985_to_fp16)[name = string("op_8986_cast_fp16")]; + fp32 var_8987_epsilon_0 = const()[name = string("op_8987_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8987_cast_fp16 = rsqrt(epsilon = var_8987_epsilon_0, x = var_8986_cast_fp16)[name = string("op_8987_cast_fp16")]; + tensor var_8988_cast_fp16 = mul(x = x_271_cast_fp16, y = var_8987_cast_fp16)[name = string("op_8988_cast_fp16")]; + tensor input_379_cast_fp16 = mul(x = var_8988_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_379_cast_fp16")]; + string input_381_pad_type_0 = const()[name = string("input_381_pad_type_0"), val = string("valid")]; + tensor input_381_strides_0 = const()[name = string("input_381_strides_0"), val = tensor([1, 1])]; + tensor input_381_pad_0 = const()[name = string("input_381_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_381_dilations_0 = const()[name = string("input_381_dilations_0"), val = tensor([1, 1])]; + int32 input_381_groups_0 = const()[name = string("input_381_groups_0"), val = int32(1)]; + tensor input_381_cast_fp16 = conv(dilations = input_381_dilations_0, groups = input_381_groups_0, pad = input_381_pad_0, pad_type = input_381_pad_type_0, strides = input_381_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_379_cast_fp16)[name = string("input_381_cast_fp16")]; + tensor var_8996_cast_fp16 = silu(x = input_381_cast_fp16)[name = string("op_8996_cast_fp16")]; + string var_9002_pad_type_0 = const()[name = string("op_9002_pad_type_0"), val = string("valid")]; + tensor var_9002_strides_0 = const()[name = string("op_9002_strides_0"), val = tensor([1, 1])]; + tensor var_9002_pad_0 = const()[name = string("op_9002_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9002_dilations_0 = const()[name = string("op_9002_dilations_0"), val = tensor([1, 1])]; + int32 var_9002_groups_0 = const()[name = string("op_9002_groups_0"), val = int32(1)]; + tensor var_9002_cast_fp16 = conv(dilations = var_9002_dilations_0, groups = var_9002_groups_0, pad = var_9002_pad_0, pad_type = var_9002_pad_type_0, strides = var_9002_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_379_cast_fp16)[name = string("op_9002_cast_fp16")]; + tensor input_383_cast_fp16 = mul(x = var_8996_cast_fp16, y = var_9002_cast_fp16)[name = string("input_383_cast_fp16")]; + string h_71_pad_type_0 = const()[name = string("h_71_pad_type_0"), val = string("valid")]; + tensor h_71_strides_0 = const()[name = string("h_71_strides_0"), val = tensor([1, 1])]; + tensor h_71_pad_0 = const()[name = string("h_71_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_71_dilations_0 = const()[name = string("h_71_dilations_0"), val = tensor([1, 1])]; + int32 h_71_groups_0 = const()[name = string("h_71_groups_0"), val = int32(1)]; + tensor h_71_cast_fp16 = conv(dilations = h_71_dilations_0, groups = h_71_groups_0, pad = h_71_pad_0, pad_type = h_71_pad_type_0, strides = h_71_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_383_cast_fp16)[name = string("h_71_cast_fp16")]; + tensor x_273_cast_fp16 = add(x = x_271_cast_fp16, y = h_71_cast_fp16)[name = string("x_273_cast_fp16")]; + tensor key_cache_73_begin_0 = const()[name = string("key_cache_73_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_73_end_0 = const()[name = string("key_cache_73_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_73_end_mask_0 = const()[name = string("key_cache_73_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_73_cast_fp16 = slice_by_index(begin = key_cache_73_begin_0, end = key_cache_73_end_0, end_mask = key_cache_73_end_mask_0, x = layer_key_caches_15_cast_fp16)[name = string("key_cache_73_cast_fp16")]; + tensor value_cache_73_begin_0 = const()[name = string("value_cache_73_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_73_end_0 = const()[name = string("value_cache_73_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_73_end_mask_0 = const()[name = string("value_cache_73_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_73_cast_fp16 = slice_by_index(begin = value_cache_73_begin_0, end = value_cache_73_end_0, end_mask = value_cache_73_end_mask_0, x = layer_value_caches_15_cast_fp16)[name = string("value_cache_73_cast_fp16")]; + int32 var_9055 = const()[name = string("op_9055"), val = int32(2)]; + int32 var_9059 = const()[name = string("op_9059"), val = int32(3)]; + tensor var_9074_cast_fp16 = mul(x = x_273_cast_fp16, y = x_273_cast_fp16)[name = string("op_9074_cast_fp16")]; + tensor variance_301_axes_0 = const()[name = string("variance_301_axes_0"), val = tensor([1])]; + bool variance_301_keep_dims_0 = const()[name = string("variance_301_keep_dims_0"), val = bool(true)]; + tensor variance_301_cast_fp16 = reduce_mean(axes = variance_301_axes_0, keep_dims = variance_301_keep_dims_0, x = var_9074_cast_fp16)[name = string("variance_301_cast_fp16")]; + fp16 var_9077_to_fp16 = const()[name = string("op_9077_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9078_cast_fp16 = add(x = variance_301_cast_fp16, y = var_9077_to_fp16)[name = string("op_9078_cast_fp16")]; + fp32 var_9079_epsilon_0 = const()[name = string("op_9079_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9079_cast_fp16 = rsqrt(epsilon = var_9079_epsilon_0, x = var_9078_cast_fp16)[name = string("op_9079_cast_fp16")]; + tensor var_9080_cast_fp16 = mul(x = x_273_cast_fp16, y = var_9079_cast_fp16)[name = string("op_9080_cast_fp16")]; + tensor input_385_cast_fp16 = mul(x = var_9080_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_385_cast_fp16")]; + string q_217_pad_type_0 = const()[name = string("q_217_pad_type_0"), val = string("valid")]; + tensor q_217_strides_0 = const()[name = string("q_217_strides_0"), val = tensor([1, 1])]; + tensor q_217_pad_0 = const()[name = string("q_217_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_217_dilations_0 = const()[name = string("q_217_dilations_0"), val = tensor([1, 1])]; + int32 q_217_groups_0 = const()[name = string("q_217_groups_0"), val = int32(1)]; + tensor q_217_cast_fp16 = conv(dilations = q_217_dilations_0, groups = q_217_groups_0, pad = q_217_pad_0, pad_type = q_217_pad_type_0, strides = q_217_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_385_cast_fp16)[name = string("q_217_cast_fp16")]; + string k_217_pad_type_0 = const()[name = string("k_217_pad_type_0"), val = string("valid")]; + tensor k_217_strides_0 = const()[name = string("k_217_strides_0"), val = tensor([1, 1])]; + tensor k_217_pad_0 = const()[name = string("k_217_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_217_dilations_0 = const()[name = string("k_217_dilations_0"), val = tensor([1, 1])]; + int32 k_217_groups_0 = const()[name = string("k_217_groups_0"), val = int32(1)]; + tensor k_217_cast_fp16 = conv(dilations = k_217_dilations_0, groups = k_217_groups_0, pad = k_217_pad_0, pad_type = k_217_pad_type_0, strides = k_217_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_385_cast_fp16)[name = string("k_217_cast_fp16")]; + string v_73_pad_type_0 = const()[name = string("v_73_pad_type_0"), val = string("valid")]; + tensor v_73_strides_0 = const()[name = string("v_73_strides_0"), val = tensor([1, 1])]; + tensor v_73_pad_0 = const()[name = string("v_73_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_73_dilations_0 = const()[name = string("v_73_dilations_0"), val = tensor([1, 1])]; + int32 v_73_groups_0 = const()[name = string("v_73_groups_0"), val = int32(1)]; + tensor v_73_cast_fp16 = conv(dilations = v_73_dilations_0, groups = v_73_groups_0, pad = v_73_pad_0, pad_type = v_73_pad_type_0, strides = v_73_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_385_cast_fp16)[name = string("v_73_cast_fp16")]; + tensor var_9114 = const()[name = string("op_9114"), val = tensor([16, 128, 1, 1])]; + tensor x_275_cast_fp16 = reshape(shape = var_9114, x = q_217_cast_fp16)[name = string("x_275_cast_fp16")]; + tensor var_9117_cast_fp16 = mul(x = x_275_cast_fp16, y = x_275_cast_fp16)[name = string("op_9117_cast_fp16")]; + tensor variance_303_axes_0 = const()[name = string("variance_303_axes_0"), val = tensor([1])]; + bool variance_303_keep_dims_0 = const()[name = string("variance_303_keep_dims_0"), val = bool(true)]; + tensor variance_303_cast_fp16 = reduce_mean(axes = variance_303_axes_0, keep_dims = variance_303_keep_dims_0, x = var_9117_cast_fp16)[name = string("variance_303_cast_fp16")]; + fp16 var_9120_to_fp16 = const()[name = string("op_9120_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9121_cast_fp16 = add(x = variance_303_cast_fp16, y = var_9120_to_fp16)[name = string("op_9121_cast_fp16")]; + fp32 var_9122_epsilon_0 = const()[name = string("op_9122_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9122_cast_fp16 = rsqrt(epsilon = var_9122_epsilon_0, x = var_9121_cast_fp16)[name = string("op_9122_cast_fp16")]; + tensor var_9123_cast_fp16 = mul(x = x_275_cast_fp16, y = var_9122_cast_fp16)[name = string("op_9123_cast_fp16")]; + tensor q_219_cast_fp16 = mul(x = var_9123_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_219_cast_fp16")]; + tensor var_9125 = const()[name = string("op_9125"), val = tensor([8, 128, 1, 1])]; + tensor x_277_cast_fp16 = reshape(shape = var_9125, x = k_217_cast_fp16)[name = string("x_277_cast_fp16")]; + tensor var_9128_cast_fp16 = mul(x = x_277_cast_fp16, y = x_277_cast_fp16)[name = string("op_9128_cast_fp16")]; + tensor variance_305_axes_0 = const()[name = string("variance_305_axes_0"), val = tensor([1])]; + bool variance_305_keep_dims_0 = const()[name = string("variance_305_keep_dims_0"), val = bool(true)]; + tensor variance_305_cast_fp16 = reduce_mean(axes = variance_305_axes_0, keep_dims = variance_305_keep_dims_0, x = var_9128_cast_fp16)[name = string("variance_305_cast_fp16")]; + fp16 var_9131_to_fp16 = const()[name = string("op_9131_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9132_cast_fp16 = add(x = variance_305_cast_fp16, y = var_9131_to_fp16)[name = string("op_9132_cast_fp16")]; + fp32 var_9133_epsilon_0 = const()[name = string("op_9133_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9133_cast_fp16 = rsqrt(epsilon = var_9133_epsilon_0, x = var_9132_cast_fp16)[name = string("op_9133_cast_fp16")]; + tensor var_9134_cast_fp16 = mul(x = x_277_cast_fp16, y = var_9133_cast_fp16)[name = string("op_9134_cast_fp16")]; + tensor k_219_cast_fp16 = mul(x = var_9134_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_219_cast_fp16")]; + tensor var_9136 = const()[name = string("op_9136"), val = tensor([1, 16, 128, 1])]; + tensor z_145_cast_fp16 = reshape(shape = var_9136, x = q_219_cast_fp16)[name = string("z_145_cast_fp16")]; + tensor var_9138 = const()[name = string("op_9138"), val = tensor([1, 8, 128, 1])]; + tensor z_147_cast_fp16 = reshape(shape = var_9138, x = k_219_cast_fp16)[name = string("z_147_cast_fp16")]; + tensor z1_145_begin_0 = const()[name = string("z1_145_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_145_end_0 = const()[name = string("z1_145_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_145_end_mask_0 = const()[name = string("z1_145_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_145_cast_fp16 = slice_by_index(begin = z1_145_begin_0, end = z1_145_end_0, end_mask = z1_145_end_mask_0, x = z_145_cast_fp16)[name = string("z1_145_cast_fp16")]; + tensor z2_145_begin_0 = const()[name = string("z2_145_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_145_end_0 = const()[name = string("z2_145_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_145_end_mask_0 = const()[name = string("z2_145_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_145_cast_fp16 = slice_by_index(begin = z2_145_begin_0, end = z2_145_end_0, end_mask = z2_145_end_mask_0, x = z_145_cast_fp16)[name = string("z2_145_cast_fp16")]; + tensor var_9146_cast_fp16 = mul(x = z_145_cast_fp16, y = cos_71_to_fp16)[name = string("op_9146_cast_fp16")]; + fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9147_cast_fp16 = mul(x = z2_145_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_9147_cast_fp16")]; + bool var_9149_interleave_0 = const()[name = string("op_9149_interleave_0"), val = bool(false)]; + tensor var_9149_cast_fp16 = concat(axis = var_9055, interleave = var_9149_interleave_0, values = (var_9147_cast_fp16, z1_145_cast_fp16))[name = string("op_9149_cast_fp16")]; + tensor var_9150_cast_fp16 = mul(x = var_9149_cast_fp16, y = sin_71_to_fp16)[name = string("op_9150_cast_fp16")]; + tensor q_221_cast_fp16 = add(x = var_9146_cast_fp16, y = var_9150_cast_fp16)[name = string("q_221_cast_fp16")]; + tensor z1_147_begin_0 = const()[name = string("z1_147_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_147_end_0 = const()[name = string("z1_147_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_147_end_mask_0 = const()[name = string("z1_147_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_147_cast_fp16 = slice_by_index(begin = z1_147_begin_0, end = z1_147_end_0, end_mask = z1_147_end_mask_0, x = z_147_cast_fp16)[name = string("z1_147_cast_fp16")]; + tensor z2_147_begin_0 = const()[name = string("z2_147_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_147_end_0 = const()[name = string("z2_147_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_147_end_mask_0 = const()[name = string("z2_147_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_147_cast_fp16 = slice_by_index(begin = z2_147_begin_0, end = z2_147_end_0, end_mask = z2_147_end_mask_0, x = z_147_cast_fp16)[name = string("z2_147_cast_fp16")]; + tensor var_9158_cast_fp16 = mul(x = z_147_cast_fp16, y = cos_71_to_fp16)[name = string("op_9158_cast_fp16")]; + fp16 const_81_promoted_to_fp16 = const()[name = string("const_81_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9159_cast_fp16 = mul(x = z2_147_cast_fp16, y = const_81_promoted_to_fp16)[name = string("op_9159_cast_fp16")]; + bool var_9161_interleave_0 = const()[name = string("op_9161_interleave_0"), val = bool(false)]; + tensor var_9161_cast_fp16 = concat(axis = var_9055, interleave = var_9161_interleave_0, values = (var_9159_cast_fp16, z1_147_cast_fp16))[name = string("op_9161_cast_fp16")]; + tensor var_9162_cast_fp16 = mul(x = var_9161_cast_fp16, y = sin_71_to_fp16)[name = string("op_9162_cast_fp16")]; + tensor k_221_cast_fp16 = add(x = var_9158_cast_fp16, y = var_9162_cast_fp16)[name = string("k_221_cast_fp16")]; + tensor var_9164 = const()[name = string("op_9164"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_73_cast_fp16 = reshape(shape = var_9164, x = k_221_cast_fp16)[name = string("cur_key_73_cast_fp16")]; + tensor var_9166_to_fp16 = const()[name = string("op_9166_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639168)))]; + tensor var_9167_cast_fp16 = mul(x = key_cache_73_cast_fp16, y = var_9166_to_fp16)[name = string("op_9167_cast_fp16")]; + tensor upd_73_to_fp16 = const()[name = string("upd_73_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639296)))]; + tensor var_9168_cast_fp16 = mul(x = cur_key_73_cast_fp16, y = upd_73_to_fp16)[name = string("op_9168_cast_fp16")]; + tensor key_73_cast_fp16 = add(x = var_9167_cast_fp16, y = var_9168_cast_fp16)[name = string("key_73_cast_fp16")]; + tensor var_9170_to_fp16 = const()[name = string("op_9170_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639168)))]; + tensor var_9171_cast_fp16 = mul(x = value_cache_73_cast_fp16, y = var_9170_to_fp16)[name = string("op_9171_cast_fp16")]; + tensor var_9172_cast_fp16 = mul(x = v_73_cast_fp16, y = upd_73_to_fp16)[name = string("op_9172_cast_fp16")]; + tensor value_73_cast_fp16 = add(x = var_9171_cast_fp16, y = var_9172_cast_fp16)[name = string("value_73_cast_fp16")]; + tensor var_9174 = const()[name = string("op_9174"), val = tensor([1, 8, 128, 16])]; + tensor kh_145_cast_fp16 = reshape(shape = var_9174, x = key_73_cast_fp16)[name = string("kh_145_cast_fp16")]; + tensor var_9176 = const()[name = string("op_9176"), val = tensor([1, 8, 128, 16])]; + tensor vh_145_cast_fp16 = reshape(shape = var_9176, x = value_73_cast_fp16)[name = string("vh_145_cast_fp16")]; + tensor transpose_144_perm_0 = const()[name = string("transpose_144_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_72_reps_0 = const()[name = string("tile_72_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_144_cast_fp16 = transpose(perm = transpose_144_perm_0, x = kh_145_cast_fp16)[name = string("transpose_263")]; + tensor tile_72_cast_fp16 = tile(reps = tile_72_reps_0, x = transpose_144_cast_fp16)[name = string("tile_72_cast_fp16")]; + tensor concat_182 = const()[name = string("concat_182"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_144_cast_fp16 = reshape(shape = concat_182, x = tile_72_cast_fp16)[name = string("reshape_144_cast_fp16")]; + tensor transpose_145_perm_0 = const()[name = string("transpose_145_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_183 = const()[name = string("concat_183"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_145_cast_fp16 = transpose(perm = transpose_145_perm_0, x = reshape_144_cast_fp16)[name = string("transpose_262")]; + tensor reshape_145_cast_fp16 = reshape(shape = concat_183, x = transpose_145_cast_fp16)[name = string("reshape_145_cast_fp16")]; + tensor transpose_146_perm_0 = const()[name = string("transpose_146_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_73_reps_0 = const()[name = string("tile_73_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_146_cast_fp16 = transpose(perm = transpose_146_perm_0, x = vh_145_cast_fp16)[name = string("transpose_261")]; + tensor tile_73_cast_fp16 = tile(reps = tile_73_reps_0, x = transpose_146_cast_fp16)[name = string("tile_73_cast_fp16")]; + tensor concat_184 = const()[name = string("concat_184"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_146_cast_fp16 = reshape(shape = concat_184, x = tile_73_cast_fp16)[name = string("reshape_146_cast_fp16")]; + tensor transpose_147_perm_0 = const()[name = string("transpose_147_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_185 = const()[name = string("concat_185"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_147_cast_fp16 = transpose(perm = transpose_147_perm_0, x = reshape_146_cast_fp16)[name = string("transpose_260")]; + tensor reshape_147_cast_fp16 = reshape(shape = concat_185, x = transpose_147_cast_fp16)[name = string("reshape_147_cast_fp16")]; + fp16 var_9180_to_fp16 = const()[name = string("op_9180_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_9181_cast_fp16 = mul(x = q_221_cast_fp16, y = var_9180_to_fp16)[name = string("op_9181_cast_fp16")]; + tensor transpose_461_perm_0 = const()[name = string("transpose_461_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_157_transpose_x_1 = const()[name = string("w_157_transpose_x_1"), val = bool(true)]; + bool w_157_transpose_y_1 = const()[name = string("w_157_transpose_y_1"), val = bool(false)]; + tensor transpose_461_cast_fp16 = transpose(perm = transpose_461_perm_0, x = reshape_145_cast_fp16)[name = string("transpose_259")]; + tensor w_157_cast_fp16 = matmul(transpose_x = w_157_transpose_x_1, transpose_y = w_157_transpose_y_1, x = var_9181_cast_fp16, y = transpose_461_cast_fp16)[name = string("w_157_cast_fp16")]; + tensor pad_73_to_fp16 = const()[name = string("pad_73_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639424)))]; + tensor var_9184_cast_fp16 = add(x = w_157_cast_fp16, y = pad_73_to_fp16)[name = string("op_9184_cast_fp16")]; + tensor w_159_cast_fp16 = softmax(axis = var_9059, x = var_9184_cast_fp16)[name = string("w_159_cast_fp16")]; + tensor transpose_462_perm_0 = const()[name = string("transpose_462_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_73_transpose_x_1 = const()[name = string("attn_73_transpose_x_1"), val = bool(false)]; + bool attn_73_transpose_y_1 = const()[name = string("attn_73_transpose_y_1"), val = bool(true)]; + tensor transpose_462_cast_fp16 = transpose(perm = transpose_462_perm_0, x = reshape_147_cast_fp16)[name = string("transpose_258")]; + tensor attn_73_cast_fp16 = matmul(transpose_x = attn_73_transpose_x_1, transpose_y = attn_73_transpose_y_1, x = transpose_462_cast_fp16, y = w_159_cast_fp16)[name = string("attn_73_cast_fp16")]; + tensor var_9188 = const()[name = string("op_9188"), val = tensor([1, 2048, 1, 1])]; + tensor input_387_cast_fp16 = reshape(shape = var_9188, x = attn_73_cast_fp16)[name = string("input_387_cast_fp16")]; + string attn_output_73_pad_type_0 = const()[name = string("attn_output_73_pad_type_0"), val = string("valid")]; + tensor attn_output_73_strides_0 = const()[name = string("attn_output_73_strides_0"), val = tensor([1, 1])]; + tensor attn_output_73_pad_0 = const()[name = string("attn_output_73_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_73_dilations_0 = const()[name = string("attn_output_73_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_73_groups_0 = const()[name = string("attn_output_73_groups_0"), val = int32(1)]; + tensor attn_output_73_cast_fp16 = conv(dilations = attn_output_73_dilations_0, groups = attn_output_73_groups_0, pad = attn_output_73_pad_0, pad_type = attn_output_73_pad_type_0, strides = attn_output_73_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_387_cast_fp16)[name = string("attn_output_73_cast_fp16")]; + tensor x_279_cast_fp16 = add(x = x_273_cast_fp16, y = attn_output_73_cast_fp16)[name = string("x_279_cast_fp16")]; + tensor var_9202_cast_fp16 = mul(x = x_279_cast_fp16, y = x_279_cast_fp16)[name = string("op_9202_cast_fp16")]; + tensor variance_307_axes_0 = const()[name = string("variance_307_axes_0"), val = tensor([1])]; + bool variance_307_keep_dims_0 = const()[name = string("variance_307_keep_dims_0"), val = bool(true)]; + tensor variance_307_cast_fp16 = reduce_mean(axes = variance_307_axes_0, keep_dims = variance_307_keep_dims_0, x = var_9202_cast_fp16)[name = string("variance_307_cast_fp16")]; + fp16 var_9205_to_fp16 = const()[name = string("op_9205_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9206_cast_fp16 = add(x = variance_307_cast_fp16, y = var_9205_to_fp16)[name = string("op_9206_cast_fp16")]; + fp32 var_9207_epsilon_0 = const()[name = string("op_9207_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9207_cast_fp16 = rsqrt(epsilon = var_9207_epsilon_0, x = var_9206_cast_fp16)[name = string("op_9207_cast_fp16")]; + tensor var_9208_cast_fp16 = mul(x = x_279_cast_fp16, y = var_9207_cast_fp16)[name = string("op_9208_cast_fp16")]; + tensor input_389_cast_fp16 = mul(x = var_9208_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_389_cast_fp16")]; + string input_391_pad_type_0 = const()[name = string("input_391_pad_type_0"), val = string("valid")]; + tensor input_391_strides_0 = const()[name = string("input_391_strides_0"), val = tensor([1, 1])]; + tensor input_391_pad_0 = const()[name = string("input_391_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_391_dilations_0 = const()[name = string("input_391_dilations_0"), val = tensor([1, 1])]; + int32 input_391_groups_0 = const()[name = string("input_391_groups_0"), val = int32(1)]; + tensor input_391_cast_fp16 = conv(dilations = input_391_dilations_0, groups = input_391_groups_0, pad = input_391_pad_0, pad_type = input_391_pad_type_0, strides = input_391_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_389_cast_fp16)[name = string("input_391_cast_fp16")]; + tensor var_9216_cast_fp16 = silu(x = input_391_cast_fp16)[name = string("op_9216_cast_fp16")]; + string var_9222_pad_type_0 = const()[name = string("op_9222_pad_type_0"), val = string("valid")]; + tensor var_9222_strides_0 = const()[name = string("op_9222_strides_0"), val = tensor([1, 1])]; + tensor var_9222_pad_0 = const()[name = string("op_9222_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9222_dilations_0 = const()[name = string("op_9222_dilations_0"), val = tensor([1, 1])]; + int32 var_9222_groups_0 = const()[name = string("op_9222_groups_0"), val = int32(1)]; + tensor var_9222_cast_fp16 = conv(dilations = var_9222_dilations_0, groups = var_9222_groups_0, pad = var_9222_pad_0, pad_type = var_9222_pad_type_0, strides = var_9222_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_389_cast_fp16)[name = string("op_9222_cast_fp16")]; + tensor input_393_cast_fp16 = mul(x = var_9216_cast_fp16, y = var_9222_cast_fp16)[name = string("input_393_cast_fp16")]; + string h_73_pad_type_0 = const()[name = string("h_73_pad_type_0"), val = string("valid")]; + tensor h_73_strides_0 = const()[name = string("h_73_strides_0"), val = tensor([1, 1])]; + tensor h_73_pad_0 = const()[name = string("h_73_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_73_dilations_0 = const()[name = string("h_73_dilations_0"), val = tensor([1, 1])]; + int32 h_73_groups_0 = const()[name = string("h_73_groups_0"), val = int32(1)]; + tensor h_73_cast_fp16 = conv(dilations = h_73_dilations_0, groups = h_73_groups_0, pad = h_73_pad_0, pad_type = h_73_pad_type_0, strides = h_73_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_393_cast_fp16)[name = string("h_73_cast_fp16")]; + tensor x_281_cast_fp16 = add(x = x_279_cast_fp16, y = h_73_cast_fp16)[name = string("x_281_cast_fp16")]; + tensor key_cache_75_begin_0 = const()[name = string("key_cache_75_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_75_end_0 = const()[name = string("key_cache_75_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_75_end_mask_0 = const()[name = string("key_cache_75_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_75_cast_fp16 = slice_by_index(begin = key_cache_75_begin_0, end = key_cache_75_end_0, end_mask = key_cache_75_end_mask_0, x = layer_key_caches_15_cast_fp16)[name = string("key_cache_75_cast_fp16")]; + tensor value_cache_75_begin_0 = const()[name = string("value_cache_75_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_75_end_0 = const()[name = string("value_cache_75_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_75_end_mask_0 = const()[name = string("value_cache_75_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_75_cast_fp16 = slice_by_index(begin = value_cache_75_begin_0, end = value_cache_75_end_0, end_mask = value_cache_75_end_mask_0, x = layer_value_caches_15_cast_fp16)[name = string("value_cache_75_cast_fp16")]; + int32 var_9275 = const()[name = string("op_9275"), val = int32(2)]; + int32 var_9279 = const()[name = string("op_9279"), val = int32(3)]; + tensor var_9294_cast_fp16 = mul(x = x_281_cast_fp16, y = x_281_cast_fp16)[name = string("op_9294_cast_fp16")]; + tensor variance_309_axes_0 = const()[name = string("variance_309_axes_0"), val = tensor([1])]; + bool variance_309_keep_dims_0 = const()[name = string("variance_309_keep_dims_0"), val = bool(true)]; + tensor variance_309_cast_fp16 = reduce_mean(axes = variance_309_axes_0, keep_dims = variance_309_keep_dims_0, x = var_9294_cast_fp16)[name = string("variance_309_cast_fp16")]; + fp16 var_9297_to_fp16 = const()[name = string("op_9297_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9298_cast_fp16 = add(x = variance_309_cast_fp16, y = var_9297_to_fp16)[name = string("op_9298_cast_fp16")]; + fp32 var_9299_epsilon_0 = const()[name = string("op_9299_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9299_cast_fp16 = rsqrt(epsilon = var_9299_epsilon_0, x = var_9298_cast_fp16)[name = string("op_9299_cast_fp16")]; + tensor var_9300_cast_fp16 = mul(x = x_281_cast_fp16, y = var_9299_cast_fp16)[name = string("op_9300_cast_fp16")]; + tensor input_395_cast_fp16 = mul(x = var_9300_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_395_cast_fp16")]; + string q_223_pad_type_0 = const()[name = string("q_223_pad_type_0"), val = string("valid")]; + tensor q_223_strides_0 = const()[name = string("q_223_strides_0"), val = tensor([1, 1])]; + tensor q_223_pad_0 = const()[name = string("q_223_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_223_dilations_0 = const()[name = string("q_223_dilations_0"), val = tensor([1, 1])]; + int32 q_223_groups_0 = const()[name = string("q_223_groups_0"), val = int32(1)]; + tensor q_223_cast_fp16 = conv(dilations = q_223_dilations_0, groups = q_223_groups_0, pad = q_223_pad_0, pad_type = q_223_pad_type_0, strides = q_223_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_395_cast_fp16)[name = string("q_223_cast_fp16")]; + string k_223_pad_type_0 = const()[name = string("k_223_pad_type_0"), val = string("valid")]; + tensor k_223_strides_0 = const()[name = string("k_223_strides_0"), val = tensor([1, 1])]; + tensor k_223_pad_0 = const()[name = string("k_223_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_223_dilations_0 = const()[name = string("k_223_dilations_0"), val = tensor([1, 1])]; + int32 k_223_groups_0 = const()[name = string("k_223_groups_0"), val = int32(1)]; + tensor k_223_cast_fp16 = conv(dilations = k_223_dilations_0, groups = k_223_groups_0, pad = k_223_pad_0, pad_type = k_223_pad_type_0, strides = k_223_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_395_cast_fp16)[name = string("k_223_cast_fp16")]; + string v_75_pad_type_0 = const()[name = string("v_75_pad_type_0"), val = string("valid")]; + tensor v_75_strides_0 = const()[name = string("v_75_strides_0"), val = tensor([1, 1])]; + tensor v_75_pad_0 = const()[name = string("v_75_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_75_dilations_0 = const()[name = string("v_75_dilations_0"), val = tensor([1, 1])]; + int32 v_75_groups_0 = const()[name = string("v_75_groups_0"), val = int32(1)]; + tensor v_75_cast_fp16 = conv(dilations = v_75_dilations_0, groups = v_75_groups_0, pad = v_75_pad_0, pad_type = v_75_pad_type_0, strides = v_75_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_395_cast_fp16)[name = string("v_75_cast_fp16")]; + tensor var_9334 = const()[name = string("op_9334"), val = tensor([16, 128, 1, 1])]; + tensor x_283_cast_fp16 = reshape(shape = var_9334, x = q_223_cast_fp16)[name = string("x_283_cast_fp16")]; + tensor var_9337_cast_fp16 = mul(x = x_283_cast_fp16, y = x_283_cast_fp16)[name = string("op_9337_cast_fp16")]; + tensor variance_311_axes_0 = const()[name = string("variance_311_axes_0"), val = tensor([1])]; + bool variance_311_keep_dims_0 = const()[name = string("variance_311_keep_dims_0"), val = bool(true)]; + tensor variance_311_cast_fp16 = reduce_mean(axes = variance_311_axes_0, keep_dims = variance_311_keep_dims_0, x = var_9337_cast_fp16)[name = string("variance_311_cast_fp16")]; + fp16 var_9340_to_fp16 = const()[name = string("op_9340_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9341_cast_fp16 = add(x = variance_311_cast_fp16, y = var_9340_to_fp16)[name = string("op_9341_cast_fp16")]; + fp32 var_9342_epsilon_0 = const()[name = string("op_9342_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9342_cast_fp16 = rsqrt(epsilon = var_9342_epsilon_0, x = var_9341_cast_fp16)[name = string("op_9342_cast_fp16")]; + tensor var_9343_cast_fp16 = mul(x = x_283_cast_fp16, y = var_9342_cast_fp16)[name = string("op_9343_cast_fp16")]; + tensor q_225_cast_fp16 = mul(x = var_9343_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_225_cast_fp16")]; + tensor var_9345 = const()[name = string("op_9345"), val = tensor([8, 128, 1, 1])]; + tensor x_285_cast_fp16 = reshape(shape = var_9345, x = k_223_cast_fp16)[name = string("x_285_cast_fp16")]; + tensor var_9348_cast_fp16 = mul(x = x_285_cast_fp16, y = x_285_cast_fp16)[name = string("op_9348_cast_fp16")]; + tensor variance_313_axes_0 = const()[name = string("variance_313_axes_0"), val = tensor([1])]; + bool variance_313_keep_dims_0 = const()[name = string("variance_313_keep_dims_0"), val = bool(true)]; + tensor variance_313_cast_fp16 = reduce_mean(axes = variance_313_axes_0, keep_dims = variance_313_keep_dims_0, x = var_9348_cast_fp16)[name = string("variance_313_cast_fp16")]; + fp16 var_9351_to_fp16 = const()[name = string("op_9351_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9352_cast_fp16 = add(x = variance_313_cast_fp16, y = var_9351_to_fp16)[name = string("op_9352_cast_fp16")]; + fp32 var_9353_epsilon_0 = const()[name = string("op_9353_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9353_cast_fp16 = rsqrt(epsilon = var_9353_epsilon_0, x = var_9352_cast_fp16)[name = string("op_9353_cast_fp16")]; + tensor var_9354_cast_fp16 = mul(x = x_285_cast_fp16, y = var_9353_cast_fp16)[name = string("op_9354_cast_fp16")]; + tensor k_225_cast_fp16 = mul(x = var_9354_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_225_cast_fp16")]; + tensor var_9356 = const()[name = string("op_9356"), val = tensor([1, 16, 128, 1])]; + tensor z_149_cast_fp16 = reshape(shape = var_9356, x = q_225_cast_fp16)[name = string("z_149_cast_fp16")]; + tensor var_9358 = const()[name = string("op_9358"), val = tensor([1, 8, 128, 1])]; + tensor z_151_cast_fp16 = reshape(shape = var_9358, x = k_225_cast_fp16)[name = string("z_151_cast_fp16")]; + tensor z1_149_begin_0 = const()[name = string("z1_149_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_149_end_0 = const()[name = string("z1_149_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_149_end_mask_0 = const()[name = string("z1_149_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_149_cast_fp16 = slice_by_index(begin = z1_149_begin_0, end = z1_149_end_0, end_mask = z1_149_end_mask_0, x = z_149_cast_fp16)[name = string("z1_149_cast_fp16")]; + tensor z2_149_begin_0 = const()[name = string("z2_149_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_149_end_0 = const()[name = string("z2_149_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_149_end_mask_0 = const()[name = string("z2_149_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_149_cast_fp16 = slice_by_index(begin = z2_149_begin_0, end = z2_149_end_0, end_mask = z2_149_end_mask_0, x = z_149_cast_fp16)[name = string("z2_149_cast_fp16")]; + tensor var_9366_cast_fp16 = mul(x = z_149_cast_fp16, y = cos_71_to_fp16)[name = string("op_9366_cast_fp16")]; + fp16 const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9367_cast_fp16 = mul(x = z2_149_cast_fp16, y = const_82_promoted_to_fp16)[name = string("op_9367_cast_fp16")]; + bool var_9369_interleave_0 = const()[name = string("op_9369_interleave_0"), val = bool(false)]; + tensor var_9369_cast_fp16 = concat(axis = var_9275, interleave = var_9369_interleave_0, values = (var_9367_cast_fp16, z1_149_cast_fp16))[name = string("op_9369_cast_fp16")]; + tensor var_9370_cast_fp16 = mul(x = var_9369_cast_fp16, y = sin_71_to_fp16)[name = string("op_9370_cast_fp16")]; + tensor q_227_cast_fp16 = add(x = var_9366_cast_fp16, y = var_9370_cast_fp16)[name = string("q_227_cast_fp16")]; + tensor z1_151_begin_0 = const()[name = string("z1_151_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_151_end_0 = const()[name = string("z1_151_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_151_end_mask_0 = const()[name = string("z1_151_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_151_cast_fp16 = slice_by_index(begin = z1_151_begin_0, end = z1_151_end_0, end_mask = z1_151_end_mask_0, x = z_151_cast_fp16)[name = string("z1_151_cast_fp16")]; + tensor z2_151_begin_0 = const()[name = string("z2_151_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_151_end_0 = const()[name = string("z2_151_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_151_end_mask_0 = const()[name = string("z2_151_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_151_cast_fp16 = slice_by_index(begin = z2_151_begin_0, end = z2_151_end_0, end_mask = z2_151_end_mask_0, x = z_151_cast_fp16)[name = string("z2_151_cast_fp16")]; + tensor var_9378_cast_fp16 = mul(x = z_151_cast_fp16, y = cos_71_to_fp16)[name = string("op_9378_cast_fp16")]; + fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9379_cast_fp16 = mul(x = z2_151_cast_fp16, y = const_83_promoted_to_fp16)[name = string("op_9379_cast_fp16")]; + bool var_9381_interleave_0 = const()[name = string("op_9381_interleave_0"), val = bool(false)]; + tensor var_9381_cast_fp16 = concat(axis = var_9275, interleave = var_9381_interleave_0, values = (var_9379_cast_fp16, z1_151_cast_fp16))[name = string("op_9381_cast_fp16")]; + tensor var_9382_cast_fp16 = mul(x = var_9381_cast_fp16, y = sin_71_to_fp16)[name = string("op_9382_cast_fp16")]; + tensor k_227_cast_fp16 = add(x = var_9378_cast_fp16, y = var_9382_cast_fp16)[name = string("k_227_cast_fp16")]; + tensor var_9384 = const()[name = string("op_9384"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_75_cast_fp16 = reshape(shape = var_9384, x = k_227_cast_fp16)[name = string("cur_key_75_cast_fp16")]; + tensor var_9386_to_fp16 = const()[name = string("op_9386_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639168)))]; + tensor var_9387_cast_fp16 = mul(x = key_cache_75_cast_fp16, y = var_9386_to_fp16)[name = string("op_9387_cast_fp16")]; + tensor upd_75_to_fp16 = const()[name = string("upd_75_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639296)))]; + tensor var_9388_cast_fp16 = mul(x = cur_key_75_cast_fp16, y = upd_75_to_fp16)[name = string("op_9388_cast_fp16")]; + tensor key_75_cast_fp16 = add(x = var_9387_cast_fp16, y = var_9388_cast_fp16)[name = string("key_75_cast_fp16")]; + tensor var_9390_to_fp16 = const()[name = string("op_9390_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639168)))]; + tensor var_9391_cast_fp16 = mul(x = value_cache_75_cast_fp16, y = var_9390_to_fp16)[name = string("op_9391_cast_fp16")]; + tensor var_9392_cast_fp16 = mul(x = v_75_cast_fp16, y = upd_75_to_fp16)[name = string("op_9392_cast_fp16")]; + tensor value_75_cast_fp16 = add(x = var_9391_cast_fp16, y = var_9392_cast_fp16)[name = string("value_75_cast_fp16")]; + tensor var_9394 = const()[name = string("op_9394"), val = tensor([1, 8, 128, 16])]; + tensor kh_149_cast_fp16 = reshape(shape = var_9394, x = key_75_cast_fp16)[name = string("kh_149_cast_fp16")]; + tensor var_9396 = const()[name = string("op_9396"), val = tensor([1, 8, 128, 16])]; + tensor vh_149_cast_fp16 = reshape(shape = var_9396, x = value_75_cast_fp16)[name = string("vh_149_cast_fp16")]; + tensor transpose_148_perm_0 = const()[name = string("transpose_148_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_74_reps_0 = const()[name = string("tile_74_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_148_cast_fp16 = transpose(perm = transpose_148_perm_0, x = kh_149_cast_fp16)[name = string("transpose_257")]; + tensor tile_74_cast_fp16 = tile(reps = tile_74_reps_0, x = transpose_148_cast_fp16)[name = string("tile_74_cast_fp16")]; + tensor concat_186 = const()[name = string("concat_186"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_148_cast_fp16 = reshape(shape = concat_186, x = tile_74_cast_fp16)[name = string("reshape_148_cast_fp16")]; + tensor transpose_149_perm_0 = const()[name = string("transpose_149_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_187 = const()[name = string("concat_187"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_149_cast_fp16 = transpose(perm = transpose_149_perm_0, x = reshape_148_cast_fp16)[name = string("transpose_256")]; + tensor reshape_149_cast_fp16 = reshape(shape = concat_187, x = transpose_149_cast_fp16)[name = string("reshape_149_cast_fp16")]; + tensor transpose_150_perm_0 = const()[name = string("transpose_150_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_75_reps_0 = const()[name = string("tile_75_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_150_cast_fp16 = transpose(perm = transpose_150_perm_0, x = vh_149_cast_fp16)[name = string("transpose_255")]; + tensor tile_75_cast_fp16 = tile(reps = tile_75_reps_0, x = transpose_150_cast_fp16)[name = string("tile_75_cast_fp16")]; + tensor concat_188 = const()[name = string("concat_188"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_150_cast_fp16 = reshape(shape = concat_188, x = tile_75_cast_fp16)[name = string("reshape_150_cast_fp16")]; + tensor transpose_151_perm_0 = const()[name = string("transpose_151_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_189 = const()[name = string("concat_189"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_151_cast_fp16 = transpose(perm = transpose_151_perm_0, x = reshape_150_cast_fp16)[name = string("transpose_254")]; + tensor reshape_151_cast_fp16 = reshape(shape = concat_189, x = transpose_151_cast_fp16)[name = string("reshape_151_cast_fp16")]; + fp16 var_9400_to_fp16 = const()[name = string("op_9400_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_9401_cast_fp16 = mul(x = q_227_cast_fp16, y = var_9400_to_fp16)[name = string("op_9401_cast_fp16")]; + tensor transpose_465_perm_0 = const()[name = string("transpose_465_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_161_transpose_x_1 = const()[name = string("w_161_transpose_x_1"), val = bool(true)]; + bool w_161_transpose_y_1 = const()[name = string("w_161_transpose_y_1"), val = bool(false)]; + tensor transpose_465_cast_fp16 = transpose(perm = transpose_465_perm_0, x = reshape_149_cast_fp16)[name = string("transpose_253")]; + tensor w_161_cast_fp16 = matmul(transpose_x = w_161_transpose_x_1, transpose_y = w_161_transpose_y_1, x = var_9401_cast_fp16, y = transpose_465_cast_fp16)[name = string("w_161_cast_fp16")]; + tensor pad_75_to_fp16 = const()[name = string("pad_75_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639424)))]; + tensor var_9404_cast_fp16 = add(x = w_161_cast_fp16, y = pad_75_to_fp16)[name = string("op_9404_cast_fp16")]; + tensor w_163_cast_fp16 = softmax(axis = var_9279, x = var_9404_cast_fp16)[name = string("w_163_cast_fp16")]; + tensor transpose_466_perm_0 = const()[name = string("transpose_466_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_75_transpose_x_1 = const()[name = string("attn_75_transpose_x_1"), val = bool(false)]; + bool attn_75_transpose_y_1 = const()[name = string("attn_75_transpose_y_1"), val = bool(true)]; + tensor transpose_466_cast_fp16 = transpose(perm = transpose_466_perm_0, x = reshape_151_cast_fp16)[name = string("transpose_252")]; + tensor attn_75_cast_fp16 = matmul(transpose_x = attn_75_transpose_x_1, transpose_y = attn_75_transpose_y_1, x = transpose_466_cast_fp16, y = w_163_cast_fp16)[name = string("attn_75_cast_fp16")]; + tensor var_9408 = const()[name = string("op_9408"), val = tensor([1, 2048, 1, 1])]; + tensor input_397_cast_fp16 = reshape(shape = var_9408, x = attn_75_cast_fp16)[name = string("input_397_cast_fp16")]; + string attn_output_75_pad_type_0 = const()[name = string("attn_output_75_pad_type_0"), val = string("valid")]; + tensor attn_output_75_strides_0 = const()[name = string("attn_output_75_strides_0"), val = tensor([1, 1])]; + tensor attn_output_75_pad_0 = const()[name = string("attn_output_75_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_75_dilations_0 = const()[name = string("attn_output_75_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_75_groups_0 = const()[name = string("attn_output_75_groups_0"), val = int32(1)]; + tensor attn_output_75_cast_fp16 = conv(dilations = attn_output_75_dilations_0, groups = attn_output_75_groups_0, pad = attn_output_75_pad_0, pad_type = attn_output_75_pad_type_0, strides = attn_output_75_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_397_cast_fp16)[name = string("attn_output_75_cast_fp16")]; + tensor x_287_cast_fp16 = add(x = x_281_cast_fp16, y = attn_output_75_cast_fp16)[name = string("x_287_cast_fp16")]; + tensor var_9422_cast_fp16 = mul(x = x_287_cast_fp16, y = x_287_cast_fp16)[name = string("op_9422_cast_fp16")]; + tensor variance_315_axes_0 = const()[name = string("variance_315_axes_0"), val = tensor([1])]; + bool variance_315_keep_dims_0 = const()[name = string("variance_315_keep_dims_0"), val = bool(true)]; + tensor variance_315_cast_fp16 = reduce_mean(axes = variance_315_axes_0, keep_dims = variance_315_keep_dims_0, x = var_9422_cast_fp16)[name = string("variance_315_cast_fp16")]; + fp16 var_9425_to_fp16 = const()[name = string("op_9425_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9426_cast_fp16 = add(x = variance_315_cast_fp16, y = var_9425_to_fp16)[name = string("op_9426_cast_fp16")]; + fp32 var_9427_epsilon_0 = const()[name = string("op_9427_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9427_cast_fp16 = rsqrt(epsilon = var_9427_epsilon_0, x = var_9426_cast_fp16)[name = string("op_9427_cast_fp16")]; + tensor var_9428_cast_fp16 = mul(x = x_287_cast_fp16, y = var_9427_cast_fp16)[name = string("op_9428_cast_fp16")]; + tensor input_399_cast_fp16 = mul(x = var_9428_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_399_cast_fp16")]; + string input_401_pad_type_0 = const()[name = string("input_401_pad_type_0"), val = string("valid")]; + tensor input_401_strides_0 = const()[name = string("input_401_strides_0"), val = tensor([1, 1])]; + tensor input_401_pad_0 = const()[name = string("input_401_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_401_dilations_0 = const()[name = string("input_401_dilations_0"), val = tensor([1, 1])]; + int32 input_401_groups_0 = const()[name = string("input_401_groups_0"), val = int32(1)]; + tensor input_401_cast_fp16 = conv(dilations = input_401_dilations_0, groups = input_401_groups_0, pad = input_401_pad_0, pad_type = input_401_pad_type_0, strides = input_401_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_399_cast_fp16)[name = string("input_401_cast_fp16")]; + tensor var_9436_cast_fp16 = silu(x = input_401_cast_fp16)[name = string("op_9436_cast_fp16")]; + string var_9442_pad_type_0 = const()[name = string("op_9442_pad_type_0"), val = string("valid")]; + tensor var_9442_strides_0 = const()[name = string("op_9442_strides_0"), val = tensor([1, 1])]; + tensor var_9442_pad_0 = const()[name = string("op_9442_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9442_dilations_0 = const()[name = string("op_9442_dilations_0"), val = tensor([1, 1])]; + int32 var_9442_groups_0 = const()[name = string("op_9442_groups_0"), val = int32(1)]; + tensor var_9442_cast_fp16 = conv(dilations = var_9442_dilations_0, groups = var_9442_groups_0, pad = var_9442_pad_0, pad_type = var_9442_pad_type_0, strides = var_9442_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_399_cast_fp16)[name = string("op_9442_cast_fp16")]; + tensor input_403_cast_fp16 = mul(x = var_9436_cast_fp16, y = var_9442_cast_fp16)[name = string("input_403_cast_fp16")]; + string h_75_pad_type_0 = const()[name = string("h_75_pad_type_0"), val = string("valid")]; + tensor h_75_strides_0 = const()[name = string("h_75_strides_0"), val = tensor([1, 1])]; + tensor h_75_pad_0 = const()[name = string("h_75_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_75_dilations_0 = const()[name = string("h_75_dilations_0"), val = tensor([1, 1])]; + int32 h_75_groups_0 = const()[name = string("h_75_groups_0"), val = int32(1)]; + tensor h_75_cast_fp16 = conv(dilations = h_75_dilations_0, groups = h_75_groups_0, pad = h_75_pad_0, pad_type = h_75_pad_type_0, strides = h_75_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_403_cast_fp16)[name = string("h_75_cast_fp16")]; + tensor x_289_cast_fp16 = add(x = x_287_cast_fp16, y = h_75_cast_fp16)[name = string("x_289_cast_fp16")]; + tensor key_cache_77_begin_0 = const()[name = string("key_cache_77_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_77_end_0 = const()[name = string("key_cache_77_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_77_end_mask_0 = const()[name = string("key_cache_77_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_77_cast_fp16 = slice_by_index(begin = key_cache_77_begin_0, end = key_cache_77_end_0, end_mask = key_cache_77_end_mask_0, x = layer_key_caches_15_cast_fp16)[name = string("key_cache_77_cast_fp16")]; + tensor value_cache_77_begin_0 = const()[name = string("value_cache_77_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_77_end_0 = const()[name = string("value_cache_77_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_77_end_mask_0 = const()[name = string("value_cache_77_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_77_cast_fp16 = slice_by_index(begin = value_cache_77_begin_0, end = value_cache_77_end_0, end_mask = value_cache_77_end_mask_0, x = layer_value_caches_15_cast_fp16)[name = string("value_cache_77_cast_fp16")]; + int32 var_9495 = const()[name = string("op_9495"), val = int32(2)]; + int32 var_9499 = const()[name = string("op_9499"), val = int32(3)]; + tensor var_9514_cast_fp16 = mul(x = x_289_cast_fp16, y = x_289_cast_fp16)[name = string("op_9514_cast_fp16")]; + tensor variance_317_axes_0 = const()[name = string("variance_317_axes_0"), val = tensor([1])]; + bool variance_317_keep_dims_0 = const()[name = string("variance_317_keep_dims_0"), val = bool(true)]; + tensor variance_317_cast_fp16 = reduce_mean(axes = variance_317_axes_0, keep_dims = variance_317_keep_dims_0, x = var_9514_cast_fp16)[name = string("variance_317_cast_fp16")]; + fp16 var_9517_to_fp16 = const()[name = string("op_9517_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9518_cast_fp16 = add(x = variance_317_cast_fp16, y = var_9517_to_fp16)[name = string("op_9518_cast_fp16")]; + fp32 var_9519_epsilon_0 = const()[name = string("op_9519_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9519_cast_fp16 = rsqrt(epsilon = var_9519_epsilon_0, x = var_9518_cast_fp16)[name = string("op_9519_cast_fp16")]; + tensor var_9520_cast_fp16 = mul(x = x_289_cast_fp16, y = var_9519_cast_fp16)[name = string("op_9520_cast_fp16")]; + tensor input_405_cast_fp16 = mul(x = var_9520_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_405_cast_fp16")]; + string q_229_pad_type_0 = const()[name = string("q_229_pad_type_0"), val = string("valid")]; + tensor q_229_strides_0 = const()[name = string("q_229_strides_0"), val = tensor([1, 1])]; + tensor q_229_pad_0 = const()[name = string("q_229_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_229_dilations_0 = const()[name = string("q_229_dilations_0"), val = tensor([1, 1])]; + int32 q_229_groups_0 = const()[name = string("q_229_groups_0"), val = int32(1)]; + tensor q_229_cast_fp16 = conv(dilations = q_229_dilations_0, groups = q_229_groups_0, pad = q_229_pad_0, pad_type = q_229_pad_type_0, strides = q_229_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_405_cast_fp16)[name = string("q_229_cast_fp16")]; + string k_229_pad_type_0 = const()[name = string("k_229_pad_type_0"), val = string("valid")]; + tensor k_229_strides_0 = const()[name = string("k_229_strides_0"), val = tensor([1, 1])]; + tensor k_229_pad_0 = const()[name = string("k_229_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_229_dilations_0 = const()[name = string("k_229_dilations_0"), val = tensor([1, 1])]; + int32 k_229_groups_0 = const()[name = string("k_229_groups_0"), val = int32(1)]; + tensor k_229_cast_fp16 = conv(dilations = k_229_dilations_0, groups = k_229_groups_0, pad = k_229_pad_0, pad_type = k_229_pad_type_0, strides = k_229_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_405_cast_fp16)[name = string("k_229_cast_fp16")]; + string v_77_pad_type_0 = const()[name = string("v_77_pad_type_0"), val = string("valid")]; + tensor v_77_strides_0 = const()[name = string("v_77_strides_0"), val = tensor([1, 1])]; + tensor v_77_pad_0 = const()[name = string("v_77_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_77_dilations_0 = const()[name = string("v_77_dilations_0"), val = tensor([1, 1])]; + int32 v_77_groups_0 = const()[name = string("v_77_groups_0"), val = int32(1)]; + tensor v_77_cast_fp16 = conv(dilations = v_77_dilations_0, groups = v_77_groups_0, pad = v_77_pad_0, pad_type = v_77_pad_type_0, strides = v_77_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_405_cast_fp16)[name = string("v_77_cast_fp16")]; + tensor var_9554 = const()[name = string("op_9554"), val = tensor([16, 128, 1, 1])]; + tensor x_291_cast_fp16 = reshape(shape = var_9554, x = q_229_cast_fp16)[name = string("x_291_cast_fp16")]; + tensor var_9557_cast_fp16 = mul(x = x_291_cast_fp16, y = x_291_cast_fp16)[name = string("op_9557_cast_fp16")]; + tensor variance_319_axes_0 = const()[name = string("variance_319_axes_0"), val = tensor([1])]; + bool variance_319_keep_dims_0 = const()[name = string("variance_319_keep_dims_0"), val = bool(true)]; + tensor variance_319_cast_fp16 = reduce_mean(axes = variance_319_axes_0, keep_dims = variance_319_keep_dims_0, x = var_9557_cast_fp16)[name = string("variance_319_cast_fp16")]; + fp16 var_9560_to_fp16 = const()[name = string("op_9560_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9561_cast_fp16 = add(x = variance_319_cast_fp16, y = var_9560_to_fp16)[name = string("op_9561_cast_fp16")]; + fp32 var_9562_epsilon_0 = const()[name = string("op_9562_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9562_cast_fp16 = rsqrt(epsilon = var_9562_epsilon_0, x = var_9561_cast_fp16)[name = string("op_9562_cast_fp16")]; + tensor var_9563_cast_fp16 = mul(x = x_291_cast_fp16, y = var_9562_cast_fp16)[name = string("op_9563_cast_fp16")]; + tensor q_231_cast_fp16 = mul(x = var_9563_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_231_cast_fp16")]; + tensor var_9565 = const()[name = string("op_9565"), val = tensor([8, 128, 1, 1])]; + tensor x_293_cast_fp16 = reshape(shape = var_9565, x = k_229_cast_fp16)[name = string("x_293_cast_fp16")]; + tensor var_9568_cast_fp16 = mul(x = x_293_cast_fp16, y = x_293_cast_fp16)[name = string("op_9568_cast_fp16")]; + tensor variance_321_axes_0 = const()[name = string("variance_321_axes_0"), val = tensor([1])]; + bool variance_321_keep_dims_0 = const()[name = string("variance_321_keep_dims_0"), val = bool(true)]; + tensor variance_321_cast_fp16 = reduce_mean(axes = variance_321_axes_0, keep_dims = variance_321_keep_dims_0, x = var_9568_cast_fp16)[name = string("variance_321_cast_fp16")]; + fp16 var_9571_to_fp16 = const()[name = string("op_9571_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9572_cast_fp16 = add(x = variance_321_cast_fp16, y = var_9571_to_fp16)[name = string("op_9572_cast_fp16")]; + fp32 var_9573_epsilon_0 = const()[name = string("op_9573_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9573_cast_fp16 = rsqrt(epsilon = var_9573_epsilon_0, x = var_9572_cast_fp16)[name = string("op_9573_cast_fp16")]; + tensor var_9574_cast_fp16 = mul(x = x_293_cast_fp16, y = var_9573_cast_fp16)[name = string("op_9574_cast_fp16")]; + tensor k_231_cast_fp16 = mul(x = var_9574_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_231_cast_fp16")]; + tensor var_9576 = const()[name = string("op_9576"), val = tensor([1, 16, 128, 1])]; + tensor z_153_cast_fp16 = reshape(shape = var_9576, x = q_231_cast_fp16)[name = string("z_153_cast_fp16")]; + tensor var_9578 = const()[name = string("op_9578"), val = tensor([1, 8, 128, 1])]; + tensor z_155_cast_fp16 = reshape(shape = var_9578, x = k_231_cast_fp16)[name = string("z_155_cast_fp16")]; + tensor z1_153_begin_0 = const()[name = string("z1_153_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_153_end_0 = const()[name = string("z1_153_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_153_end_mask_0 = const()[name = string("z1_153_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_153_cast_fp16 = slice_by_index(begin = z1_153_begin_0, end = z1_153_end_0, end_mask = z1_153_end_mask_0, x = z_153_cast_fp16)[name = string("z1_153_cast_fp16")]; + tensor z2_153_begin_0 = const()[name = string("z2_153_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_153_end_0 = const()[name = string("z2_153_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_153_end_mask_0 = const()[name = string("z2_153_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_153_cast_fp16 = slice_by_index(begin = z2_153_begin_0, end = z2_153_end_0, end_mask = z2_153_end_mask_0, x = z_153_cast_fp16)[name = string("z2_153_cast_fp16")]; + tensor var_9586_cast_fp16 = mul(x = z_153_cast_fp16, y = cos_71_to_fp16)[name = string("op_9586_cast_fp16")]; + fp16 const_84_promoted_to_fp16 = const()[name = string("const_84_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9587_cast_fp16 = mul(x = z2_153_cast_fp16, y = const_84_promoted_to_fp16)[name = string("op_9587_cast_fp16")]; + bool var_9589_interleave_0 = const()[name = string("op_9589_interleave_0"), val = bool(false)]; + tensor var_9589_cast_fp16 = concat(axis = var_9495, interleave = var_9589_interleave_0, values = (var_9587_cast_fp16, z1_153_cast_fp16))[name = string("op_9589_cast_fp16")]; + tensor var_9590_cast_fp16 = mul(x = var_9589_cast_fp16, y = sin_71_to_fp16)[name = string("op_9590_cast_fp16")]; + tensor q_233_cast_fp16 = add(x = var_9586_cast_fp16, y = var_9590_cast_fp16)[name = string("q_233_cast_fp16")]; + tensor z1_155_begin_0 = const()[name = string("z1_155_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_155_end_0 = const()[name = string("z1_155_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_155_end_mask_0 = const()[name = string("z1_155_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_155_cast_fp16 = slice_by_index(begin = z1_155_begin_0, end = z1_155_end_0, end_mask = z1_155_end_mask_0, x = z_155_cast_fp16)[name = string("z1_155_cast_fp16")]; + tensor z2_155_begin_0 = const()[name = string("z2_155_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_155_end_0 = const()[name = string("z2_155_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_155_end_mask_0 = const()[name = string("z2_155_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_155_cast_fp16 = slice_by_index(begin = z2_155_begin_0, end = z2_155_end_0, end_mask = z2_155_end_mask_0, x = z_155_cast_fp16)[name = string("z2_155_cast_fp16")]; + tensor var_9598_cast_fp16 = mul(x = z_155_cast_fp16, y = cos_71_to_fp16)[name = string("op_9598_cast_fp16")]; + fp16 const_85_promoted_to_fp16 = const()[name = string("const_85_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9599_cast_fp16 = mul(x = z2_155_cast_fp16, y = const_85_promoted_to_fp16)[name = string("op_9599_cast_fp16")]; + bool var_9601_interleave_0 = const()[name = string("op_9601_interleave_0"), val = bool(false)]; + tensor var_9601_cast_fp16 = concat(axis = var_9495, interleave = var_9601_interleave_0, values = (var_9599_cast_fp16, z1_155_cast_fp16))[name = string("op_9601_cast_fp16")]; + tensor var_9602_cast_fp16 = mul(x = var_9601_cast_fp16, y = sin_71_to_fp16)[name = string("op_9602_cast_fp16")]; + tensor k_233_cast_fp16 = add(x = var_9598_cast_fp16, y = var_9602_cast_fp16)[name = string("k_233_cast_fp16")]; + tensor var_9604 = const()[name = string("op_9604"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_77_cast_fp16 = reshape(shape = var_9604, x = k_233_cast_fp16)[name = string("cur_key_77_cast_fp16")]; + tensor var_9606_to_fp16 = const()[name = string("op_9606_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639168)))]; + tensor var_9607_cast_fp16 = mul(x = key_cache_77_cast_fp16, y = var_9606_to_fp16)[name = string("op_9607_cast_fp16")]; + tensor upd_77_to_fp16 = const()[name = string("upd_77_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639296)))]; + tensor var_9608_cast_fp16 = mul(x = cur_key_77_cast_fp16, y = upd_77_to_fp16)[name = string("op_9608_cast_fp16")]; + tensor key_77_cast_fp16 = add(x = var_9607_cast_fp16, y = var_9608_cast_fp16)[name = string("key_77_cast_fp16")]; + tensor var_9610_to_fp16 = const()[name = string("op_9610_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639168)))]; + tensor var_9611_cast_fp16 = mul(x = value_cache_77_cast_fp16, y = var_9610_to_fp16)[name = string("op_9611_cast_fp16")]; + tensor var_9612_cast_fp16 = mul(x = v_77_cast_fp16, y = upd_77_to_fp16)[name = string("op_9612_cast_fp16")]; + tensor value_77_cast_fp16 = add(x = var_9611_cast_fp16, y = var_9612_cast_fp16)[name = string("value_77_cast_fp16")]; + tensor var_9614 = const()[name = string("op_9614"), val = tensor([1, 8, 128, 16])]; + tensor kh_153_cast_fp16 = reshape(shape = var_9614, x = key_77_cast_fp16)[name = string("kh_153_cast_fp16")]; + tensor var_9616 = const()[name = string("op_9616"), val = tensor([1, 8, 128, 16])]; + tensor vh_153_cast_fp16 = reshape(shape = var_9616, x = value_77_cast_fp16)[name = string("vh_153_cast_fp16")]; + tensor transpose_152_perm_0 = const()[name = string("transpose_152_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_76_reps_0 = const()[name = string("tile_76_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_152_cast_fp16 = transpose(perm = transpose_152_perm_0, x = kh_153_cast_fp16)[name = string("transpose_251")]; + tensor tile_76_cast_fp16 = tile(reps = tile_76_reps_0, x = transpose_152_cast_fp16)[name = string("tile_76_cast_fp16")]; + tensor concat_190 = const()[name = string("concat_190"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_152_cast_fp16 = reshape(shape = concat_190, x = tile_76_cast_fp16)[name = string("reshape_152_cast_fp16")]; + tensor transpose_153_perm_0 = const()[name = string("transpose_153_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_191 = const()[name = string("concat_191"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_153_cast_fp16 = transpose(perm = transpose_153_perm_0, x = reshape_152_cast_fp16)[name = string("transpose_250")]; + tensor reshape_153_cast_fp16 = reshape(shape = concat_191, x = transpose_153_cast_fp16)[name = string("reshape_153_cast_fp16")]; + tensor transpose_154_perm_0 = const()[name = string("transpose_154_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_77_reps_0 = const()[name = string("tile_77_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_154_cast_fp16 = transpose(perm = transpose_154_perm_0, x = vh_153_cast_fp16)[name = string("transpose_249")]; + tensor tile_77_cast_fp16 = tile(reps = tile_77_reps_0, x = transpose_154_cast_fp16)[name = string("tile_77_cast_fp16")]; + tensor concat_192 = const()[name = string("concat_192"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_154_cast_fp16 = reshape(shape = concat_192, x = tile_77_cast_fp16)[name = string("reshape_154_cast_fp16")]; + tensor transpose_155_perm_0 = const()[name = string("transpose_155_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_193 = const()[name = string("concat_193"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_155_cast_fp16 = transpose(perm = transpose_155_perm_0, x = reshape_154_cast_fp16)[name = string("transpose_248")]; + tensor reshape_155_cast_fp16 = reshape(shape = concat_193, x = transpose_155_cast_fp16)[name = string("reshape_155_cast_fp16")]; + fp16 var_9620_to_fp16 = const()[name = string("op_9620_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_9621_cast_fp16 = mul(x = q_233_cast_fp16, y = var_9620_to_fp16)[name = string("op_9621_cast_fp16")]; + tensor transpose_469_perm_0 = const()[name = string("transpose_469_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_165_transpose_x_1 = const()[name = string("w_165_transpose_x_1"), val = bool(true)]; + bool w_165_transpose_y_1 = const()[name = string("w_165_transpose_y_1"), val = bool(false)]; + tensor transpose_469_cast_fp16 = transpose(perm = transpose_469_perm_0, x = reshape_153_cast_fp16)[name = string("transpose_247")]; + tensor w_165_cast_fp16 = matmul(transpose_x = w_165_transpose_x_1, transpose_y = w_165_transpose_y_1, x = var_9621_cast_fp16, y = transpose_469_cast_fp16)[name = string("w_165_cast_fp16")]; + tensor pad_77_to_fp16 = const()[name = string("pad_77_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639424)))]; + tensor var_9624_cast_fp16 = add(x = w_165_cast_fp16, y = pad_77_to_fp16)[name = string("op_9624_cast_fp16")]; + tensor w_167_cast_fp16 = softmax(axis = var_9499, x = var_9624_cast_fp16)[name = string("w_167_cast_fp16")]; + tensor transpose_470_perm_0 = const()[name = string("transpose_470_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_77_transpose_x_1 = const()[name = string("attn_77_transpose_x_1"), val = bool(false)]; + bool attn_77_transpose_y_1 = const()[name = string("attn_77_transpose_y_1"), val = bool(true)]; + tensor transpose_470_cast_fp16 = transpose(perm = transpose_470_perm_0, x = reshape_155_cast_fp16)[name = string("transpose_246")]; + tensor attn_77_cast_fp16 = matmul(transpose_x = attn_77_transpose_x_1, transpose_y = attn_77_transpose_y_1, x = transpose_470_cast_fp16, y = w_167_cast_fp16)[name = string("attn_77_cast_fp16")]; + tensor var_9628 = const()[name = string("op_9628"), val = tensor([1, 2048, 1, 1])]; + tensor input_407_cast_fp16 = reshape(shape = var_9628, x = attn_77_cast_fp16)[name = string("input_407_cast_fp16")]; + string attn_output_77_pad_type_0 = const()[name = string("attn_output_77_pad_type_0"), val = string("valid")]; + tensor attn_output_77_strides_0 = const()[name = string("attn_output_77_strides_0"), val = tensor([1, 1])]; + tensor attn_output_77_pad_0 = const()[name = string("attn_output_77_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_77_dilations_0 = const()[name = string("attn_output_77_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_77_groups_0 = const()[name = string("attn_output_77_groups_0"), val = int32(1)]; + tensor attn_output_77_cast_fp16 = conv(dilations = attn_output_77_dilations_0, groups = attn_output_77_groups_0, pad = attn_output_77_pad_0, pad_type = attn_output_77_pad_type_0, strides = attn_output_77_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_407_cast_fp16)[name = string("attn_output_77_cast_fp16")]; + tensor x_295_cast_fp16 = add(x = x_289_cast_fp16, y = attn_output_77_cast_fp16)[name = string("x_295_cast_fp16")]; + tensor var_9642_cast_fp16 = mul(x = x_295_cast_fp16, y = x_295_cast_fp16)[name = string("op_9642_cast_fp16")]; + tensor variance_323_axes_0 = const()[name = string("variance_323_axes_0"), val = tensor([1])]; + bool variance_323_keep_dims_0 = const()[name = string("variance_323_keep_dims_0"), val = bool(true)]; + tensor variance_323_cast_fp16 = reduce_mean(axes = variance_323_axes_0, keep_dims = variance_323_keep_dims_0, x = var_9642_cast_fp16)[name = string("variance_323_cast_fp16")]; + fp16 var_9645_to_fp16 = const()[name = string("op_9645_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9646_cast_fp16 = add(x = variance_323_cast_fp16, y = var_9645_to_fp16)[name = string("op_9646_cast_fp16")]; + fp32 var_9647_epsilon_0 = const()[name = string("op_9647_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9647_cast_fp16 = rsqrt(epsilon = var_9647_epsilon_0, x = var_9646_cast_fp16)[name = string("op_9647_cast_fp16")]; + tensor var_9648_cast_fp16 = mul(x = x_295_cast_fp16, y = var_9647_cast_fp16)[name = string("op_9648_cast_fp16")]; + tensor input_409_cast_fp16 = mul(x = var_9648_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_409_cast_fp16")]; + string input_411_pad_type_0 = const()[name = string("input_411_pad_type_0"), val = string("valid")]; + tensor input_411_strides_0 = const()[name = string("input_411_strides_0"), val = tensor([1, 1])]; + tensor input_411_pad_0 = const()[name = string("input_411_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_411_dilations_0 = const()[name = string("input_411_dilations_0"), val = tensor([1, 1])]; + int32 input_411_groups_0 = const()[name = string("input_411_groups_0"), val = int32(1)]; + tensor input_411_cast_fp16 = conv(dilations = input_411_dilations_0, groups = input_411_groups_0, pad = input_411_pad_0, pad_type = input_411_pad_type_0, strides = input_411_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_409_cast_fp16)[name = string("input_411_cast_fp16")]; + tensor var_9656_cast_fp16 = silu(x = input_411_cast_fp16)[name = string("op_9656_cast_fp16")]; + string var_9662_pad_type_0 = const()[name = string("op_9662_pad_type_0"), val = string("valid")]; + tensor var_9662_strides_0 = const()[name = string("op_9662_strides_0"), val = tensor([1, 1])]; + tensor var_9662_pad_0 = const()[name = string("op_9662_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9662_dilations_0 = const()[name = string("op_9662_dilations_0"), val = tensor([1, 1])]; + int32 var_9662_groups_0 = const()[name = string("op_9662_groups_0"), val = int32(1)]; + tensor var_9662_cast_fp16 = conv(dilations = var_9662_dilations_0, groups = var_9662_groups_0, pad = var_9662_pad_0, pad_type = var_9662_pad_type_0, strides = var_9662_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_409_cast_fp16)[name = string("op_9662_cast_fp16")]; + tensor input_413_cast_fp16 = mul(x = var_9656_cast_fp16, y = var_9662_cast_fp16)[name = string("input_413_cast_fp16")]; + string h_77_pad_type_0 = const()[name = string("h_77_pad_type_0"), val = string("valid")]; + tensor h_77_strides_0 = const()[name = string("h_77_strides_0"), val = tensor([1, 1])]; + tensor h_77_pad_0 = const()[name = string("h_77_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_77_dilations_0 = const()[name = string("h_77_dilations_0"), val = tensor([1, 1])]; + int32 h_77_groups_0 = const()[name = string("h_77_groups_0"), val = int32(1)]; + tensor h_77_cast_fp16 = conv(dilations = h_77_dilations_0, groups = h_77_groups_0, pad = h_77_pad_0, pad_type = h_77_pad_type_0, strides = h_77_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_413_cast_fp16)[name = string("h_77_cast_fp16")]; + tensor x_297_cast_fp16 = add(x = x_295_cast_fp16, y = h_77_cast_fp16)[name = string("x_297_cast_fp16")]; + tensor key_cache_79_begin_0 = const()[name = string("key_cache_79_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_79_end_0 = const()[name = string("key_cache_79_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_79_end_mask_0 = const()[name = string("key_cache_79_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_79_cast_fp16 = slice_by_index(begin = key_cache_79_begin_0, end = key_cache_79_end_0, end_mask = key_cache_79_end_mask_0, x = layer_key_caches_15_cast_fp16)[name = string("key_cache_79_cast_fp16")]; + tensor value_cache_79_begin_0 = const()[name = string("value_cache_79_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_79_end_0 = const()[name = string("value_cache_79_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_79_end_mask_0 = const()[name = string("value_cache_79_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_79_cast_fp16 = slice_by_index(begin = value_cache_79_begin_0, end = value_cache_79_end_0, end_mask = value_cache_79_end_mask_0, x = layer_value_caches_15_cast_fp16)[name = string("value_cache_79_cast_fp16")]; + int32 var_9715 = const()[name = string("op_9715"), val = int32(2)]; + int32 var_9719 = const()[name = string("op_9719"), val = int32(3)]; + tensor var_9734_cast_fp16 = mul(x = x_297_cast_fp16, y = x_297_cast_fp16)[name = string("op_9734_cast_fp16")]; + tensor variance_325_axes_0 = const()[name = string("variance_325_axes_0"), val = tensor([1])]; + bool variance_325_keep_dims_0 = const()[name = string("variance_325_keep_dims_0"), val = bool(true)]; + tensor variance_325_cast_fp16 = reduce_mean(axes = variance_325_axes_0, keep_dims = variance_325_keep_dims_0, x = var_9734_cast_fp16)[name = string("variance_325_cast_fp16")]; + fp16 var_9737_to_fp16 = const()[name = string("op_9737_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9738_cast_fp16 = add(x = variance_325_cast_fp16, y = var_9737_to_fp16)[name = string("op_9738_cast_fp16")]; + fp32 var_9739_epsilon_0 = const()[name = string("op_9739_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9739_cast_fp16 = rsqrt(epsilon = var_9739_epsilon_0, x = var_9738_cast_fp16)[name = string("op_9739_cast_fp16")]; + tensor var_9740_cast_fp16 = mul(x = x_297_cast_fp16, y = var_9739_cast_fp16)[name = string("op_9740_cast_fp16")]; + tensor input_415_cast_fp16 = mul(x = var_9740_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_415_cast_fp16")]; + string q_235_pad_type_0 = const()[name = string("q_235_pad_type_0"), val = string("valid")]; + tensor q_235_strides_0 = const()[name = string("q_235_strides_0"), val = tensor([1, 1])]; + tensor q_235_pad_0 = const()[name = string("q_235_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_235_dilations_0 = const()[name = string("q_235_dilations_0"), val = tensor([1, 1])]; + int32 q_235_groups_0 = const()[name = string("q_235_groups_0"), val = int32(1)]; + tensor q_235_cast_fp16 = conv(dilations = q_235_dilations_0, groups = q_235_groups_0, pad = q_235_pad_0, pad_type = q_235_pad_type_0, strides = q_235_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_415_cast_fp16)[name = string("q_235_cast_fp16")]; + string k_235_pad_type_0 = const()[name = string("k_235_pad_type_0"), val = string("valid")]; + tensor k_235_strides_0 = const()[name = string("k_235_strides_0"), val = tensor([1, 1])]; + tensor k_235_pad_0 = const()[name = string("k_235_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_235_dilations_0 = const()[name = string("k_235_dilations_0"), val = tensor([1, 1])]; + int32 k_235_groups_0 = const()[name = string("k_235_groups_0"), val = int32(1)]; + tensor k_235_cast_fp16 = conv(dilations = k_235_dilations_0, groups = k_235_groups_0, pad = k_235_pad_0, pad_type = k_235_pad_type_0, strides = k_235_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_415_cast_fp16)[name = string("k_235_cast_fp16")]; + string v_79_pad_type_0 = const()[name = string("v_79_pad_type_0"), val = string("valid")]; + tensor v_79_strides_0 = const()[name = string("v_79_strides_0"), val = tensor([1, 1])]; + tensor v_79_pad_0 = const()[name = string("v_79_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_79_dilations_0 = const()[name = string("v_79_dilations_0"), val = tensor([1, 1])]; + int32 v_79_groups_0 = const()[name = string("v_79_groups_0"), val = int32(1)]; + tensor v_79_cast_fp16 = conv(dilations = v_79_dilations_0, groups = v_79_groups_0, pad = v_79_pad_0, pad_type = v_79_pad_type_0, strides = v_79_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_415_cast_fp16)[name = string("v_79_cast_fp16")]; + tensor var_9774 = const()[name = string("op_9774"), val = tensor([16, 128, 1, 1])]; + tensor x_299_cast_fp16 = reshape(shape = var_9774, x = q_235_cast_fp16)[name = string("x_299_cast_fp16")]; + tensor var_9777_cast_fp16 = mul(x = x_299_cast_fp16, y = x_299_cast_fp16)[name = string("op_9777_cast_fp16")]; + tensor variance_327_axes_0 = const()[name = string("variance_327_axes_0"), val = tensor([1])]; + bool variance_327_keep_dims_0 = const()[name = string("variance_327_keep_dims_0"), val = bool(true)]; + tensor variance_327_cast_fp16 = reduce_mean(axes = variance_327_axes_0, keep_dims = variance_327_keep_dims_0, x = var_9777_cast_fp16)[name = string("variance_327_cast_fp16")]; + fp16 var_9780_to_fp16 = const()[name = string("op_9780_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9781_cast_fp16 = add(x = variance_327_cast_fp16, y = var_9780_to_fp16)[name = string("op_9781_cast_fp16")]; + fp32 var_9782_epsilon_0 = const()[name = string("op_9782_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9782_cast_fp16 = rsqrt(epsilon = var_9782_epsilon_0, x = var_9781_cast_fp16)[name = string("op_9782_cast_fp16")]; + tensor var_9783_cast_fp16 = mul(x = x_299_cast_fp16, y = var_9782_cast_fp16)[name = string("op_9783_cast_fp16")]; + tensor q_237_cast_fp16 = mul(x = var_9783_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_237_cast_fp16")]; + tensor var_9785 = const()[name = string("op_9785"), val = tensor([8, 128, 1, 1])]; + tensor x_301_cast_fp16 = reshape(shape = var_9785, x = k_235_cast_fp16)[name = string("x_301_cast_fp16")]; + tensor var_9788_cast_fp16 = mul(x = x_301_cast_fp16, y = x_301_cast_fp16)[name = string("op_9788_cast_fp16")]; + tensor variance_329_axes_0 = const()[name = string("variance_329_axes_0"), val = tensor([1])]; + bool variance_329_keep_dims_0 = const()[name = string("variance_329_keep_dims_0"), val = bool(true)]; + tensor variance_329_cast_fp16 = reduce_mean(axes = variance_329_axes_0, keep_dims = variance_329_keep_dims_0, x = var_9788_cast_fp16)[name = string("variance_329_cast_fp16")]; + fp16 var_9791_to_fp16 = const()[name = string("op_9791_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9792_cast_fp16 = add(x = variance_329_cast_fp16, y = var_9791_to_fp16)[name = string("op_9792_cast_fp16")]; + fp32 var_9793_epsilon_0 = const()[name = string("op_9793_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9793_cast_fp16 = rsqrt(epsilon = var_9793_epsilon_0, x = var_9792_cast_fp16)[name = string("op_9793_cast_fp16")]; + tensor var_9794_cast_fp16 = mul(x = x_301_cast_fp16, y = var_9793_cast_fp16)[name = string("op_9794_cast_fp16")]; + tensor k_237_cast_fp16 = mul(x = var_9794_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_237_cast_fp16")]; + tensor var_9796 = const()[name = string("op_9796"), val = tensor([1, 16, 128, 1])]; + tensor z_157_cast_fp16 = reshape(shape = var_9796, x = q_237_cast_fp16)[name = string("z_157_cast_fp16")]; + tensor var_9798 = const()[name = string("op_9798"), val = tensor([1, 8, 128, 1])]; + tensor z_159_cast_fp16 = reshape(shape = var_9798, x = k_237_cast_fp16)[name = string("z_159_cast_fp16")]; + tensor z1_157_begin_0 = const()[name = string("z1_157_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_157_end_0 = const()[name = string("z1_157_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_157_end_mask_0 = const()[name = string("z1_157_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_157_cast_fp16 = slice_by_index(begin = z1_157_begin_0, end = z1_157_end_0, end_mask = z1_157_end_mask_0, x = z_157_cast_fp16)[name = string("z1_157_cast_fp16")]; + tensor z2_157_begin_0 = const()[name = string("z2_157_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_157_end_0 = const()[name = string("z2_157_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_157_end_mask_0 = const()[name = string("z2_157_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_157_cast_fp16 = slice_by_index(begin = z2_157_begin_0, end = z2_157_end_0, end_mask = z2_157_end_mask_0, x = z_157_cast_fp16)[name = string("z2_157_cast_fp16")]; + tensor var_9806_cast_fp16 = mul(x = z_157_cast_fp16, y = cos_71_to_fp16)[name = string("op_9806_cast_fp16")]; + fp16 const_86_promoted_to_fp16 = const()[name = string("const_86_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9807_cast_fp16 = mul(x = z2_157_cast_fp16, y = const_86_promoted_to_fp16)[name = string("op_9807_cast_fp16")]; + bool var_9809_interleave_0 = const()[name = string("op_9809_interleave_0"), val = bool(false)]; + tensor var_9809_cast_fp16 = concat(axis = var_9715, interleave = var_9809_interleave_0, values = (var_9807_cast_fp16, z1_157_cast_fp16))[name = string("op_9809_cast_fp16")]; + tensor var_9810_cast_fp16 = mul(x = var_9809_cast_fp16, y = sin_71_to_fp16)[name = string("op_9810_cast_fp16")]; + tensor q_239_cast_fp16 = add(x = var_9806_cast_fp16, y = var_9810_cast_fp16)[name = string("q_239_cast_fp16")]; + tensor z1_159_begin_0 = const()[name = string("z1_159_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_159_end_0 = const()[name = string("z1_159_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_159_end_mask_0 = const()[name = string("z1_159_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_159_cast_fp16 = slice_by_index(begin = z1_159_begin_0, end = z1_159_end_0, end_mask = z1_159_end_mask_0, x = z_159_cast_fp16)[name = string("z1_159_cast_fp16")]; + tensor z2_159_begin_0 = const()[name = string("z2_159_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_159_end_0 = const()[name = string("z2_159_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_159_end_mask_0 = const()[name = string("z2_159_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_159_cast_fp16 = slice_by_index(begin = z2_159_begin_0, end = z2_159_end_0, end_mask = z2_159_end_mask_0, x = z_159_cast_fp16)[name = string("z2_159_cast_fp16")]; + tensor var_9818_cast_fp16 = mul(x = z_159_cast_fp16, y = cos_71_to_fp16)[name = string("op_9818_cast_fp16")]; + fp16 const_87_promoted_to_fp16 = const()[name = string("const_87_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9819_cast_fp16 = mul(x = z2_159_cast_fp16, y = const_87_promoted_to_fp16)[name = string("op_9819_cast_fp16")]; + bool var_9821_interleave_0 = const()[name = string("op_9821_interleave_0"), val = bool(false)]; + tensor var_9821_cast_fp16 = concat(axis = var_9715, interleave = var_9821_interleave_0, values = (var_9819_cast_fp16, z1_159_cast_fp16))[name = string("op_9821_cast_fp16")]; + tensor var_9822_cast_fp16 = mul(x = var_9821_cast_fp16, y = sin_71_to_fp16)[name = string("op_9822_cast_fp16")]; + tensor k_239_cast_fp16 = add(x = var_9818_cast_fp16, y = var_9822_cast_fp16)[name = string("k_239_cast_fp16")]; + tensor var_9824 = const()[name = string("op_9824"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_79_cast_fp16 = reshape(shape = var_9824, x = k_239_cast_fp16)[name = string("cur_key_79_cast_fp16")]; + tensor var_9826_to_fp16 = const()[name = string("op_9826_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639168)))]; + tensor var_9827_cast_fp16 = mul(x = key_cache_79_cast_fp16, y = var_9826_to_fp16)[name = string("op_9827_cast_fp16")]; + tensor upd_79_to_fp16 = const()[name = string("upd_79_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639296)))]; + tensor var_9828_cast_fp16 = mul(x = cur_key_79_cast_fp16, y = upd_79_to_fp16)[name = string("op_9828_cast_fp16")]; + tensor key_79_cast_fp16 = add(x = var_9827_cast_fp16, y = var_9828_cast_fp16)[name = string("key_79_cast_fp16")]; + tensor var_9830_to_fp16 = const()[name = string("op_9830_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639168)))]; + tensor var_9831_cast_fp16 = mul(x = value_cache_79_cast_fp16, y = var_9830_to_fp16)[name = string("op_9831_cast_fp16")]; + tensor var_9832_cast_fp16 = mul(x = v_79_cast_fp16, y = upd_79_to_fp16)[name = string("op_9832_cast_fp16")]; + tensor value_79_cast_fp16 = add(x = var_9831_cast_fp16, y = var_9832_cast_fp16)[name = string("value_79_cast_fp16")]; + tensor var_9834 = const()[name = string("op_9834"), val = tensor([1, 8, 128, 16])]; + tensor kh_157_cast_fp16 = reshape(shape = var_9834, x = key_79_cast_fp16)[name = string("kh_157_cast_fp16")]; + tensor var_9836 = const()[name = string("op_9836"), val = tensor([1, 8, 128, 16])]; + tensor vh_157_cast_fp16 = reshape(shape = var_9836, x = value_79_cast_fp16)[name = string("vh_157_cast_fp16")]; + tensor transpose_156_perm_0 = const()[name = string("transpose_156_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_78_reps_0 = const()[name = string("tile_78_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_156_cast_fp16 = transpose(perm = transpose_156_perm_0, x = kh_157_cast_fp16)[name = string("transpose_245")]; + tensor tile_78_cast_fp16 = tile(reps = tile_78_reps_0, x = transpose_156_cast_fp16)[name = string("tile_78_cast_fp16")]; + tensor concat_194 = const()[name = string("concat_194"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_156_cast_fp16 = reshape(shape = concat_194, x = tile_78_cast_fp16)[name = string("reshape_156_cast_fp16")]; + tensor transpose_157_perm_0 = const()[name = string("transpose_157_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_195 = const()[name = string("concat_195"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_157_cast_fp16 = transpose(perm = transpose_157_perm_0, x = reshape_156_cast_fp16)[name = string("transpose_244")]; + tensor reshape_157_cast_fp16 = reshape(shape = concat_195, x = transpose_157_cast_fp16)[name = string("reshape_157_cast_fp16")]; + tensor transpose_158_perm_0 = const()[name = string("transpose_158_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_79_reps_0 = const()[name = string("tile_79_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_158_cast_fp16 = transpose(perm = transpose_158_perm_0, x = vh_157_cast_fp16)[name = string("transpose_243")]; + tensor tile_79_cast_fp16 = tile(reps = tile_79_reps_0, x = transpose_158_cast_fp16)[name = string("tile_79_cast_fp16")]; + tensor concat_196 = const()[name = string("concat_196"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_158_cast_fp16 = reshape(shape = concat_196, x = tile_79_cast_fp16)[name = string("reshape_158_cast_fp16")]; + tensor transpose_159_perm_0 = const()[name = string("transpose_159_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_197 = const()[name = string("concat_197"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_159_cast_fp16 = transpose(perm = transpose_159_perm_0, x = reshape_158_cast_fp16)[name = string("transpose_242")]; + tensor reshape_159_cast_fp16 = reshape(shape = concat_197, x = transpose_159_cast_fp16)[name = string("reshape_159_cast_fp16")]; + fp16 var_9840_to_fp16 = const()[name = string("op_9840_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_9841_cast_fp16 = mul(x = q_239_cast_fp16, y = var_9840_to_fp16)[name = string("op_9841_cast_fp16")]; + tensor transpose_473_perm_0 = const()[name = string("transpose_473_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_169_transpose_x_1 = const()[name = string("w_169_transpose_x_1"), val = bool(true)]; + bool w_169_transpose_y_1 = const()[name = string("w_169_transpose_y_1"), val = bool(false)]; + tensor transpose_473_cast_fp16 = transpose(perm = transpose_473_perm_0, x = reshape_157_cast_fp16)[name = string("transpose_241")]; + tensor w_169_cast_fp16 = matmul(transpose_x = w_169_transpose_x_1, transpose_y = w_169_transpose_y_1, x = var_9841_cast_fp16, y = transpose_473_cast_fp16)[name = string("w_169_cast_fp16")]; + tensor pad_79_to_fp16 = const()[name = string("pad_79_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639424)))]; + tensor var_9844_cast_fp16 = add(x = w_169_cast_fp16, y = pad_79_to_fp16)[name = string("op_9844_cast_fp16")]; + tensor w_171_cast_fp16 = softmax(axis = var_9719, x = var_9844_cast_fp16)[name = string("w_171_cast_fp16")]; + tensor transpose_474_perm_0 = const()[name = string("transpose_474_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_79_transpose_x_1 = const()[name = string("attn_79_transpose_x_1"), val = bool(false)]; + bool attn_79_transpose_y_1 = const()[name = string("attn_79_transpose_y_1"), val = bool(true)]; + tensor transpose_474_cast_fp16 = transpose(perm = transpose_474_perm_0, x = reshape_159_cast_fp16)[name = string("transpose_240")]; + tensor attn_79_cast_fp16 = matmul(transpose_x = attn_79_transpose_x_1, transpose_y = attn_79_transpose_y_1, x = transpose_474_cast_fp16, y = w_171_cast_fp16)[name = string("attn_79_cast_fp16")]; + tensor var_9848 = const()[name = string("op_9848"), val = tensor([1, 2048, 1, 1])]; + tensor input_417_cast_fp16 = reshape(shape = var_9848, x = attn_79_cast_fp16)[name = string("input_417_cast_fp16")]; + string attn_output_79_pad_type_0 = const()[name = string("attn_output_79_pad_type_0"), val = string("valid")]; + tensor attn_output_79_strides_0 = const()[name = string("attn_output_79_strides_0"), val = tensor([1, 1])]; + tensor attn_output_79_pad_0 = const()[name = string("attn_output_79_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_79_dilations_0 = const()[name = string("attn_output_79_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_79_groups_0 = const()[name = string("attn_output_79_groups_0"), val = int32(1)]; + tensor attn_output_79_cast_fp16 = conv(dilations = attn_output_79_dilations_0, groups = attn_output_79_groups_0, pad = attn_output_79_pad_0, pad_type = attn_output_79_pad_type_0, strides = attn_output_79_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_417_cast_fp16)[name = string("attn_output_79_cast_fp16")]; + tensor x_303_cast_fp16 = add(x = x_297_cast_fp16, y = attn_output_79_cast_fp16)[name = string("x_303_cast_fp16")]; + tensor var_9862_cast_fp16 = mul(x = x_303_cast_fp16, y = x_303_cast_fp16)[name = string("op_9862_cast_fp16")]; + tensor variance_331_axes_0 = const()[name = string("variance_331_axes_0"), val = tensor([1])]; + bool variance_331_keep_dims_0 = const()[name = string("variance_331_keep_dims_0"), val = bool(true)]; + tensor variance_331_cast_fp16 = reduce_mean(axes = variance_331_axes_0, keep_dims = variance_331_keep_dims_0, x = var_9862_cast_fp16)[name = string("variance_331_cast_fp16")]; + fp16 var_9865_to_fp16 = const()[name = string("op_9865_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9866_cast_fp16 = add(x = variance_331_cast_fp16, y = var_9865_to_fp16)[name = string("op_9866_cast_fp16")]; + fp32 var_9867_epsilon_0 = const()[name = string("op_9867_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9867_cast_fp16 = rsqrt(epsilon = var_9867_epsilon_0, x = var_9866_cast_fp16)[name = string("op_9867_cast_fp16")]; + tensor var_9868_cast_fp16 = mul(x = x_303_cast_fp16, y = var_9867_cast_fp16)[name = string("op_9868_cast_fp16")]; + tensor input_419_cast_fp16 = mul(x = var_9868_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_419_cast_fp16")]; + string input_421_pad_type_0 = const()[name = string("input_421_pad_type_0"), val = string("valid")]; + tensor input_421_strides_0 = const()[name = string("input_421_strides_0"), val = tensor([1, 1])]; + tensor input_421_pad_0 = const()[name = string("input_421_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_421_dilations_0 = const()[name = string("input_421_dilations_0"), val = tensor([1, 1])]; + int32 input_421_groups_0 = const()[name = string("input_421_groups_0"), val = int32(1)]; + tensor input_421_cast_fp16 = conv(dilations = input_421_dilations_0, groups = input_421_groups_0, pad = input_421_pad_0, pad_type = input_421_pad_type_0, strides = input_421_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_419_cast_fp16)[name = string("input_421_cast_fp16")]; + tensor var_9876_cast_fp16 = silu(x = input_421_cast_fp16)[name = string("op_9876_cast_fp16")]; + string var_9882_pad_type_0 = const()[name = string("op_9882_pad_type_0"), val = string("valid")]; + tensor var_9882_strides_0 = const()[name = string("op_9882_strides_0"), val = tensor([1, 1])]; + tensor var_9882_pad_0 = const()[name = string("op_9882_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9882_dilations_0 = const()[name = string("op_9882_dilations_0"), val = tensor([1, 1])]; + int32 var_9882_groups_0 = const()[name = string("op_9882_groups_0"), val = int32(1)]; + tensor var_9882_cast_fp16 = conv(dilations = var_9882_dilations_0, groups = var_9882_groups_0, pad = var_9882_pad_0, pad_type = var_9882_pad_type_0, strides = var_9882_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_419_cast_fp16)[name = string("op_9882_cast_fp16")]; + tensor input_423_cast_fp16 = mul(x = var_9876_cast_fp16, y = var_9882_cast_fp16)[name = string("input_423_cast_fp16")]; + string h_79_pad_type_0 = const()[name = string("h_79_pad_type_0"), val = string("valid")]; + tensor h_79_strides_0 = const()[name = string("h_79_strides_0"), val = tensor([1, 1])]; + tensor h_79_pad_0 = const()[name = string("h_79_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_79_dilations_0 = const()[name = string("h_79_dilations_0"), val = tensor([1, 1])]; + int32 h_79_groups_0 = const()[name = string("h_79_groups_0"), val = int32(1)]; + tensor h_79_cast_fp16 = conv(dilations = h_79_dilations_0, groups = h_79_groups_0, pad = h_79_pad_0, pad_type = h_79_pad_type_0, strides = h_79_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_423_cast_fp16)[name = string("h_79_cast_fp16")]; + tensor inputs_13_cast_fp16 = add(x = x_303_cast_fp16, y = h_79_cast_fp16)[name = string("inputs_13_cast_fp16")]; + int32 var_9910 = const()[name = string("op_9910"), val = int32(1)]; + bool layer_key_caches_17_interleave_0 = const()[name = string("layer_key_caches_17_interleave_0"), val = bool(false)]; + tensor layer_key_caches_17_cast_fp16 = concat(axis = var_9910, interleave = layer_key_caches_17_interleave_0, values = (key_71_cast_fp16, key_73_cast_fp16, key_75_cast_fp16, key_77_cast_fp16, key_79_cast_fp16))[name = string("layer_key_caches_17_cast_fp16")]; + int32 var_9913 = const()[name = string("op_9913"), val = int32(1)]; + bool layer_value_caches_17_interleave_0 = const()[name = string("layer_value_caches_17_interleave_0"), val = bool(false)]; + tensor layer_value_caches_17_cast_fp16 = concat(axis = var_9913, interleave = layer_value_caches_17_interleave_0, values = (value_71_cast_fp16, value_73_cast_fp16, value_75_cast_fp16, value_77_cast_fp16, value_79_cast_fp16))[name = string("layer_value_caches_17_cast_fp16")]; + tensor inputs_sq_13_cast_fp16 = mul(x = inputs_13_cast_fp16, y = inputs_13_cast_fp16)[name = string("inputs_sq_13_cast_fp16")]; + tensor variance_333_axes_0 = const()[name = string("variance_333_axes_0"), val = tensor([1])]; + bool variance_333_keep_dims_0 = const()[name = string("variance_333_keep_dims_0"), val = bool(true)]; + tensor variance_333_cast_fp16 = reduce_mean(axes = variance_333_axes_0, keep_dims = variance_333_keep_dims_0, x = inputs_sq_13_cast_fp16)[name = string("variance_333_cast_fp16")]; + fp16 var_9923_to_fp16 = const()[name = string("op_9923_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9924_cast_fp16 = add(x = variance_333_cast_fp16, y = var_9923_to_fp16)[name = string("op_9924_cast_fp16")]; + fp32 var_9925_epsilon_0 = const()[name = string("op_9925_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9925_cast_fp16 = rsqrt(epsilon = var_9925_epsilon_0, x = var_9924_cast_fp16)[name = string("op_9925_cast_fp16")]; + tensor hidden_states_13_cast_fp16 = mul(x = inputs_13_cast_fp16, y = var_9925_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; + tensor input_425_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_13_cast_fp16)[name = string("input_425_cast_fp16")]; + string logits_25_pad_type_0 = const()[name = string("logits_25_pad_type_0"), val = string("valid")]; + tensor logits_25_strides_0 = const()[name = string("logits_25_strides_0"), val = tensor([1, 1])]; + tensor logits_25_pad_0 = const()[name = string("logits_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_25_dilations_0 = const()[name = string("logits_25_dilations_0"), val = tensor([1, 1])]; + int32 logits_25_groups_0 = const()[name = string("logits_25_groups_0"), val = int32(1)]; + tensor lm_heads_6_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91293440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93390656))))[name = string("lm_heads_6_weight_to_fp16_palettized")]; + tensor logits_25_cast_fp16 = conv(dilations = logits_25_dilations_0, groups = logits_25_groups_0, pad = logits_25_pad_0, pad_type = logits_25_pad_type_0, strides = logits_25_strides_0, weight = lm_heads_6_weight_to_fp16_palettized, x = input_425_cast_fp16)[name = string("logits_25_cast_fp16")]; + tensor var_9943 = const()[name = string("op_9943"), val = tensor([1, 2048])]; + tensor logits_27_cast_fp16 = reshape(shape = var_9943, x = logits_25_cast_fp16)[name = string("logits_27_cast_fp16")]; + tensor scaled_logits_13_cast_fp16 = real_div(x = logits_27_cast_fp16, y = temperature)[name = string("scaled_logits_13_cast_fp16")]; + int32 var_9953 = const()[name = string("op_9953"), val = int32(100)]; + int32 top_values_13_axis_0 = const()[name = string("top_values_13_axis_0"), val = int32(1)]; + bool top_values_13_ascending_0 = const()[name = string("top_values_13_ascending_0"), val = bool(false)]; + bool top_values_13_sort_0 = const()[name = string("top_values_13_sort_0"), val = bool(true)]; + bool top_values_13_return_indices_0 = const()[name = string("top_values_13_return_indices_0"), val = bool(true)]; + string top_values_13_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_13_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_13_cast_fp16_cast_uint16_0, tensor top_values_13_cast_fp16_cast_uint16_1 = topk(ascending = top_values_13_ascending_0, axis = top_values_13_axis_0, k = var_9953, output_indices_dtype = top_values_13_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_13_return_indices_0, sort = top_values_13_sort_0, x = scaled_logits_13_cast_fp16)[name = string("top_values_13_cast_fp16_cast_uint16")]; + tensor var_9959_cast_fp16 = mul(x = top_values_13_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_9959_cast_fp16")]; + tensor var_9963_cast_fp16 = add(x = var_9959_cast_fp16, y = var_2438_cast_fp16)[name = string("op_9963_cast_fp16")]; + tensor reduce_min_6_axes_0 = const()[name = string("reduce_min_6_axes_0"), val = tensor([1])]; + bool reduce_min_6_keep_dims_0 = const()[name = string("reduce_min_6_keep_dims_0"), val = bool(true)]; + tensor reduce_min_6_cast_fp16 = reduce_min(axes = reduce_min_6_axes_0, keep_dims = reduce_min_6_keep_dims_0, x = var_9963_cast_fp16)[name = string("reduce_min_6_cast_fp16")]; + tensor var_9966_cast_fp16 = greater_equal(x = scaled_logits_13_cast_fp16, y = reduce_min_6_cast_fp16)[name = string("op_9966_cast_fp16")]; + fp16 var_9967_value_0_to_fp16 = const()[name = string("op_9967_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_9967_cast_fp16 = fill_like(ref_tensor = scaled_logits_13_cast_fp16, value = var_9967_value_0_to_fp16)[name = string("op_9967_cast_fp16")]; + tensor masked_logits_13_cast_fp16 = select(a = scaled_logits_13_cast_fp16, b = var_9967_cast_fp16, cond = var_9966_cast_fp16)[name = string("masked_logits_13_cast_fp16")]; + tensor var_9971_begin_0 = const()[name = string("op_9971_begin_0"), val = tensor([6, 0])]; + tensor var_9971_end_0 = const()[name = string("op_9971_end_0"), val = tensor([7, 2048])]; + tensor var_9971_end_mask_0 = const()[name = string("op_9971_end_mask_0"), val = tensor([false, true])]; + tensor var_9971_squeeze_mask_0 = const()[name = string("op_9971_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_9971_cast_fp16 = slice_by_index(begin = var_9971_begin_0, end = var_9971_end_0, end_mask = var_9971_end_mask_0, squeeze_mask = var_9971_squeeze_mask_0, x = gumbel)[name = string("op_9971_cast_fp16")]; + tensor var_9974 = const()[name = string("op_9974"), val = tensor([1, 2048])]; + tensor var_9975_cast_fp16 = reshape(shape = var_9974, x = var_9971_cast_fp16)[name = string("op_9975_cast_fp16")]; + tensor noisy_logits_13_cast_fp16 = add(x = masked_logits_13_cast_fp16, y = var_9975_cast_fp16)[name = string("noisy_logits_13_cast_fp16")]; + int32 code_13_axis_0 = const()[name = string("code_13_axis_0"), val = int32(1)]; + bool code_13_keep_dims_0 = const()[name = string("code_13_keep_dims_0"), val = bool(false)]; + string code_13_output_dtype_0 = const()[name = string("code_13_output_dtype_0"), val = string("int32")]; + tensor code_13_cast_fp16 = reduce_argmax(axis = code_13_axis_0, keep_dims = code_13_keep_dims_0, output_dtype = code_13_output_dtype_0, x = noisy_logits_13_cast_fp16)[name = string("code_13_cast_fp16")]; + int32 var_9986 = const()[name = string("op_9986"), val = int32(12288)]; + tensor input_427 = add(x = code_13_cast_fp16, y = var_9986)[name = string("input_427")]; + int32 code_embed_25_axis_0 = const()[name = string("code_embed_25_axis_0"), val = int32(0)]; + int32 code_embed_25_batch_dims_0 = const()[name = string("code_embed_25_batch_dims_0"), val = int32(0)]; + bool code_embed_25_validate_indices_0 = const()[name = string("code_embed_25_validate_indices_0"), val = bool(false)]; + string input_427_to_uint16_dtype_0 = const()[name = string("input_427_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_427_to_uint16 = cast(dtype = input_427_to_uint16_dtype_0, x = input_427)[name = string("cast_8")]; + tensor code_embed_25_cast_fp16_cast_uint16 = gather(axis = code_embed_25_axis_0, batch_dims = code_embed_25_batch_dims_0, indices = input_427_to_uint16, validate_indices = code_embed_25_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_25_cast_fp16_cast_uint16")]; + tensor var_9990 = const()[name = string("op_9990"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_27_cast_fp16 = reshape(shape = var_9990, x = code_embed_25_cast_fp16_cast_uint16)[name = string("code_embed_27_cast_fp16")]; + tensor embed_sum_15_cast_fp16 = add(x = embed_sum_13_cast_fp16, y = code_embed_27_cast_fp16)[name = string("embed_sum_15_cast_fp16")]; + tensor key_cache_81_begin_0 = const()[name = string("key_cache_81_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_81_end_0 = const()[name = string("key_cache_81_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_81_end_mask_0 = const()[name = string("key_cache_81_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_81_cast_fp16 = slice_by_index(begin = key_cache_81_begin_0, end = key_cache_81_end_0, end_mask = key_cache_81_end_mask_0, x = layer_key_caches_17_cast_fp16)[name = string("key_cache_81_cast_fp16")]; + tensor value_cache_81_begin_0 = const()[name = string("value_cache_81_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_81_end_0 = const()[name = string("value_cache_81_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_81_end_mask_0 = const()[name = string("value_cache_81_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_81_cast_fp16 = slice_by_index(begin = value_cache_81_begin_0, end = value_cache_81_end_0, end_mask = value_cache_81_end_mask_0, x = layer_value_caches_17_cast_fp16)[name = string("value_cache_81_cast_fp16")]; + int32 var_10089 = const()[name = string("op_10089"), val = int32(2)]; + int32 var_10093 = const()[name = string("op_10093"), val = int32(3)]; + tensor var_10108_cast_fp16 = mul(x = code_embed_27_cast_fp16, y = code_embed_27_cast_fp16)[name = string("op_10108_cast_fp16")]; + tensor variance_335_axes_0 = const()[name = string("variance_335_axes_0"), val = tensor([1])]; + bool variance_335_keep_dims_0 = const()[name = string("variance_335_keep_dims_0"), val = bool(true)]; + tensor variance_335_cast_fp16 = reduce_mean(axes = variance_335_axes_0, keep_dims = variance_335_keep_dims_0, x = var_10108_cast_fp16)[name = string("variance_335_cast_fp16")]; + fp16 var_10111_to_fp16 = const()[name = string("op_10111_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10112_cast_fp16 = add(x = variance_335_cast_fp16, y = var_10111_to_fp16)[name = string("op_10112_cast_fp16")]; + fp32 var_10113_epsilon_0 = const()[name = string("op_10113_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10113_cast_fp16 = rsqrt(epsilon = var_10113_epsilon_0, x = var_10112_cast_fp16)[name = string("op_10113_cast_fp16")]; + tensor var_10114_cast_fp16 = mul(x = code_embed_27_cast_fp16, y = var_10113_cast_fp16)[name = string("op_10114_cast_fp16")]; + tensor input_429_cast_fp16 = mul(x = var_10114_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_429_cast_fp16")]; + string q_241_pad_type_0 = const()[name = string("q_241_pad_type_0"), val = string("valid")]; + tensor q_241_strides_0 = const()[name = string("q_241_strides_0"), val = tensor([1, 1])]; + tensor q_241_pad_0 = const()[name = string("q_241_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_241_dilations_0 = const()[name = string("q_241_dilations_0"), val = tensor([1, 1])]; + int32 q_241_groups_0 = const()[name = string("q_241_groups_0"), val = int32(1)]; + tensor q_241_cast_fp16 = conv(dilations = q_241_dilations_0, groups = q_241_groups_0, pad = q_241_pad_0, pad_type = q_241_pad_type_0, strides = q_241_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_429_cast_fp16)[name = string("q_241_cast_fp16")]; + string k_241_pad_type_0 = const()[name = string("k_241_pad_type_0"), val = string("valid")]; + tensor k_241_strides_0 = const()[name = string("k_241_strides_0"), val = tensor([1, 1])]; + tensor k_241_pad_0 = const()[name = string("k_241_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_241_dilations_0 = const()[name = string("k_241_dilations_0"), val = tensor([1, 1])]; + int32 k_241_groups_0 = const()[name = string("k_241_groups_0"), val = int32(1)]; + tensor k_241_cast_fp16 = conv(dilations = k_241_dilations_0, groups = k_241_groups_0, pad = k_241_pad_0, pad_type = k_241_pad_type_0, strides = k_241_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_429_cast_fp16)[name = string("k_241_cast_fp16")]; + string v_81_pad_type_0 = const()[name = string("v_81_pad_type_0"), val = string("valid")]; + tensor v_81_strides_0 = const()[name = string("v_81_strides_0"), val = tensor([1, 1])]; + tensor v_81_pad_0 = const()[name = string("v_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_81_dilations_0 = const()[name = string("v_81_dilations_0"), val = tensor([1, 1])]; + int32 v_81_groups_0 = const()[name = string("v_81_groups_0"), val = int32(1)]; + tensor v_81_cast_fp16 = conv(dilations = v_81_dilations_0, groups = v_81_groups_0, pad = v_81_pad_0, pad_type = v_81_pad_type_0, strides = v_81_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_429_cast_fp16)[name = string("v_81_cast_fp16")]; + tensor var_10148 = const()[name = string("op_10148"), val = tensor([16, 128, 1, 1])]; + tensor x_305_cast_fp16 = reshape(shape = var_10148, x = q_241_cast_fp16)[name = string("x_305_cast_fp16")]; + tensor var_10151_cast_fp16 = mul(x = x_305_cast_fp16, y = x_305_cast_fp16)[name = string("op_10151_cast_fp16")]; + tensor variance_337_axes_0 = const()[name = string("variance_337_axes_0"), val = tensor([1])]; + bool variance_337_keep_dims_0 = const()[name = string("variance_337_keep_dims_0"), val = bool(true)]; + tensor variance_337_cast_fp16 = reduce_mean(axes = variance_337_axes_0, keep_dims = variance_337_keep_dims_0, x = var_10151_cast_fp16)[name = string("variance_337_cast_fp16")]; + fp16 var_10154_to_fp16 = const()[name = string("op_10154_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10155_cast_fp16 = add(x = variance_337_cast_fp16, y = var_10154_to_fp16)[name = string("op_10155_cast_fp16")]; + fp32 var_10156_epsilon_0 = const()[name = string("op_10156_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10156_cast_fp16 = rsqrt(epsilon = var_10156_epsilon_0, x = var_10155_cast_fp16)[name = string("op_10156_cast_fp16")]; + tensor var_10157_cast_fp16 = mul(x = x_305_cast_fp16, y = var_10156_cast_fp16)[name = string("op_10157_cast_fp16")]; + tensor q_243_cast_fp16 = mul(x = var_10157_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_243_cast_fp16")]; + tensor var_10159 = const()[name = string("op_10159"), val = tensor([8, 128, 1, 1])]; + tensor x_307_cast_fp16 = reshape(shape = var_10159, x = k_241_cast_fp16)[name = string("x_307_cast_fp16")]; + tensor var_10162_cast_fp16 = mul(x = x_307_cast_fp16, y = x_307_cast_fp16)[name = string("op_10162_cast_fp16")]; + tensor variance_339_axes_0 = const()[name = string("variance_339_axes_0"), val = tensor([1])]; + bool variance_339_keep_dims_0 = const()[name = string("variance_339_keep_dims_0"), val = bool(true)]; + tensor variance_339_cast_fp16 = reduce_mean(axes = variance_339_axes_0, keep_dims = variance_339_keep_dims_0, x = var_10162_cast_fp16)[name = string("variance_339_cast_fp16")]; + fp16 var_10165_to_fp16 = const()[name = string("op_10165_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10166_cast_fp16 = add(x = variance_339_cast_fp16, y = var_10165_to_fp16)[name = string("op_10166_cast_fp16")]; + fp32 var_10167_epsilon_0 = const()[name = string("op_10167_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10167_cast_fp16 = rsqrt(epsilon = var_10167_epsilon_0, x = var_10166_cast_fp16)[name = string("op_10167_cast_fp16")]; + tensor var_10168_cast_fp16 = mul(x = x_307_cast_fp16, y = var_10167_cast_fp16)[name = string("op_10168_cast_fp16")]; + tensor k_243_cast_fp16 = mul(x = var_10168_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_243_cast_fp16")]; + tensor var_10170 = const()[name = string("op_10170"), val = tensor([1, 16, 128, 1])]; + tensor z_161_cast_fp16 = reshape(shape = var_10170, x = q_243_cast_fp16)[name = string("z_161_cast_fp16")]; + tensor var_10172 = const()[name = string("op_10172"), val = tensor([1, 8, 128, 1])]; + tensor z_163_cast_fp16 = reshape(shape = var_10172, x = k_243_cast_fp16)[name = string("z_163_cast_fp16")]; + tensor z1_161_begin_0 = const()[name = string("z1_161_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_161_end_0 = const()[name = string("z1_161_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_161_end_mask_0 = const()[name = string("z1_161_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_161_cast_fp16 = slice_by_index(begin = z1_161_begin_0, end = z1_161_end_0, end_mask = z1_161_end_mask_0, x = z_161_cast_fp16)[name = string("z1_161_cast_fp16")]; + tensor z2_161_begin_0 = const()[name = string("z2_161_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_161_end_0 = const()[name = string("z2_161_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_161_end_mask_0 = const()[name = string("z2_161_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_161_cast_fp16 = slice_by_index(begin = z2_161_begin_0, end = z2_161_end_0, end_mask = z2_161_end_mask_0, x = z_161_cast_fp16)[name = string("z2_161_cast_fp16")]; + tensor cos_81_to_fp16 = const()[name = string("cos_81_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639552)))]; + tensor var_10180_cast_fp16 = mul(x = z_161_cast_fp16, y = cos_81_to_fp16)[name = string("op_10180_cast_fp16")]; + fp16 const_89_promoted_to_fp16 = const()[name = string("const_89_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10181_cast_fp16 = mul(x = z2_161_cast_fp16, y = const_89_promoted_to_fp16)[name = string("op_10181_cast_fp16")]; + bool var_10183_interleave_0 = const()[name = string("op_10183_interleave_0"), val = bool(false)]; + tensor var_10183_cast_fp16 = concat(axis = var_10089, interleave = var_10183_interleave_0, values = (var_10181_cast_fp16, z1_161_cast_fp16))[name = string("op_10183_cast_fp16")]; + tensor sin_81_to_fp16 = const()[name = string("sin_81_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141639872)))]; + tensor var_10184_cast_fp16 = mul(x = var_10183_cast_fp16, y = sin_81_to_fp16)[name = string("op_10184_cast_fp16")]; + tensor q_245_cast_fp16 = add(x = var_10180_cast_fp16, y = var_10184_cast_fp16)[name = string("q_245_cast_fp16")]; + tensor z1_163_begin_0 = const()[name = string("z1_163_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_163_end_0 = const()[name = string("z1_163_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_163_end_mask_0 = const()[name = string("z1_163_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_163_cast_fp16 = slice_by_index(begin = z1_163_begin_0, end = z1_163_end_0, end_mask = z1_163_end_mask_0, x = z_163_cast_fp16)[name = string("z1_163_cast_fp16")]; + tensor z2_163_begin_0 = const()[name = string("z2_163_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_163_end_0 = const()[name = string("z2_163_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_163_end_mask_0 = const()[name = string("z2_163_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_163_cast_fp16 = slice_by_index(begin = z2_163_begin_0, end = z2_163_end_0, end_mask = z2_163_end_mask_0, x = z_163_cast_fp16)[name = string("z2_163_cast_fp16")]; + tensor var_10192_cast_fp16 = mul(x = z_163_cast_fp16, y = cos_81_to_fp16)[name = string("op_10192_cast_fp16")]; + fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10193_cast_fp16 = mul(x = z2_163_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_10193_cast_fp16")]; + bool var_10195_interleave_0 = const()[name = string("op_10195_interleave_0"), val = bool(false)]; + tensor var_10195_cast_fp16 = concat(axis = var_10089, interleave = var_10195_interleave_0, values = (var_10193_cast_fp16, z1_163_cast_fp16))[name = string("op_10195_cast_fp16")]; + tensor var_10196_cast_fp16 = mul(x = var_10195_cast_fp16, y = sin_81_to_fp16)[name = string("op_10196_cast_fp16")]; + tensor k_245_cast_fp16 = add(x = var_10192_cast_fp16, y = var_10196_cast_fp16)[name = string("k_245_cast_fp16")]; + tensor var_10198 = const()[name = string("op_10198"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_81_cast_fp16 = reshape(shape = var_10198, x = k_245_cast_fp16)[name = string("cur_key_81_cast_fp16")]; + tensor var_10200_to_fp16 = const()[name = string("op_10200_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640192)))]; + tensor var_10201_cast_fp16 = mul(x = key_cache_81_cast_fp16, y = var_10200_to_fp16)[name = string("op_10201_cast_fp16")]; + tensor upd_81_to_fp16 = const()[name = string("upd_81_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640320)))]; + tensor var_10202_cast_fp16 = mul(x = cur_key_81_cast_fp16, y = upd_81_to_fp16)[name = string("op_10202_cast_fp16")]; + tensor key_81_cast_fp16 = add(x = var_10201_cast_fp16, y = var_10202_cast_fp16)[name = string("key_81_cast_fp16")]; + tensor var_10204_to_fp16 = const()[name = string("op_10204_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640192)))]; + tensor var_10205_cast_fp16 = mul(x = value_cache_81_cast_fp16, y = var_10204_to_fp16)[name = string("op_10205_cast_fp16")]; + tensor var_10206_cast_fp16 = mul(x = v_81_cast_fp16, y = upd_81_to_fp16)[name = string("op_10206_cast_fp16")]; + tensor value_81_cast_fp16 = add(x = var_10205_cast_fp16, y = var_10206_cast_fp16)[name = string("value_81_cast_fp16")]; + tensor var_10208 = const()[name = string("op_10208"), val = tensor([1, 8, 128, 16])]; + tensor kh_161_cast_fp16 = reshape(shape = var_10208, x = key_81_cast_fp16)[name = string("kh_161_cast_fp16")]; + tensor var_10210 = const()[name = string("op_10210"), val = tensor([1, 8, 128, 16])]; + tensor vh_161_cast_fp16 = reshape(shape = var_10210, x = value_81_cast_fp16)[name = string("vh_161_cast_fp16")]; + tensor transpose_160_perm_0 = const()[name = string("transpose_160_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_80_reps_0 = const()[name = string("tile_80_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_160_cast_fp16 = transpose(perm = transpose_160_perm_0, x = kh_161_cast_fp16)[name = string("transpose_239")]; + tensor tile_80_cast_fp16 = tile(reps = tile_80_reps_0, x = transpose_160_cast_fp16)[name = string("tile_80_cast_fp16")]; + tensor concat_203 = const()[name = string("concat_203"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_160_cast_fp16 = reshape(shape = concat_203, x = tile_80_cast_fp16)[name = string("reshape_160_cast_fp16")]; + tensor transpose_161_perm_0 = const()[name = string("transpose_161_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_204 = const()[name = string("concat_204"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_161_cast_fp16 = transpose(perm = transpose_161_perm_0, x = reshape_160_cast_fp16)[name = string("transpose_238")]; + tensor reshape_161_cast_fp16 = reshape(shape = concat_204, x = transpose_161_cast_fp16)[name = string("reshape_161_cast_fp16")]; + tensor transpose_162_perm_0 = const()[name = string("transpose_162_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_81_reps_0 = const()[name = string("tile_81_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_162_cast_fp16 = transpose(perm = transpose_162_perm_0, x = vh_161_cast_fp16)[name = string("transpose_237")]; + tensor tile_81_cast_fp16 = tile(reps = tile_81_reps_0, x = transpose_162_cast_fp16)[name = string("tile_81_cast_fp16")]; + tensor concat_205 = const()[name = string("concat_205"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_162_cast_fp16 = reshape(shape = concat_205, x = tile_81_cast_fp16)[name = string("reshape_162_cast_fp16")]; + tensor transpose_163_perm_0 = const()[name = string("transpose_163_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_206 = const()[name = string("concat_206"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_163_cast_fp16 = transpose(perm = transpose_163_perm_0, x = reshape_162_cast_fp16)[name = string("transpose_236")]; + tensor reshape_163_cast_fp16 = reshape(shape = concat_206, x = transpose_163_cast_fp16)[name = string("reshape_163_cast_fp16")]; + fp16 var_10214_to_fp16 = const()[name = string("op_10214_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_10215_cast_fp16 = mul(x = q_245_cast_fp16, y = var_10214_to_fp16)[name = string("op_10215_cast_fp16")]; + tensor transpose_477_perm_0 = const()[name = string("transpose_477_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_175_transpose_x_1 = const()[name = string("w_175_transpose_x_1"), val = bool(true)]; + bool w_175_transpose_y_1 = const()[name = string("w_175_transpose_y_1"), val = bool(false)]; + tensor transpose_477_cast_fp16 = transpose(perm = transpose_477_perm_0, x = reshape_161_cast_fp16)[name = string("transpose_235")]; + tensor w_175_cast_fp16 = matmul(transpose_x = w_175_transpose_x_1, transpose_y = w_175_transpose_y_1, x = var_10215_cast_fp16, y = transpose_477_cast_fp16)[name = string("w_175_cast_fp16")]; + tensor pad_81_to_fp16 = const()[name = string("pad_81_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640448)))]; + tensor var_10218_cast_fp16 = add(x = w_175_cast_fp16, y = pad_81_to_fp16)[name = string("op_10218_cast_fp16")]; + tensor w_177_cast_fp16 = softmax(axis = var_10093, x = var_10218_cast_fp16)[name = string("w_177_cast_fp16")]; + tensor transpose_478_perm_0 = const()[name = string("transpose_478_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_81_transpose_x_1 = const()[name = string("attn_81_transpose_x_1"), val = bool(false)]; + bool attn_81_transpose_y_1 = const()[name = string("attn_81_transpose_y_1"), val = bool(true)]; + tensor transpose_478_cast_fp16 = transpose(perm = transpose_478_perm_0, x = reshape_163_cast_fp16)[name = string("transpose_234")]; + tensor attn_81_cast_fp16 = matmul(transpose_x = attn_81_transpose_x_1, transpose_y = attn_81_transpose_y_1, x = transpose_478_cast_fp16, y = w_177_cast_fp16)[name = string("attn_81_cast_fp16")]; + tensor var_10222 = const()[name = string("op_10222"), val = tensor([1, 2048, 1, 1])]; + tensor input_431_cast_fp16 = reshape(shape = var_10222, x = attn_81_cast_fp16)[name = string("input_431_cast_fp16")]; + string attn_output_81_pad_type_0 = const()[name = string("attn_output_81_pad_type_0"), val = string("valid")]; + tensor attn_output_81_strides_0 = const()[name = string("attn_output_81_strides_0"), val = tensor([1, 1])]; + tensor attn_output_81_pad_0 = const()[name = string("attn_output_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_81_dilations_0 = const()[name = string("attn_output_81_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_81_groups_0 = const()[name = string("attn_output_81_groups_0"), val = int32(1)]; + tensor attn_output_81_cast_fp16 = conv(dilations = attn_output_81_dilations_0, groups = attn_output_81_groups_0, pad = attn_output_81_pad_0, pad_type = attn_output_81_pad_type_0, strides = attn_output_81_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_431_cast_fp16)[name = string("attn_output_81_cast_fp16")]; + tensor x_309_cast_fp16 = add(x = code_embed_27_cast_fp16, y = attn_output_81_cast_fp16)[name = string("x_309_cast_fp16")]; + tensor var_10236_cast_fp16 = mul(x = x_309_cast_fp16, y = x_309_cast_fp16)[name = string("op_10236_cast_fp16")]; + tensor variance_341_axes_0 = const()[name = string("variance_341_axes_0"), val = tensor([1])]; + bool variance_341_keep_dims_0 = const()[name = string("variance_341_keep_dims_0"), val = bool(true)]; + tensor variance_341_cast_fp16 = reduce_mean(axes = variance_341_axes_0, keep_dims = variance_341_keep_dims_0, x = var_10236_cast_fp16)[name = string("variance_341_cast_fp16")]; + fp16 var_10239_to_fp16 = const()[name = string("op_10239_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10240_cast_fp16 = add(x = variance_341_cast_fp16, y = var_10239_to_fp16)[name = string("op_10240_cast_fp16")]; + fp32 var_10241_epsilon_0 = const()[name = string("op_10241_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10241_cast_fp16 = rsqrt(epsilon = var_10241_epsilon_0, x = var_10240_cast_fp16)[name = string("op_10241_cast_fp16")]; + tensor var_10242_cast_fp16 = mul(x = x_309_cast_fp16, y = var_10241_cast_fp16)[name = string("op_10242_cast_fp16")]; + tensor input_433_cast_fp16 = mul(x = var_10242_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_433_cast_fp16")]; + string input_435_pad_type_0 = const()[name = string("input_435_pad_type_0"), val = string("valid")]; + tensor input_435_strides_0 = const()[name = string("input_435_strides_0"), val = tensor([1, 1])]; + tensor input_435_pad_0 = const()[name = string("input_435_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_435_dilations_0 = const()[name = string("input_435_dilations_0"), val = tensor([1, 1])]; + int32 input_435_groups_0 = const()[name = string("input_435_groups_0"), val = int32(1)]; + tensor input_435_cast_fp16 = conv(dilations = input_435_dilations_0, groups = input_435_groups_0, pad = input_435_pad_0, pad_type = input_435_pad_type_0, strides = input_435_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_433_cast_fp16)[name = string("input_435_cast_fp16")]; + tensor var_10250_cast_fp16 = silu(x = input_435_cast_fp16)[name = string("op_10250_cast_fp16")]; + string var_10256_pad_type_0 = const()[name = string("op_10256_pad_type_0"), val = string("valid")]; + tensor var_10256_strides_0 = const()[name = string("op_10256_strides_0"), val = tensor([1, 1])]; + tensor var_10256_pad_0 = const()[name = string("op_10256_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10256_dilations_0 = const()[name = string("op_10256_dilations_0"), val = tensor([1, 1])]; + int32 var_10256_groups_0 = const()[name = string("op_10256_groups_0"), val = int32(1)]; + tensor var_10256_cast_fp16 = conv(dilations = var_10256_dilations_0, groups = var_10256_groups_0, pad = var_10256_pad_0, pad_type = var_10256_pad_type_0, strides = var_10256_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_433_cast_fp16)[name = string("op_10256_cast_fp16")]; + tensor input_437_cast_fp16 = mul(x = var_10250_cast_fp16, y = var_10256_cast_fp16)[name = string("input_437_cast_fp16")]; + string h_81_pad_type_0 = const()[name = string("h_81_pad_type_0"), val = string("valid")]; + tensor h_81_strides_0 = const()[name = string("h_81_strides_0"), val = tensor([1, 1])]; + tensor h_81_pad_0 = const()[name = string("h_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_81_dilations_0 = const()[name = string("h_81_dilations_0"), val = tensor([1, 1])]; + int32 h_81_groups_0 = const()[name = string("h_81_groups_0"), val = int32(1)]; + tensor h_81_cast_fp16 = conv(dilations = h_81_dilations_0, groups = h_81_groups_0, pad = h_81_pad_0, pad_type = h_81_pad_type_0, strides = h_81_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_437_cast_fp16)[name = string("h_81_cast_fp16")]; + tensor x_311_cast_fp16 = add(x = x_309_cast_fp16, y = h_81_cast_fp16)[name = string("x_311_cast_fp16")]; + tensor key_cache_83_begin_0 = const()[name = string("key_cache_83_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_83_end_0 = const()[name = string("key_cache_83_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_83_end_mask_0 = const()[name = string("key_cache_83_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_83_cast_fp16 = slice_by_index(begin = key_cache_83_begin_0, end = key_cache_83_end_0, end_mask = key_cache_83_end_mask_0, x = layer_key_caches_17_cast_fp16)[name = string("key_cache_83_cast_fp16")]; + tensor value_cache_83_begin_0 = const()[name = string("value_cache_83_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_83_end_0 = const()[name = string("value_cache_83_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_83_end_mask_0 = const()[name = string("value_cache_83_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_83_cast_fp16 = slice_by_index(begin = value_cache_83_begin_0, end = value_cache_83_end_0, end_mask = value_cache_83_end_mask_0, x = layer_value_caches_17_cast_fp16)[name = string("value_cache_83_cast_fp16")]; + int32 var_10309 = const()[name = string("op_10309"), val = int32(2)]; + int32 var_10313 = const()[name = string("op_10313"), val = int32(3)]; + tensor var_10328_cast_fp16 = mul(x = x_311_cast_fp16, y = x_311_cast_fp16)[name = string("op_10328_cast_fp16")]; + tensor variance_343_axes_0 = const()[name = string("variance_343_axes_0"), val = tensor([1])]; + bool variance_343_keep_dims_0 = const()[name = string("variance_343_keep_dims_0"), val = bool(true)]; + tensor variance_343_cast_fp16 = reduce_mean(axes = variance_343_axes_0, keep_dims = variance_343_keep_dims_0, x = var_10328_cast_fp16)[name = string("variance_343_cast_fp16")]; + fp16 var_10331_to_fp16 = const()[name = string("op_10331_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10332_cast_fp16 = add(x = variance_343_cast_fp16, y = var_10331_to_fp16)[name = string("op_10332_cast_fp16")]; + fp32 var_10333_epsilon_0 = const()[name = string("op_10333_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10333_cast_fp16 = rsqrt(epsilon = var_10333_epsilon_0, x = var_10332_cast_fp16)[name = string("op_10333_cast_fp16")]; + tensor var_10334_cast_fp16 = mul(x = x_311_cast_fp16, y = var_10333_cast_fp16)[name = string("op_10334_cast_fp16")]; + tensor input_439_cast_fp16 = mul(x = var_10334_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_439_cast_fp16")]; + string q_247_pad_type_0 = const()[name = string("q_247_pad_type_0"), val = string("valid")]; + tensor q_247_strides_0 = const()[name = string("q_247_strides_0"), val = tensor([1, 1])]; + tensor q_247_pad_0 = const()[name = string("q_247_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_247_dilations_0 = const()[name = string("q_247_dilations_0"), val = tensor([1, 1])]; + int32 q_247_groups_0 = const()[name = string("q_247_groups_0"), val = int32(1)]; + tensor q_247_cast_fp16 = conv(dilations = q_247_dilations_0, groups = q_247_groups_0, pad = q_247_pad_0, pad_type = q_247_pad_type_0, strides = q_247_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_439_cast_fp16)[name = string("q_247_cast_fp16")]; + string k_247_pad_type_0 = const()[name = string("k_247_pad_type_0"), val = string("valid")]; + tensor k_247_strides_0 = const()[name = string("k_247_strides_0"), val = tensor([1, 1])]; + tensor k_247_pad_0 = const()[name = string("k_247_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_247_dilations_0 = const()[name = string("k_247_dilations_0"), val = tensor([1, 1])]; + int32 k_247_groups_0 = const()[name = string("k_247_groups_0"), val = int32(1)]; + tensor k_247_cast_fp16 = conv(dilations = k_247_dilations_0, groups = k_247_groups_0, pad = k_247_pad_0, pad_type = k_247_pad_type_0, strides = k_247_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_439_cast_fp16)[name = string("k_247_cast_fp16")]; + string v_83_pad_type_0 = const()[name = string("v_83_pad_type_0"), val = string("valid")]; + tensor v_83_strides_0 = const()[name = string("v_83_strides_0"), val = tensor([1, 1])]; + tensor v_83_pad_0 = const()[name = string("v_83_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_83_dilations_0 = const()[name = string("v_83_dilations_0"), val = tensor([1, 1])]; + int32 v_83_groups_0 = const()[name = string("v_83_groups_0"), val = int32(1)]; + tensor v_83_cast_fp16 = conv(dilations = v_83_dilations_0, groups = v_83_groups_0, pad = v_83_pad_0, pad_type = v_83_pad_type_0, strides = v_83_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_439_cast_fp16)[name = string("v_83_cast_fp16")]; + tensor var_10368 = const()[name = string("op_10368"), val = tensor([16, 128, 1, 1])]; + tensor x_313_cast_fp16 = reshape(shape = var_10368, x = q_247_cast_fp16)[name = string("x_313_cast_fp16")]; + tensor var_10371_cast_fp16 = mul(x = x_313_cast_fp16, y = x_313_cast_fp16)[name = string("op_10371_cast_fp16")]; + tensor variance_345_axes_0 = const()[name = string("variance_345_axes_0"), val = tensor([1])]; + bool variance_345_keep_dims_0 = const()[name = string("variance_345_keep_dims_0"), val = bool(true)]; + tensor variance_345_cast_fp16 = reduce_mean(axes = variance_345_axes_0, keep_dims = variance_345_keep_dims_0, x = var_10371_cast_fp16)[name = string("variance_345_cast_fp16")]; + fp16 var_10374_to_fp16 = const()[name = string("op_10374_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10375_cast_fp16 = add(x = variance_345_cast_fp16, y = var_10374_to_fp16)[name = string("op_10375_cast_fp16")]; + fp32 var_10376_epsilon_0 = const()[name = string("op_10376_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10376_cast_fp16 = rsqrt(epsilon = var_10376_epsilon_0, x = var_10375_cast_fp16)[name = string("op_10376_cast_fp16")]; + tensor var_10377_cast_fp16 = mul(x = x_313_cast_fp16, y = var_10376_cast_fp16)[name = string("op_10377_cast_fp16")]; + tensor q_249_cast_fp16 = mul(x = var_10377_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_249_cast_fp16")]; + tensor var_10379 = const()[name = string("op_10379"), val = tensor([8, 128, 1, 1])]; + tensor x_315_cast_fp16 = reshape(shape = var_10379, x = k_247_cast_fp16)[name = string("x_315_cast_fp16")]; + tensor var_10382_cast_fp16 = mul(x = x_315_cast_fp16, y = x_315_cast_fp16)[name = string("op_10382_cast_fp16")]; + tensor variance_347_axes_0 = const()[name = string("variance_347_axes_0"), val = tensor([1])]; + bool variance_347_keep_dims_0 = const()[name = string("variance_347_keep_dims_0"), val = bool(true)]; + tensor variance_347_cast_fp16 = reduce_mean(axes = variance_347_axes_0, keep_dims = variance_347_keep_dims_0, x = var_10382_cast_fp16)[name = string("variance_347_cast_fp16")]; + fp16 var_10385_to_fp16 = const()[name = string("op_10385_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10386_cast_fp16 = add(x = variance_347_cast_fp16, y = var_10385_to_fp16)[name = string("op_10386_cast_fp16")]; + fp32 var_10387_epsilon_0 = const()[name = string("op_10387_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10387_cast_fp16 = rsqrt(epsilon = var_10387_epsilon_0, x = var_10386_cast_fp16)[name = string("op_10387_cast_fp16")]; + tensor var_10388_cast_fp16 = mul(x = x_315_cast_fp16, y = var_10387_cast_fp16)[name = string("op_10388_cast_fp16")]; + tensor k_249_cast_fp16 = mul(x = var_10388_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_249_cast_fp16")]; + tensor var_10390 = const()[name = string("op_10390"), val = tensor([1, 16, 128, 1])]; + tensor z_165_cast_fp16 = reshape(shape = var_10390, x = q_249_cast_fp16)[name = string("z_165_cast_fp16")]; + tensor var_10392 = const()[name = string("op_10392"), val = tensor([1, 8, 128, 1])]; + tensor z_167_cast_fp16 = reshape(shape = var_10392, x = k_249_cast_fp16)[name = string("z_167_cast_fp16")]; + tensor z1_165_begin_0 = const()[name = string("z1_165_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_165_end_0 = const()[name = string("z1_165_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_165_end_mask_0 = const()[name = string("z1_165_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_165_cast_fp16 = slice_by_index(begin = z1_165_begin_0, end = z1_165_end_0, end_mask = z1_165_end_mask_0, x = z_165_cast_fp16)[name = string("z1_165_cast_fp16")]; + tensor z2_165_begin_0 = const()[name = string("z2_165_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_165_end_0 = const()[name = string("z2_165_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_165_end_mask_0 = const()[name = string("z2_165_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_165_cast_fp16 = slice_by_index(begin = z2_165_begin_0, end = z2_165_end_0, end_mask = z2_165_end_mask_0, x = z_165_cast_fp16)[name = string("z2_165_cast_fp16")]; + tensor var_10400_cast_fp16 = mul(x = z_165_cast_fp16, y = cos_81_to_fp16)[name = string("op_10400_cast_fp16")]; + fp16 const_91_promoted_to_fp16 = const()[name = string("const_91_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10401_cast_fp16 = mul(x = z2_165_cast_fp16, y = const_91_promoted_to_fp16)[name = string("op_10401_cast_fp16")]; + bool var_10403_interleave_0 = const()[name = string("op_10403_interleave_0"), val = bool(false)]; + tensor var_10403_cast_fp16 = concat(axis = var_10309, interleave = var_10403_interleave_0, values = (var_10401_cast_fp16, z1_165_cast_fp16))[name = string("op_10403_cast_fp16")]; + tensor var_10404_cast_fp16 = mul(x = var_10403_cast_fp16, y = sin_81_to_fp16)[name = string("op_10404_cast_fp16")]; + tensor q_251_cast_fp16 = add(x = var_10400_cast_fp16, y = var_10404_cast_fp16)[name = string("q_251_cast_fp16")]; + tensor z1_167_begin_0 = const()[name = string("z1_167_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_167_end_0 = const()[name = string("z1_167_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_167_end_mask_0 = const()[name = string("z1_167_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_167_cast_fp16 = slice_by_index(begin = z1_167_begin_0, end = z1_167_end_0, end_mask = z1_167_end_mask_0, x = z_167_cast_fp16)[name = string("z1_167_cast_fp16")]; + tensor z2_167_begin_0 = const()[name = string("z2_167_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_167_end_0 = const()[name = string("z2_167_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_167_end_mask_0 = const()[name = string("z2_167_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_167_cast_fp16 = slice_by_index(begin = z2_167_begin_0, end = z2_167_end_0, end_mask = z2_167_end_mask_0, x = z_167_cast_fp16)[name = string("z2_167_cast_fp16")]; + tensor var_10412_cast_fp16 = mul(x = z_167_cast_fp16, y = cos_81_to_fp16)[name = string("op_10412_cast_fp16")]; + fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10413_cast_fp16 = mul(x = z2_167_cast_fp16, y = const_92_promoted_to_fp16)[name = string("op_10413_cast_fp16")]; + bool var_10415_interleave_0 = const()[name = string("op_10415_interleave_0"), val = bool(false)]; + tensor var_10415_cast_fp16 = concat(axis = var_10309, interleave = var_10415_interleave_0, values = (var_10413_cast_fp16, z1_167_cast_fp16))[name = string("op_10415_cast_fp16")]; + tensor var_10416_cast_fp16 = mul(x = var_10415_cast_fp16, y = sin_81_to_fp16)[name = string("op_10416_cast_fp16")]; + tensor k_251_cast_fp16 = add(x = var_10412_cast_fp16, y = var_10416_cast_fp16)[name = string("k_251_cast_fp16")]; + tensor var_10418 = const()[name = string("op_10418"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_83_cast_fp16 = reshape(shape = var_10418, x = k_251_cast_fp16)[name = string("cur_key_83_cast_fp16")]; + tensor var_10420_to_fp16 = const()[name = string("op_10420_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640192)))]; + tensor var_10421_cast_fp16 = mul(x = key_cache_83_cast_fp16, y = var_10420_to_fp16)[name = string("op_10421_cast_fp16")]; + tensor upd_83_to_fp16 = const()[name = string("upd_83_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640320)))]; + tensor var_10422_cast_fp16 = mul(x = cur_key_83_cast_fp16, y = upd_83_to_fp16)[name = string("op_10422_cast_fp16")]; + tensor key_83_cast_fp16 = add(x = var_10421_cast_fp16, y = var_10422_cast_fp16)[name = string("key_83_cast_fp16")]; + tensor var_10424_to_fp16 = const()[name = string("op_10424_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640192)))]; + tensor var_10425_cast_fp16 = mul(x = value_cache_83_cast_fp16, y = var_10424_to_fp16)[name = string("op_10425_cast_fp16")]; + tensor var_10426_cast_fp16 = mul(x = v_83_cast_fp16, y = upd_83_to_fp16)[name = string("op_10426_cast_fp16")]; + tensor value_83_cast_fp16 = add(x = var_10425_cast_fp16, y = var_10426_cast_fp16)[name = string("value_83_cast_fp16")]; + tensor var_10428 = const()[name = string("op_10428"), val = tensor([1, 8, 128, 16])]; + tensor kh_165_cast_fp16 = reshape(shape = var_10428, x = key_83_cast_fp16)[name = string("kh_165_cast_fp16")]; + tensor var_10430 = const()[name = string("op_10430"), val = tensor([1, 8, 128, 16])]; + tensor vh_165_cast_fp16 = reshape(shape = var_10430, x = value_83_cast_fp16)[name = string("vh_165_cast_fp16")]; + tensor transpose_164_perm_0 = const()[name = string("transpose_164_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_82_reps_0 = const()[name = string("tile_82_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_164_cast_fp16 = transpose(perm = transpose_164_perm_0, x = kh_165_cast_fp16)[name = string("transpose_233")]; + tensor tile_82_cast_fp16 = tile(reps = tile_82_reps_0, x = transpose_164_cast_fp16)[name = string("tile_82_cast_fp16")]; + tensor concat_207 = const()[name = string("concat_207"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_164_cast_fp16 = reshape(shape = concat_207, x = tile_82_cast_fp16)[name = string("reshape_164_cast_fp16")]; + tensor transpose_165_perm_0 = const()[name = string("transpose_165_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_208 = const()[name = string("concat_208"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_165_cast_fp16 = transpose(perm = transpose_165_perm_0, x = reshape_164_cast_fp16)[name = string("transpose_232")]; + tensor reshape_165_cast_fp16 = reshape(shape = concat_208, x = transpose_165_cast_fp16)[name = string("reshape_165_cast_fp16")]; + tensor transpose_166_perm_0 = const()[name = string("transpose_166_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_83_reps_0 = const()[name = string("tile_83_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_166_cast_fp16 = transpose(perm = transpose_166_perm_0, x = vh_165_cast_fp16)[name = string("transpose_231")]; + tensor tile_83_cast_fp16 = tile(reps = tile_83_reps_0, x = transpose_166_cast_fp16)[name = string("tile_83_cast_fp16")]; + tensor concat_209 = const()[name = string("concat_209"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_166_cast_fp16 = reshape(shape = concat_209, x = tile_83_cast_fp16)[name = string("reshape_166_cast_fp16")]; + tensor transpose_167_perm_0 = const()[name = string("transpose_167_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_210 = const()[name = string("concat_210"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_167_cast_fp16 = transpose(perm = transpose_167_perm_0, x = reshape_166_cast_fp16)[name = string("transpose_230")]; + tensor reshape_167_cast_fp16 = reshape(shape = concat_210, x = transpose_167_cast_fp16)[name = string("reshape_167_cast_fp16")]; + fp16 var_10434_to_fp16 = const()[name = string("op_10434_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_10435_cast_fp16 = mul(x = q_251_cast_fp16, y = var_10434_to_fp16)[name = string("op_10435_cast_fp16")]; + tensor transpose_481_perm_0 = const()[name = string("transpose_481_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_179_transpose_x_1 = const()[name = string("w_179_transpose_x_1"), val = bool(true)]; + bool w_179_transpose_y_1 = const()[name = string("w_179_transpose_y_1"), val = bool(false)]; + tensor transpose_481_cast_fp16 = transpose(perm = transpose_481_perm_0, x = reshape_165_cast_fp16)[name = string("transpose_229")]; + tensor w_179_cast_fp16 = matmul(transpose_x = w_179_transpose_x_1, transpose_y = w_179_transpose_y_1, x = var_10435_cast_fp16, y = transpose_481_cast_fp16)[name = string("w_179_cast_fp16")]; + tensor pad_83_to_fp16 = const()[name = string("pad_83_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640448)))]; + tensor var_10438_cast_fp16 = add(x = w_179_cast_fp16, y = pad_83_to_fp16)[name = string("op_10438_cast_fp16")]; + tensor w_181_cast_fp16 = softmax(axis = var_10313, x = var_10438_cast_fp16)[name = string("w_181_cast_fp16")]; + tensor transpose_482_perm_0 = const()[name = string("transpose_482_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_83_transpose_x_1 = const()[name = string("attn_83_transpose_x_1"), val = bool(false)]; + bool attn_83_transpose_y_1 = const()[name = string("attn_83_transpose_y_1"), val = bool(true)]; + tensor transpose_482_cast_fp16 = transpose(perm = transpose_482_perm_0, x = reshape_167_cast_fp16)[name = string("transpose_228")]; + tensor attn_83_cast_fp16 = matmul(transpose_x = attn_83_transpose_x_1, transpose_y = attn_83_transpose_y_1, x = transpose_482_cast_fp16, y = w_181_cast_fp16)[name = string("attn_83_cast_fp16")]; + tensor var_10442 = const()[name = string("op_10442"), val = tensor([1, 2048, 1, 1])]; + tensor input_441_cast_fp16 = reshape(shape = var_10442, x = attn_83_cast_fp16)[name = string("input_441_cast_fp16")]; + string attn_output_83_pad_type_0 = const()[name = string("attn_output_83_pad_type_0"), val = string("valid")]; + tensor attn_output_83_strides_0 = const()[name = string("attn_output_83_strides_0"), val = tensor([1, 1])]; + tensor attn_output_83_pad_0 = const()[name = string("attn_output_83_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_83_dilations_0 = const()[name = string("attn_output_83_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_83_groups_0 = const()[name = string("attn_output_83_groups_0"), val = int32(1)]; + tensor attn_output_83_cast_fp16 = conv(dilations = attn_output_83_dilations_0, groups = attn_output_83_groups_0, pad = attn_output_83_pad_0, pad_type = attn_output_83_pad_type_0, strides = attn_output_83_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_441_cast_fp16)[name = string("attn_output_83_cast_fp16")]; + tensor x_317_cast_fp16 = add(x = x_311_cast_fp16, y = attn_output_83_cast_fp16)[name = string("x_317_cast_fp16")]; + tensor var_10456_cast_fp16 = mul(x = x_317_cast_fp16, y = x_317_cast_fp16)[name = string("op_10456_cast_fp16")]; + tensor variance_349_axes_0 = const()[name = string("variance_349_axes_0"), val = tensor([1])]; + bool variance_349_keep_dims_0 = const()[name = string("variance_349_keep_dims_0"), val = bool(true)]; + tensor variance_349_cast_fp16 = reduce_mean(axes = variance_349_axes_0, keep_dims = variance_349_keep_dims_0, x = var_10456_cast_fp16)[name = string("variance_349_cast_fp16")]; + fp16 var_10459_to_fp16 = const()[name = string("op_10459_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10460_cast_fp16 = add(x = variance_349_cast_fp16, y = var_10459_to_fp16)[name = string("op_10460_cast_fp16")]; + fp32 var_10461_epsilon_0 = const()[name = string("op_10461_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10461_cast_fp16 = rsqrt(epsilon = var_10461_epsilon_0, x = var_10460_cast_fp16)[name = string("op_10461_cast_fp16")]; + tensor var_10462_cast_fp16 = mul(x = x_317_cast_fp16, y = var_10461_cast_fp16)[name = string("op_10462_cast_fp16")]; + tensor input_443_cast_fp16 = mul(x = var_10462_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_443_cast_fp16")]; + string input_445_pad_type_0 = const()[name = string("input_445_pad_type_0"), val = string("valid")]; + tensor input_445_strides_0 = const()[name = string("input_445_strides_0"), val = tensor([1, 1])]; + tensor input_445_pad_0 = const()[name = string("input_445_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_445_dilations_0 = const()[name = string("input_445_dilations_0"), val = tensor([1, 1])]; + int32 input_445_groups_0 = const()[name = string("input_445_groups_0"), val = int32(1)]; + tensor input_445_cast_fp16 = conv(dilations = input_445_dilations_0, groups = input_445_groups_0, pad = input_445_pad_0, pad_type = input_445_pad_type_0, strides = input_445_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_443_cast_fp16)[name = string("input_445_cast_fp16")]; + tensor var_10470_cast_fp16 = silu(x = input_445_cast_fp16)[name = string("op_10470_cast_fp16")]; + string var_10476_pad_type_0 = const()[name = string("op_10476_pad_type_0"), val = string("valid")]; + tensor var_10476_strides_0 = const()[name = string("op_10476_strides_0"), val = tensor([1, 1])]; + tensor var_10476_pad_0 = const()[name = string("op_10476_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10476_dilations_0 = const()[name = string("op_10476_dilations_0"), val = tensor([1, 1])]; + int32 var_10476_groups_0 = const()[name = string("op_10476_groups_0"), val = int32(1)]; + tensor var_10476_cast_fp16 = conv(dilations = var_10476_dilations_0, groups = var_10476_groups_0, pad = var_10476_pad_0, pad_type = var_10476_pad_type_0, strides = var_10476_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_443_cast_fp16)[name = string("op_10476_cast_fp16")]; + tensor input_447_cast_fp16 = mul(x = var_10470_cast_fp16, y = var_10476_cast_fp16)[name = string("input_447_cast_fp16")]; + string h_83_pad_type_0 = const()[name = string("h_83_pad_type_0"), val = string("valid")]; + tensor h_83_strides_0 = const()[name = string("h_83_strides_0"), val = tensor([1, 1])]; + tensor h_83_pad_0 = const()[name = string("h_83_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_83_dilations_0 = const()[name = string("h_83_dilations_0"), val = tensor([1, 1])]; + int32 h_83_groups_0 = const()[name = string("h_83_groups_0"), val = int32(1)]; + tensor h_83_cast_fp16 = conv(dilations = h_83_dilations_0, groups = h_83_groups_0, pad = h_83_pad_0, pad_type = h_83_pad_type_0, strides = h_83_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_447_cast_fp16)[name = string("h_83_cast_fp16")]; + tensor x_319_cast_fp16 = add(x = x_317_cast_fp16, y = h_83_cast_fp16)[name = string("x_319_cast_fp16")]; + tensor key_cache_85_begin_0 = const()[name = string("key_cache_85_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_85_end_0 = const()[name = string("key_cache_85_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_85_end_mask_0 = const()[name = string("key_cache_85_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_85_cast_fp16 = slice_by_index(begin = key_cache_85_begin_0, end = key_cache_85_end_0, end_mask = key_cache_85_end_mask_0, x = layer_key_caches_17_cast_fp16)[name = string("key_cache_85_cast_fp16")]; + tensor value_cache_85_begin_0 = const()[name = string("value_cache_85_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_85_end_0 = const()[name = string("value_cache_85_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_85_end_mask_0 = const()[name = string("value_cache_85_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_85_cast_fp16 = slice_by_index(begin = value_cache_85_begin_0, end = value_cache_85_end_0, end_mask = value_cache_85_end_mask_0, x = layer_value_caches_17_cast_fp16)[name = string("value_cache_85_cast_fp16")]; + int32 var_10529 = const()[name = string("op_10529"), val = int32(2)]; + int32 var_10533 = const()[name = string("op_10533"), val = int32(3)]; + tensor var_10548_cast_fp16 = mul(x = x_319_cast_fp16, y = x_319_cast_fp16)[name = string("op_10548_cast_fp16")]; + tensor variance_351_axes_0 = const()[name = string("variance_351_axes_0"), val = tensor([1])]; + bool variance_351_keep_dims_0 = const()[name = string("variance_351_keep_dims_0"), val = bool(true)]; + tensor variance_351_cast_fp16 = reduce_mean(axes = variance_351_axes_0, keep_dims = variance_351_keep_dims_0, x = var_10548_cast_fp16)[name = string("variance_351_cast_fp16")]; + fp16 var_10551_to_fp16 = const()[name = string("op_10551_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10552_cast_fp16 = add(x = variance_351_cast_fp16, y = var_10551_to_fp16)[name = string("op_10552_cast_fp16")]; + fp32 var_10553_epsilon_0 = const()[name = string("op_10553_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10553_cast_fp16 = rsqrt(epsilon = var_10553_epsilon_0, x = var_10552_cast_fp16)[name = string("op_10553_cast_fp16")]; + tensor var_10554_cast_fp16 = mul(x = x_319_cast_fp16, y = var_10553_cast_fp16)[name = string("op_10554_cast_fp16")]; + tensor input_449_cast_fp16 = mul(x = var_10554_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_449_cast_fp16")]; + string q_253_pad_type_0 = const()[name = string("q_253_pad_type_0"), val = string("valid")]; + tensor q_253_strides_0 = const()[name = string("q_253_strides_0"), val = tensor([1, 1])]; + tensor q_253_pad_0 = const()[name = string("q_253_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_253_dilations_0 = const()[name = string("q_253_dilations_0"), val = tensor([1, 1])]; + int32 q_253_groups_0 = const()[name = string("q_253_groups_0"), val = int32(1)]; + tensor q_253_cast_fp16 = conv(dilations = q_253_dilations_0, groups = q_253_groups_0, pad = q_253_pad_0, pad_type = q_253_pad_type_0, strides = q_253_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_449_cast_fp16)[name = string("q_253_cast_fp16")]; + string k_253_pad_type_0 = const()[name = string("k_253_pad_type_0"), val = string("valid")]; + tensor k_253_strides_0 = const()[name = string("k_253_strides_0"), val = tensor([1, 1])]; + tensor k_253_pad_0 = const()[name = string("k_253_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_253_dilations_0 = const()[name = string("k_253_dilations_0"), val = tensor([1, 1])]; + int32 k_253_groups_0 = const()[name = string("k_253_groups_0"), val = int32(1)]; + tensor k_253_cast_fp16 = conv(dilations = k_253_dilations_0, groups = k_253_groups_0, pad = k_253_pad_0, pad_type = k_253_pad_type_0, strides = k_253_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_449_cast_fp16)[name = string("k_253_cast_fp16")]; + string v_85_pad_type_0 = const()[name = string("v_85_pad_type_0"), val = string("valid")]; + tensor v_85_strides_0 = const()[name = string("v_85_strides_0"), val = tensor([1, 1])]; + tensor v_85_pad_0 = const()[name = string("v_85_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_85_dilations_0 = const()[name = string("v_85_dilations_0"), val = tensor([1, 1])]; + int32 v_85_groups_0 = const()[name = string("v_85_groups_0"), val = int32(1)]; + tensor v_85_cast_fp16 = conv(dilations = v_85_dilations_0, groups = v_85_groups_0, pad = v_85_pad_0, pad_type = v_85_pad_type_0, strides = v_85_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_449_cast_fp16)[name = string("v_85_cast_fp16")]; + tensor var_10588 = const()[name = string("op_10588"), val = tensor([16, 128, 1, 1])]; + tensor x_321_cast_fp16 = reshape(shape = var_10588, x = q_253_cast_fp16)[name = string("x_321_cast_fp16")]; + tensor var_10591_cast_fp16 = mul(x = x_321_cast_fp16, y = x_321_cast_fp16)[name = string("op_10591_cast_fp16")]; + tensor variance_353_axes_0 = const()[name = string("variance_353_axes_0"), val = tensor([1])]; + bool variance_353_keep_dims_0 = const()[name = string("variance_353_keep_dims_0"), val = bool(true)]; + tensor variance_353_cast_fp16 = reduce_mean(axes = variance_353_axes_0, keep_dims = variance_353_keep_dims_0, x = var_10591_cast_fp16)[name = string("variance_353_cast_fp16")]; + fp16 var_10594_to_fp16 = const()[name = string("op_10594_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10595_cast_fp16 = add(x = variance_353_cast_fp16, y = var_10594_to_fp16)[name = string("op_10595_cast_fp16")]; + fp32 var_10596_epsilon_0 = const()[name = string("op_10596_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10596_cast_fp16 = rsqrt(epsilon = var_10596_epsilon_0, x = var_10595_cast_fp16)[name = string("op_10596_cast_fp16")]; + tensor var_10597_cast_fp16 = mul(x = x_321_cast_fp16, y = var_10596_cast_fp16)[name = string("op_10597_cast_fp16")]; + tensor q_255_cast_fp16 = mul(x = var_10597_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_255_cast_fp16")]; + tensor var_10599 = const()[name = string("op_10599"), val = tensor([8, 128, 1, 1])]; + tensor x_323_cast_fp16 = reshape(shape = var_10599, x = k_253_cast_fp16)[name = string("x_323_cast_fp16")]; + tensor var_10602_cast_fp16 = mul(x = x_323_cast_fp16, y = x_323_cast_fp16)[name = string("op_10602_cast_fp16")]; + tensor variance_355_axes_0 = const()[name = string("variance_355_axes_0"), val = tensor([1])]; + bool variance_355_keep_dims_0 = const()[name = string("variance_355_keep_dims_0"), val = bool(true)]; + tensor variance_355_cast_fp16 = reduce_mean(axes = variance_355_axes_0, keep_dims = variance_355_keep_dims_0, x = var_10602_cast_fp16)[name = string("variance_355_cast_fp16")]; + fp16 var_10605_to_fp16 = const()[name = string("op_10605_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10606_cast_fp16 = add(x = variance_355_cast_fp16, y = var_10605_to_fp16)[name = string("op_10606_cast_fp16")]; + fp32 var_10607_epsilon_0 = const()[name = string("op_10607_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10607_cast_fp16 = rsqrt(epsilon = var_10607_epsilon_0, x = var_10606_cast_fp16)[name = string("op_10607_cast_fp16")]; + tensor var_10608_cast_fp16 = mul(x = x_323_cast_fp16, y = var_10607_cast_fp16)[name = string("op_10608_cast_fp16")]; + tensor k_255_cast_fp16 = mul(x = var_10608_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_255_cast_fp16")]; + tensor var_10610 = const()[name = string("op_10610"), val = tensor([1, 16, 128, 1])]; + tensor z_169_cast_fp16 = reshape(shape = var_10610, x = q_255_cast_fp16)[name = string("z_169_cast_fp16")]; + tensor var_10612 = const()[name = string("op_10612"), val = tensor([1, 8, 128, 1])]; + tensor z_171_cast_fp16 = reshape(shape = var_10612, x = k_255_cast_fp16)[name = string("z_171_cast_fp16")]; + tensor z1_169_begin_0 = const()[name = string("z1_169_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_169_end_0 = const()[name = string("z1_169_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_169_end_mask_0 = const()[name = string("z1_169_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_169_cast_fp16 = slice_by_index(begin = z1_169_begin_0, end = z1_169_end_0, end_mask = z1_169_end_mask_0, x = z_169_cast_fp16)[name = string("z1_169_cast_fp16")]; + tensor z2_169_begin_0 = const()[name = string("z2_169_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_169_end_0 = const()[name = string("z2_169_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_169_end_mask_0 = const()[name = string("z2_169_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_169_cast_fp16 = slice_by_index(begin = z2_169_begin_0, end = z2_169_end_0, end_mask = z2_169_end_mask_0, x = z_169_cast_fp16)[name = string("z2_169_cast_fp16")]; + tensor var_10620_cast_fp16 = mul(x = z_169_cast_fp16, y = cos_81_to_fp16)[name = string("op_10620_cast_fp16")]; + fp16 const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10621_cast_fp16 = mul(x = z2_169_cast_fp16, y = const_93_promoted_to_fp16)[name = string("op_10621_cast_fp16")]; + bool var_10623_interleave_0 = const()[name = string("op_10623_interleave_0"), val = bool(false)]; + tensor var_10623_cast_fp16 = concat(axis = var_10529, interleave = var_10623_interleave_0, values = (var_10621_cast_fp16, z1_169_cast_fp16))[name = string("op_10623_cast_fp16")]; + tensor var_10624_cast_fp16 = mul(x = var_10623_cast_fp16, y = sin_81_to_fp16)[name = string("op_10624_cast_fp16")]; + tensor q_257_cast_fp16 = add(x = var_10620_cast_fp16, y = var_10624_cast_fp16)[name = string("q_257_cast_fp16")]; + tensor z1_171_begin_0 = const()[name = string("z1_171_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_171_end_0 = const()[name = string("z1_171_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_171_end_mask_0 = const()[name = string("z1_171_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_171_cast_fp16 = slice_by_index(begin = z1_171_begin_0, end = z1_171_end_0, end_mask = z1_171_end_mask_0, x = z_171_cast_fp16)[name = string("z1_171_cast_fp16")]; + tensor z2_171_begin_0 = const()[name = string("z2_171_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_171_end_0 = const()[name = string("z2_171_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_171_end_mask_0 = const()[name = string("z2_171_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_171_cast_fp16 = slice_by_index(begin = z2_171_begin_0, end = z2_171_end_0, end_mask = z2_171_end_mask_0, x = z_171_cast_fp16)[name = string("z2_171_cast_fp16")]; + tensor var_10632_cast_fp16 = mul(x = z_171_cast_fp16, y = cos_81_to_fp16)[name = string("op_10632_cast_fp16")]; + fp16 const_94_promoted_to_fp16 = const()[name = string("const_94_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10633_cast_fp16 = mul(x = z2_171_cast_fp16, y = const_94_promoted_to_fp16)[name = string("op_10633_cast_fp16")]; + bool var_10635_interleave_0 = const()[name = string("op_10635_interleave_0"), val = bool(false)]; + tensor var_10635_cast_fp16 = concat(axis = var_10529, interleave = var_10635_interleave_0, values = (var_10633_cast_fp16, z1_171_cast_fp16))[name = string("op_10635_cast_fp16")]; + tensor var_10636_cast_fp16 = mul(x = var_10635_cast_fp16, y = sin_81_to_fp16)[name = string("op_10636_cast_fp16")]; + tensor k_257_cast_fp16 = add(x = var_10632_cast_fp16, y = var_10636_cast_fp16)[name = string("k_257_cast_fp16")]; + tensor var_10638 = const()[name = string("op_10638"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_85_cast_fp16 = reshape(shape = var_10638, x = k_257_cast_fp16)[name = string("cur_key_85_cast_fp16")]; + tensor var_10640_to_fp16 = const()[name = string("op_10640_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640192)))]; + tensor var_10641_cast_fp16 = mul(x = key_cache_85_cast_fp16, y = var_10640_to_fp16)[name = string("op_10641_cast_fp16")]; + tensor upd_85_to_fp16 = const()[name = string("upd_85_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640320)))]; + tensor var_10642_cast_fp16 = mul(x = cur_key_85_cast_fp16, y = upd_85_to_fp16)[name = string("op_10642_cast_fp16")]; + tensor key_85_cast_fp16 = add(x = var_10641_cast_fp16, y = var_10642_cast_fp16)[name = string("key_85_cast_fp16")]; + tensor var_10644_to_fp16 = const()[name = string("op_10644_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640192)))]; + tensor var_10645_cast_fp16 = mul(x = value_cache_85_cast_fp16, y = var_10644_to_fp16)[name = string("op_10645_cast_fp16")]; + tensor var_10646_cast_fp16 = mul(x = v_85_cast_fp16, y = upd_85_to_fp16)[name = string("op_10646_cast_fp16")]; + tensor value_85_cast_fp16 = add(x = var_10645_cast_fp16, y = var_10646_cast_fp16)[name = string("value_85_cast_fp16")]; + tensor var_10648 = const()[name = string("op_10648"), val = tensor([1, 8, 128, 16])]; + tensor kh_169_cast_fp16 = reshape(shape = var_10648, x = key_85_cast_fp16)[name = string("kh_169_cast_fp16")]; + tensor var_10650 = const()[name = string("op_10650"), val = tensor([1, 8, 128, 16])]; + tensor vh_169_cast_fp16 = reshape(shape = var_10650, x = value_85_cast_fp16)[name = string("vh_169_cast_fp16")]; + tensor transpose_168_perm_0 = const()[name = string("transpose_168_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_84_reps_0 = const()[name = string("tile_84_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_168_cast_fp16 = transpose(perm = transpose_168_perm_0, x = kh_169_cast_fp16)[name = string("transpose_227")]; + tensor tile_84_cast_fp16 = tile(reps = tile_84_reps_0, x = transpose_168_cast_fp16)[name = string("tile_84_cast_fp16")]; + tensor concat_211 = const()[name = string("concat_211"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_168_cast_fp16 = reshape(shape = concat_211, x = tile_84_cast_fp16)[name = string("reshape_168_cast_fp16")]; + tensor transpose_169_perm_0 = const()[name = string("transpose_169_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_212 = const()[name = string("concat_212"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_169_cast_fp16 = transpose(perm = transpose_169_perm_0, x = reshape_168_cast_fp16)[name = string("transpose_226")]; + tensor reshape_169_cast_fp16 = reshape(shape = concat_212, x = transpose_169_cast_fp16)[name = string("reshape_169_cast_fp16")]; + tensor transpose_170_perm_0 = const()[name = string("transpose_170_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_85_reps_0 = const()[name = string("tile_85_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_170_cast_fp16 = transpose(perm = transpose_170_perm_0, x = vh_169_cast_fp16)[name = string("transpose_225")]; + tensor tile_85_cast_fp16 = tile(reps = tile_85_reps_0, x = transpose_170_cast_fp16)[name = string("tile_85_cast_fp16")]; + tensor concat_213 = const()[name = string("concat_213"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_170_cast_fp16 = reshape(shape = concat_213, x = tile_85_cast_fp16)[name = string("reshape_170_cast_fp16")]; + tensor transpose_171_perm_0 = const()[name = string("transpose_171_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_214 = const()[name = string("concat_214"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_171_cast_fp16 = transpose(perm = transpose_171_perm_0, x = reshape_170_cast_fp16)[name = string("transpose_224")]; + tensor reshape_171_cast_fp16 = reshape(shape = concat_214, x = transpose_171_cast_fp16)[name = string("reshape_171_cast_fp16")]; + fp16 var_10654_to_fp16 = const()[name = string("op_10654_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_10655_cast_fp16 = mul(x = q_257_cast_fp16, y = var_10654_to_fp16)[name = string("op_10655_cast_fp16")]; + tensor transpose_485_perm_0 = const()[name = string("transpose_485_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_183_transpose_x_1 = const()[name = string("w_183_transpose_x_1"), val = bool(true)]; + bool w_183_transpose_y_1 = const()[name = string("w_183_transpose_y_1"), val = bool(false)]; + tensor transpose_485_cast_fp16 = transpose(perm = transpose_485_perm_0, x = reshape_169_cast_fp16)[name = string("transpose_223")]; + tensor w_183_cast_fp16 = matmul(transpose_x = w_183_transpose_x_1, transpose_y = w_183_transpose_y_1, x = var_10655_cast_fp16, y = transpose_485_cast_fp16)[name = string("w_183_cast_fp16")]; + tensor pad_85_to_fp16 = const()[name = string("pad_85_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640448)))]; + tensor var_10658_cast_fp16 = add(x = w_183_cast_fp16, y = pad_85_to_fp16)[name = string("op_10658_cast_fp16")]; + tensor w_185_cast_fp16 = softmax(axis = var_10533, x = var_10658_cast_fp16)[name = string("w_185_cast_fp16")]; + tensor transpose_486_perm_0 = const()[name = string("transpose_486_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_85_transpose_x_1 = const()[name = string("attn_85_transpose_x_1"), val = bool(false)]; + bool attn_85_transpose_y_1 = const()[name = string("attn_85_transpose_y_1"), val = bool(true)]; + tensor transpose_486_cast_fp16 = transpose(perm = transpose_486_perm_0, x = reshape_171_cast_fp16)[name = string("transpose_222")]; + tensor attn_85_cast_fp16 = matmul(transpose_x = attn_85_transpose_x_1, transpose_y = attn_85_transpose_y_1, x = transpose_486_cast_fp16, y = w_185_cast_fp16)[name = string("attn_85_cast_fp16")]; + tensor var_10662 = const()[name = string("op_10662"), val = tensor([1, 2048, 1, 1])]; + tensor input_451_cast_fp16 = reshape(shape = var_10662, x = attn_85_cast_fp16)[name = string("input_451_cast_fp16")]; + string attn_output_85_pad_type_0 = const()[name = string("attn_output_85_pad_type_0"), val = string("valid")]; + tensor attn_output_85_strides_0 = const()[name = string("attn_output_85_strides_0"), val = tensor([1, 1])]; + tensor attn_output_85_pad_0 = const()[name = string("attn_output_85_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_85_dilations_0 = const()[name = string("attn_output_85_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_85_groups_0 = const()[name = string("attn_output_85_groups_0"), val = int32(1)]; + tensor attn_output_85_cast_fp16 = conv(dilations = attn_output_85_dilations_0, groups = attn_output_85_groups_0, pad = attn_output_85_pad_0, pad_type = attn_output_85_pad_type_0, strides = attn_output_85_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_451_cast_fp16)[name = string("attn_output_85_cast_fp16")]; + tensor x_325_cast_fp16 = add(x = x_319_cast_fp16, y = attn_output_85_cast_fp16)[name = string("x_325_cast_fp16")]; + tensor var_10676_cast_fp16 = mul(x = x_325_cast_fp16, y = x_325_cast_fp16)[name = string("op_10676_cast_fp16")]; + tensor variance_357_axes_0 = const()[name = string("variance_357_axes_0"), val = tensor([1])]; + bool variance_357_keep_dims_0 = const()[name = string("variance_357_keep_dims_0"), val = bool(true)]; + tensor variance_357_cast_fp16 = reduce_mean(axes = variance_357_axes_0, keep_dims = variance_357_keep_dims_0, x = var_10676_cast_fp16)[name = string("variance_357_cast_fp16")]; + fp16 var_10679_to_fp16 = const()[name = string("op_10679_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10680_cast_fp16 = add(x = variance_357_cast_fp16, y = var_10679_to_fp16)[name = string("op_10680_cast_fp16")]; + fp32 var_10681_epsilon_0 = const()[name = string("op_10681_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10681_cast_fp16 = rsqrt(epsilon = var_10681_epsilon_0, x = var_10680_cast_fp16)[name = string("op_10681_cast_fp16")]; + tensor var_10682_cast_fp16 = mul(x = x_325_cast_fp16, y = var_10681_cast_fp16)[name = string("op_10682_cast_fp16")]; + tensor input_453_cast_fp16 = mul(x = var_10682_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_453_cast_fp16")]; + string input_455_pad_type_0 = const()[name = string("input_455_pad_type_0"), val = string("valid")]; + tensor input_455_strides_0 = const()[name = string("input_455_strides_0"), val = tensor([1, 1])]; + tensor input_455_pad_0 = const()[name = string("input_455_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_455_dilations_0 = const()[name = string("input_455_dilations_0"), val = tensor([1, 1])]; + int32 input_455_groups_0 = const()[name = string("input_455_groups_0"), val = int32(1)]; + tensor input_455_cast_fp16 = conv(dilations = input_455_dilations_0, groups = input_455_groups_0, pad = input_455_pad_0, pad_type = input_455_pad_type_0, strides = input_455_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_453_cast_fp16)[name = string("input_455_cast_fp16")]; + tensor var_10690_cast_fp16 = silu(x = input_455_cast_fp16)[name = string("op_10690_cast_fp16")]; + string var_10696_pad_type_0 = const()[name = string("op_10696_pad_type_0"), val = string("valid")]; + tensor var_10696_strides_0 = const()[name = string("op_10696_strides_0"), val = tensor([1, 1])]; + tensor var_10696_pad_0 = const()[name = string("op_10696_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10696_dilations_0 = const()[name = string("op_10696_dilations_0"), val = tensor([1, 1])]; + int32 var_10696_groups_0 = const()[name = string("op_10696_groups_0"), val = int32(1)]; + tensor var_10696_cast_fp16 = conv(dilations = var_10696_dilations_0, groups = var_10696_groups_0, pad = var_10696_pad_0, pad_type = var_10696_pad_type_0, strides = var_10696_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_453_cast_fp16)[name = string("op_10696_cast_fp16")]; + tensor input_457_cast_fp16 = mul(x = var_10690_cast_fp16, y = var_10696_cast_fp16)[name = string("input_457_cast_fp16")]; + string h_85_pad_type_0 = const()[name = string("h_85_pad_type_0"), val = string("valid")]; + tensor h_85_strides_0 = const()[name = string("h_85_strides_0"), val = tensor([1, 1])]; + tensor h_85_pad_0 = const()[name = string("h_85_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_85_dilations_0 = const()[name = string("h_85_dilations_0"), val = tensor([1, 1])]; + int32 h_85_groups_0 = const()[name = string("h_85_groups_0"), val = int32(1)]; + tensor h_85_cast_fp16 = conv(dilations = h_85_dilations_0, groups = h_85_groups_0, pad = h_85_pad_0, pad_type = h_85_pad_type_0, strides = h_85_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_457_cast_fp16)[name = string("h_85_cast_fp16")]; + tensor x_327_cast_fp16 = add(x = x_325_cast_fp16, y = h_85_cast_fp16)[name = string("x_327_cast_fp16")]; + tensor key_cache_87_begin_0 = const()[name = string("key_cache_87_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_87_end_0 = const()[name = string("key_cache_87_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_87_end_mask_0 = const()[name = string("key_cache_87_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_87_cast_fp16 = slice_by_index(begin = key_cache_87_begin_0, end = key_cache_87_end_0, end_mask = key_cache_87_end_mask_0, x = layer_key_caches_17_cast_fp16)[name = string("key_cache_87_cast_fp16")]; + tensor value_cache_87_begin_0 = const()[name = string("value_cache_87_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_87_end_0 = const()[name = string("value_cache_87_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_87_end_mask_0 = const()[name = string("value_cache_87_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_87_cast_fp16 = slice_by_index(begin = value_cache_87_begin_0, end = value_cache_87_end_0, end_mask = value_cache_87_end_mask_0, x = layer_value_caches_17_cast_fp16)[name = string("value_cache_87_cast_fp16")]; + int32 var_10749 = const()[name = string("op_10749"), val = int32(2)]; + int32 var_10753 = const()[name = string("op_10753"), val = int32(3)]; + tensor var_10768_cast_fp16 = mul(x = x_327_cast_fp16, y = x_327_cast_fp16)[name = string("op_10768_cast_fp16")]; + tensor variance_359_axes_0 = const()[name = string("variance_359_axes_0"), val = tensor([1])]; + bool variance_359_keep_dims_0 = const()[name = string("variance_359_keep_dims_0"), val = bool(true)]; + tensor variance_359_cast_fp16 = reduce_mean(axes = variance_359_axes_0, keep_dims = variance_359_keep_dims_0, x = var_10768_cast_fp16)[name = string("variance_359_cast_fp16")]; + fp16 var_10771_to_fp16 = const()[name = string("op_10771_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10772_cast_fp16 = add(x = variance_359_cast_fp16, y = var_10771_to_fp16)[name = string("op_10772_cast_fp16")]; + fp32 var_10773_epsilon_0 = const()[name = string("op_10773_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10773_cast_fp16 = rsqrt(epsilon = var_10773_epsilon_0, x = var_10772_cast_fp16)[name = string("op_10773_cast_fp16")]; + tensor var_10774_cast_fp16 = mul(x = x_327_cast_fp16, y = var_10773_cast_fp16)[name = string("op_10774_cast_fp16")]; + tensor input_459_cast_fp16 = mul(x = var_10774_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_459_cast_fp16")]; + string q_259_pad_type_0 = const()[name = string("q_259_pad_type_0"), val = string("valid")]; + tensor q_259_strides_0 = const()[name = string("q_259_strides_0"), val = tensor([1, 1])]; + tensor q_259_pad_0 = const()[name = string("q_259_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_259_dilations_0 = const()[name = string("q_259_dilations_0"), val = tensor([1, 1])]; + int32 q_259_groups_0 = const()[name = string("q_259_groups_0"), val = int32(1)]; + tensor q_259_cast_fp16 = conv(dilations = q_259_dilations_0, groups = q_259_groups_0, pad = q_259_pad_0, pad_type = q_259_pad_type_0, strides = q_259_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_459_cast_fp16)[name = string("q_259_cast_fp16")]; + string k_259_pad_type_0 = const()[name = string("k_259_pad_type_0"), val = string("valid")]; + tensor k_259_strides_0 = const()[name = string("k_259_strides_0"), val = tensor([1, 1])]; + tensor k_259_pad_0 = const()[name = string("k_259_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_259_dilations_0 = const()[name = string("k_259_dilations_0"), val = tensor([1, 1])]; + int32 k_259_groups_0 = const()[name = string("k_259_groups_0"), val = int32(1)]; + tensor k_259_cast_fp16 = conv(dilations = k_259_dilations_0, groups = k_259_groups_0, pad = k_259_pad_0, pad_type = k_259_pad_type_0, strides = k_259_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_459_cast_fp16)[name = string("k_259_cast_fp16")]; + string v_87_pad_type_0 = const()[name = string("v_87_pad_type_0"), val = string("valid")]; + tensor v_87_strides_0 = const()[name = string("v_87_strides_0"), val = tensor([1, 1])]; + tensor v_87_pad_0 = const()[name = string("v_87_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_87_dilations_0 = const()[name = string("v_87_dilations_0"), val = tensor([1, 1])]; + int32 v_87_groups_0 = const()[name = string("v_87_groups_0"), val = int32(1)]; + tensor v_87_cast_fp16 = conv(dilations = v_87_dilations_0, groups = v_87_groups_0, pad = v_87_pad_0, pad_type = v_87_pad_type_0, strides = v_87_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_459_cast_fp16)[name = string("v_87_cast_fp16")]; + tensor var_10808 = const()[name = string("op_10808"), val = tensor([16, 128, 1, 1])]; + tensor x_329_cast_fp16 = reshape(shape = var_10808, x = q_259_cast_fp16)[name = string("x_329_cast_fp16")]; + tensor var_10811_cast_fp16 = mul(x = x_329_cast_fp16, y = x_329_cast_fp16)[name = string("op_10811_cast_fp16")]; + tensor variance_361_axes_0 = const()[name = string("variance_361_axes_0"), val = tensor([1])]; + bool variance_361_keep_dims_0 = const()[name = string("variance_361_keep_dims_0"), val = bool(true)]; + tensor variance_361_cast_fp16 = reduce_mean(axes = variance_361_axes_0, keep_dims = variance_361_keep_dims_0, x = var_10811_cast_fp16)[name = string("variance_361_cast_fp16")]; + fp16 var_10814_to_fp16 = const()[name = string("op_10814_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10815_cast_fp16 = add(x = variance_361_cast_fp16, y = var_10814_to_fp16)[name = string("op_10815_cast_fp16")]; + fp32 var_10816_epsilon_0 = const()[name = string("op_10816_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10816_cast_fp16 = rsqrt(epsilon = var_10816_epsilon_0, x = var_10815_cast_fp16)[name = string("op_10816_cast_fp16")]; + tensor var_10817_cast_fp16 = mul(x = x_329_cast_fp16, y = var_10816_cast_fp16)[name = string("op_10817_cast_fp16")]; + tensor q_261_cast_fp16 = mul(x = var_10817_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_261_cast_fp16")]; + tensor var_10819 = const()[name = string("op_10819"), val = tensor([8, 128, 1, 1])]; + tensor x_331_cast_fp16 = reshape(shape = var_10819, x = k_259_cast_fp16)[name = string("x_331_cast_fp16")]; + tensor var_10822_cast_fp16 = mul(x = x_331_cast_fp16, y = x_331_cast_fp16)[name = string("op_10822_cast_fp16")]; + tensor variance_363_axes_0 = const()[name = string("variance_363_axes_0"), val = tensor([1])]; + bool variance_363_keep_dims_0 = const()[name = string("variance_363_keep_dims_0"), val = bool(true)]; + tensor variance_363_cast_fp16 = reduce_mean(axes = variance_363_axes_0, keep_dims = variance_363_keep_dims_0, x = var_10822_cast_fp16)[name = string("variance_363_cast_fp16")]; + fp16 var_10825_to_fp16 = const()[name = string("op_10825_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10826_cast_fp16 = add(x = variance_363_cast_fp16, y = var_10825_to_fp16)[name = string("op_10826_cast_fp16")]; + fp32 var_10827_epsilon_0 = const()[name = string("op_10827_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10827_cast_fp16 = rsqrt(epsilon = var_10827_epsilon_0, x = var_10826_cast_fp16)[name = string("op_10827_cast_fp16")]; + tensor var_10828_cast_fp16 = mul(x = x_331_cast_fp16, y = var_10827_cast_fp16)[name = string("op_10828_cast_fp16")]; + tensor k_261_cast_fp16 = mul(x = var_10828_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_261_cast_fp16")]; + tensor var_10830 = const()[name = string("op_10830"), val = tensor([1, 16, 128, 1])]; + tensor z_173_cast_fp16 = reshape(shape = var_10830, x = q_261_cast_fp16)[name = string("z_173_cast_fp16")]; + tensor var_10832 = const()[name = string("op_10832"), val = tensor([1, 8, 128, 1])]; + tensor z_175_cast_fp16 = reshape(shape = var_10832, x = k_261_cast_fp16)[name = string("z_175_cast_fp16")]; + tensor z1_173_begin_0 = const()[name = string("z1_173_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_173_end_0 = const()[name = string("z1_173_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_173_end_mask_0 = const()[name = string("z1_173_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_173_cast_fp16 = slice_by_index(begin = z1_173_begin_0, end = z1_173_end_0, end_mask = z1_173_end_mask_0, x = z_173_cast_fp16)[name = string("z1_173_cast_fp16")]; + tensor z2_173_begin_0 = const()[name = string("z2_173_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_173_end_0 = const()[name = string("z2_173_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_173_end_mask_0 = const()[name = string("z2_173_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_173_cast_fp16 = slice_by_index(begin = z2_173_begin_0, end = z2_173_end_0, end_mask = z2_173_end_mask_0, x = z_173_cast_fp16)[name = string("z2_173_cast_fp16")]; + tensor var_10840_cast_fp16 = mul(x = z_173_cast_fp16, y = cos_81_to_fp16)[name = string("op_10840_cast_fp16")]; + fp16 const_95_promoted_to_fp16 = const()[name = string("const_95_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10841_cast_fp16 = mul(x = z2_173_cast_fp16, y = const_95_promoted_to_fp16)[name = string("op_10841_cast_fp16")]; + bool var_10843_interleave_0 = const()[name = string("op_10843_interleave_0"), val = bool(false)]; + tensor var_10843_cast_fp16 = concat(axis = var_10749, interleave = var_10843_interleave_0, values = (var_10841_cast_fp16, z1_173_cast_fp16))[name = string("op_10843_cast_fp16")]; + tensor var_10844_cast_fp16 = mul(x = var_10843_cast_fp16, y = sin_81_to_fp16)[name = string("op_10844_cast_fp16")]; + tensor q_263_cast_fp16 = add(x = var_10840_cast_fp16, y = var_10844_cast_fp16)[name = string("q_263_cast_fp16")]; + tensor z1_175_begin_0 = const()[name = string("z1_175_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_175_end_0 = const()[name = string("z1_175_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_175_end_mask_0 = const()[name = string("z1_175_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_175_cast_fp16 = slice_by_index(begin = z1_175_begin_0, end = z1_175_end_0, end_mask = z1_175_end_mask_0, x = z_175_cast_fp16)[name = string("z1_175_cast_fp16")]; + tensor z2_175_begin_0 = const()[name = string("z2_175_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_175_end_0 = const()[name = string("z2_175_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_175_end_mask_0 = const()[name = string("z2_175_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_175_cast_fp16 = slice_by_index(begin = z2_175_begin_0, end = z2_175_end_0, end_mask = z2_175_end_mask_0, x = z_175_cast_fp16)[name = string("z2_175_cast_fp16")]; + tensor var_10852_cast_fp16 = mul(x = z_175_cast_fp16, y = cos_81_to_fp16)[name = string("op_10852_cast_fp16")]; + fp16 const_96_promoted_to_fp16 = const()[name = string("const_96_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10853_cast_fp16 = mul(x = z2_175_cast_fp16, y = const_96_promoted_to_fp16)[name = string("op_10853_cast_fp16")]; + bool var_10855_interleave_0 = const()[name = string("op_10855_interleave_0"), val = bool(false)]; + tensor var_10855_cast_fp16 = concat(axis = var_10749, interleave = var_10855_interleave_0, values = (var_10853_cast_fp16, z1_175_cast_fp16))[name = string("op_10855_cast_fp16")]; + tensor var_10856_cast_fp16 = mul(x = var_10855_cast_fp16, y = sin_81_to_fp16)[name = string("op_10856_cast_fp16")]; + tensor k_263_cast_fp16 = add(x = var_10852_cast_fp16, y = var_10856_cast_fp16)[name = string("k_263_cast_fp16")]; + tensor var_10858 = const()[name = string("op_10858"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_87_cast_fp16 = reshape(shape = var_10858, x = k_263_cast_fp16)[name = string("cur_key_87_cast_fp16")]; + tensor var_10860_to_fp16 = const()[name = string("op_10860_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640192)))]; + tensor var_10861_cast_fp16 = mul(x = key_cache_87_cast_fp16, y = var_10860_to_fp16)[name = string("op_10861_cast_fp16")]; + tensor upd_87_to_fp16 = const()[name = string("upd_87_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640320)))]; + tensor var_10862_cast_fp16 = mul(x = cur_key_87_cast_fp16, y = upd_87_to_fp16)[name = string("op_10862_cast_fp16")]; + tensor key_87_cast_fp16 = add(x = var_10861_cast_fp16, y = var_10862_cast_fp16)[name = string("key_87_cast_fp16")]; + tensor var_10864_to_fp16 = const()[name = string("op_10864_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640192)))]; + tensor var_10865_cast_fp16 = mul(x = value_cache_87_cast_fp16, y = var_10864_to_fp16)[name = string("op_10865_cast_fp16")]; + tensor var_10866_cast_fp16 = mul(x = v_87_cast_fp16, y = upd_87_to_fp16)[name = string("op_10866_cast_fp16")]; + tensor value_87_cast_fp16 = add(x = var_10865_cast_fp16, y = var_10866_cast_fp16)[name = string("value_87_cast_fp16")]; + tensor var_10868 = const()[name = string("op_10868"), val = tensor([1, 8, 128, 16])]; + tensor kh_173_cast_fp16 = reshape(shape = var_10868, x = key_87_cast_fp16)[name = string("kh_173_cast_fp16")]; + tensor var_10870 = const()[name = string("op_10870"), val = tensor([1, 8, 128, 16])]; + tensor vh_173_cast_fp16 = reshape(shape = var_10870, x = value_87_cast_fp16)[name = string("vh_173_cast_fp16")]; + tensor transpose_172_perm_0 = const()[name = string("transpose_172_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_86_reps_0 = const()[name = string("tile_86_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_172_cast_fp16 = transpose(perm = transpose_172_perm_0, x = kh_173_cast_fp16)[name = string("transpose_221")]; + tensor tile_86_cast_fp16 = tile(reps = tile_86_reps_0, x = transpose_172_cast_fp16)[name = string("tile_86_cast_fp16")]; + tensor concat_215 = const()[name = string("concat_215"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_172_cast_fp16 = reshape(shape = concat_215, x = tile_86_cast_fp16)[name = string("reshape_172_cast_fp16")]; + tensor transpose_173_perm_0 = const()[name = string("transpose_173_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_216 = const()[name = string("concat_216"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_173_cast_fp16 = transpose(perm = transpose_173_perm_0, x = reshape_172_cast_fp16)[name = string("transpose_220")]; + tensor reshape_173_cast_fp16 = reshape(shape = concat_216, x = transpose_173_cast_fp16)[name = string("reshape_173_cast_fp16")]; + tensor transpose_174_perm_0 = const()[name = string("transpose_174_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_87_reps_0 = const()[name = string("tile_87_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_174_cast_fp16 = transpose(perm = transpose_174_perm_0, x = vh_173_cast_fp16)[name = string("transpose_219")]; + tensor tile_87_cast_fp16 = tile(reps = tile_87_reps_0, x = transpose_174_cast_fp16)[name = string("tile_87_cast_fp16")]; + tensor concat_217 = const()[name = string("concat_217"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_174_cast_fp16 = reshape(shape = concat_217, x = tile_87_cast_fp16)[name = string("reshape_174_cast_fp16")]; + tensor transpose_175_perm_0 = const()[name = string("transpose_175_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_218 = const()[name = string("concat_218"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_175_cast_fp16 = transpose(perm = transpose_175_perm_0, x = reshape_174_cast_fp16)[name = string("transpose_218")]; + tensor reshape_175_cast_fp16 = reshape(shape = concat_218, x = transpose_175_cast_fp16)[name = string("reshape_175_cast_fp16")]; + fp16 var_10874_to_fp16 = const()[name = string("op_10874_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_10875_cast_fp16 = mul(x = q_263_cast_fp16, y = var_10874_to_fp16)[name = string("op_10875_cast_fp16")]; + tensor transpose_489_perm_0 = const()[name = string("transpose_489_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_187_transpose_x_1 = const()[name = string("w_187_transpose_x_1"), val = bool(true)]; + bool w_187_transpose_y_1 = const()[name = string("w_187_transpose_y_1"), val = bool(false)]; + tensor transpose_489_cast_fp16 = transpose(perm = transpose_489_perm_0, x = reshape_173_cast_fp16)[name = string("transpose_217")]; + tensor w_187_cast_fp16 = matmul(transpose_x = w_187_transpose_x_1, transpose_y = w_187_transpose_y_1, x = var_10875_cast_fp16, y = transpose_489_cast_fp16)[name = string("w_187_cast_fp16")]; + tensor pad_87_to_fp16 = const()[name = string("pad_87_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640448)))]; + tensor var_10878_cast_fp16 = add(x = w_187_cast_fp16, y = pad_87_to_fp16)[name = string("op_10878_cast_fp16")]; + tensor w_189_cast_fp16 = softmax(axis = var_10753, x = var_10878_cast_fp16)[name = string("w_189_cast_fp16")]; + tensor transpose_490_perm_0 = const()[name = string("transpose_490_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_87_transpose_x_1 = const()[name = string("attn_87_transpose_x_1"), val = bool(false)]; + bool attn_87_transpose_y_1 = const()[name = string("attn_87_transpose_y_1"), val = bool(true)]; + tensor transpose_490_cast_fp16 = transpose(perm = transpose_490_perm_0, x = reshape_175_cast_fp16)[name = string("transpose_216")]; + tensor attn_87_cast_fp16 = matmul(transpose_x = attn_87_transpose_x_1, transpose_y = attn_87_transpose_y_1, x = transpose_490_cast_fp16, y = w_189_cast_fp16)[name = string("attn_87_cast_fp16")]; + tensor var_10882 = const()[name = string("op_10882"), val = tensor([1, 2048, 1, 1])]; + tensor input_461_cast_fp16 = reshape(shape = var_10882, x = attn_87_cast_fp16)[name = string("input_461_cast_fp16")]; + string attn_output_87_pad_type_0 = const()[name = string("attn_output_87_pad_type_0"), val = string("valid")]; + tensor attn_output_87_strides_0 = const()[name = string("attn_output_87_strides_0"), val = tensor([1, 1])]; + tensor attn_output_87_pad_0 = const()[name = string("attn_output_87_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_87_dilations_0 = const()[name = string("attn_output_87_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_87_groups_0 = const()[name = string("attn_output_87_groups_0"), val = int32(1)]; + tensor attn_output_87_cast_fp16 = conv(dilations = attn_output_87_dilations_0, groups = attn_output_87_groups_0, pad = attn_output_87_pad_0, pad_type = attn_output_87_pad_type_0, strides = attn_output_87_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_461_cast_fp16)[name = string("attn_output_87_cast_fp16")]; + tensor x_333_cast_fp16 = add(x = x_327_cast_fp16, y = attn_output_87_cast_fp16)[name = string("x_333_cast_fp16")]; + tensor var_10896_cast_fp16 = mul(x = x_333_cast_fp16, y = x_333_cast_fp16)[name = string("op_10896_cast_fp16")]; + tensor variance_365_axes_0 = const()[name = string("variance_365_axes_0"), val = tensor([1])]; + bool variance_365_keep_dims_0 = const()[name = string("variance_365_keep_dims_0"), val = bool(true)]; + tensor variance_365_cast_fp16 = reduce_mean(axes = variance_365_axes_0, keep_dims = variance_365_keep_dims_0, x = var_10896_cast_fp16)[name = string("variance_365_cast_fp16")]; + fp16 var_10899_to_fp16 = const()[name = string("op_10899_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10900_cast_fp16 = add(x = variance_365_cast_fp16, y = var_10899_to_fp16)[name = string("op_10900_cast_fp16")]; + fp32 var_10901_epsilon_0 = const()[name = string("op_10901_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10901_cast_fp16 = rsqrt(epsilon = var_10901_epsilon_0, x = var_10900_cast_fp16)[name = string("op_10901_cast_fp16")]; + tensor var_10902_cast_fp16 = mul(x = x_333_cast_fp16, y = var_10901_cast_fp16)[name = string("op_10902_cast_fp16")]; + tensor input_463_cast_fp16 = mul(x = var_10902_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_463_cast_fp16")]; + string input_465_pad_type_0 = const()[name = string("input_465_pad_type_0"), val = string("valid")]; + tensor input_465_strides_0 = const()[name = string("input_465_strides_0"), val = tensor([1, 1])]; + tensor input_465_pad_0 = const()[name = string("input_465_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_465_dilations_0 = const()[name = string("input_465_dilations_0"), val = tensor([1, 1])]; + int32 input_465_groups_0 = const()[name = string("input_465_groups_0"), val = int32(1)]; + tensor input_465_cast_fp16 = conv(dilations = input_465_dilations_0, groups = input_465_groups_0, pad = input_465_pad_0, pad_type = input_465_pad_type_0, strides = input_465_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_463_cast_fp16)[name = string("input_465_cast_fp16")]; + tensor var_10910_cast_fp16 = silu(x = input_465_cast_fp16)[name = string("op_10910_cast_fp16")]; + string var_10916_pad_type_0 = const()[name = string("op_10916_pad_type_0"), val = string("valid")]; + tensor var_10916_strides_0 = const()[name = string("op_10916_strides_0"), val = tensor([1, 1])]; + tensor var_10916_pad_0 = const()[name = string("op_10916_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10916_dilations_0 = const()[name = string("op_10916_dilations_0"), val = tensor([1, 1])]; + int32 var_10916_groups_0 = const()[name = string("op_10916_groups_0"), val = int32(1)]; + tensor var_10916_cast_fp16 = conv(dilations = var_10916_dilations_0, groups = var_10916_groups_0, pad = var_10916_pad_0, pad_type = var_10916_pad_type_0, strides = var_10916_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_463_cast_fp16)[name = string("op_10916_cast_fp16")]; + tensor input_467_cast_fp16 = mul(x = var_10910_cast_fp16, y = var_10916_cast_fp16)[name = string("input_467_cast_fp16")]; + string h_87_pad_type_0 = const()[name = string("h_87_pad_type_0"), val = string("valid")]; + tensor h_87_strides_0 = const()[name = string("h_87_strides_0"), val = tensor([1, 1])]; + tensor h_87_pad_0 = const()[name = string("h_87_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_87_dilations_0 = const()[name = string("h_87_dilations_0"), val = tensor([1, 1])]; + int32 h_87_groups_0 = const()[name = string("h_87_groups_0"), val = int32(1)]; + tensor h_87_cast_fp16 = conv(dilations = h_87_dilations_0, groups = h_87_groups_0, pad = h_87_pad_0, pad_type = h_87_pad_type_0, strides = h_87_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_467_cast_fp16)[name = string("h_87_cast_fp16")]; + tensor x_335_cast_fp16 = add(x = x_333_cast_fp16, y = h_87_cast_fp16)[name = string("x_335_cast_fp16")]; + tensor key_cache_89_begin_0 = const()[name = string("key_cache_89_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_89_end_0 = const()[name = string("key_cache_89_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_89_end_mask_0 = const()[name = string("key_cache_89_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_89_cast_fp16 = slice_by_index(begin = key_cache_89_begin_0, end = key_cache_89_end_0, end_mask = key_cache_89_end_mask_0, x = layer_key_caches_17_cast_fp16)[name = string("key_cache_89_cast_fp16")]; + tensor value_cache_89_begin_0 = const()[name = string("value_cache_89_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_89_end_0 = const()[name = string("value_cache_89_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_89_end_mask_0 = const()[name = string("value_cache_89_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_89_cast_fp16 = slice_by_index(begin = value_cache_89_begin_0, end = value_cache_89_end_0, end_mask = value_cache_89_end_mask_0, x = layer_value_caches_17_cast_fp16)[name = string("value_cache_89_cast_fp16")]; + int32 var_10969 = const()[name = string("op_10969"), val = int32(2)]; + int32 var_10973 = const()[name = string("op_10973"), val = int32(3)]; + tensor var_10988_cast_fp16 = mul(x = x_335_cast_fp16, y = x_335_cast_fp16)[name = string("op_10988_cast_fp16")]; + tensor variance_367_axes_0 = const()[name = string("variance_367_axes_0"), val = tensor([1])]; + bool variance_367_keep_dims_0 = const()[name = string("variance_367_keep_dims_0"), val = bool(true)]; + tensor variance_367_cast_fp16 = reduce_mean(axes = variance_367_axes_0, keep_dims = variance_367_keep_dims_0, x = var_10988_cast_fp16)[name = string("variance_367_cast_fp16")]; + fp16 var_10991_to_fp16 = const()[name = string("op_10991_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10992_cast_fp16 = add(x = variance_367_cast_fp16, y = var_10991_to_fp16)[name = string("op_10992_cast_fp16")]; + fp32 var_10993_epsilon_0 = const()[name = string("op_10993_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10993_cast_fp16 = rsqrt(epsilon = var_10993_epsilon_0, x = var_10992_cast_fp16)[name = string("op_10993_cast_fp16")]; + tensor var_10994_cast_fp16 = mul(x = x_335_cast_fp16, y = var_10993_cast_fp16)[name = string("op_10994_cast_fp16")]; + tensor input_469_cast_fp16 = mul(x = var_10994_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_469_cast_fp16")]; + string q_265_pad_type_0 = const()[name = string("q_265_pad_type_0"), val = string("valid")]; + tensor q_265_strides_0 = const()[name = string("q_265_strides_0"), val = tensor([1, 1])]; + tensor q_265_pad_0 = const()[name = string("q_265_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_265_dilations_0 = const()[name = string("q_265_dilations_0"), val = tensor([1, 1])]; + int32 q_265_groups_0 = const()[name = string("q_265_groups_0"), val = int32(1)]; + tensor q_265_cast_fp16 = conv(dilations = q_265_dilations_0, groups = q_265_groups_0, pad = q_265_pad_0, pad_type = q_265_pad_type_0, strides = q_265_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_469_cast_fp16)[name = string("q_265_cast_fp16")]; + string k_265_pad_type_0 = const()[name = string("k_265_pad_type_0"), val = string("valid")]; + tensor k_265_strides_0 = const()[name = string("k_265_strides_0"), val = tensor([1, 1])]; + tensor k_265_pad_0 = const()[name = string("k_265_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_265_dilations_0 = const()[name = string("k_265_dilations_0"), val = tensor([1, 1])]; + int32 k_265_groups_0 = const()[name = string("k_265_groups_0"), val = int32(1)]; + tensor k_265_cast_fp16 = conv(dilations = k_265_dilations_0, groups = k_265_groups_0, pad = k_265_pad_0, pad_type = k_265_pad_type_0, strides = k_265_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_469_cast_fp16)[name = string("k_265_cast_fp16")]; + string v_89_pad_type_0 = const()[name = string("v_89_pad_type_0"), val = string("valid")]; + tensor v_89_strides_0 = const()[name = string("v_89_strides_0"), val = tensor([1, 1])]; + tensor v_89_pad_0 = const()[name = string("v_89_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_89_dilations_0 = const()[name = string("v_89_dilations_0"), val = tensor([1, 1])]; + int32 v_89_groups_0 = const()[name = string("v_89_groups_0"), val = int32(1)]; + tensor v_89_cast_fp16 = conv(dilations = v_89_dilations_0, groups = v_89_groups_0, pad = v_89_pad_0, pad_type = v_89_pad_type_0, strides = v_89_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_469_cast_fp16)[name = string("v_89_cast_fp16")]; + tensor var_11028 = const()[name = string("op_11028"), val = tensor([16, 128, 1, 1])]; + tensor x_337_cast_fp16 = reshape(shape = var_11028, x = q_265_cast_fp16)[name = string("x_337_cast_fp16")]; + tensor var_11031_cast_fp16 = mul(x = x_337_cast_fp16, y = x_337_cast_fp16)[name = string("op_11031_cast_fp16")]; + tensor variance_369_axes_0 = const()[name = string("variance_369_axes_0"), val = tensor([1])]; + bool variance_369_keep_dims_0 = const()[name = string("variance_369_keep_dims_0"), val = bool(true)]; + tensor variance_369_cast_fp16 = reduce_mean(axes = variance_369_axes_0, keep_dims = variance_369_keep_dims_0, x = var_11031_cast_fp16)[name = string("variance_369_cast_fp16")]; + fp16 var_11034_to_fp16 = const()[name = string("op_11034_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11035_cast_fp16 = add(x = variance_369_cast_fp16, y = var_11034_to_fp16)[name = string("op_11035_cast_fp16")]; + fp32 var_11036_epsilon_0 = const()[name = string("op_11036_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11036_cast_fp16 = rsqrt(epsilon = var_11036_epsilon_0, x = var_11035_cast_fp16)[name = string("op_11036_cast_fp16")]; + tensor var_11037_cast_fp16 = mul(x = x_337_cast_fp16, y = var_11036_cast_fp16)[name = string("op_11037_cast_fp16")]; + tensor q_267_cast_fp16 = mul(x = var_11037_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_267_cast_fp16")]; + tensor var_11039 = const()[name = string("op_11039"), val = tensor([8, 128, 1, 1])]; + tensor x_339_cast_fp16 = reshape(shape = var_11039, x = k_265_cast_fp16)[name = string("x_339_cast_fp16")]; + tensor var_11042_cast_fp16 = mul(x = x_339_cast_fp16, y = x_339_cast_fp16)[name = string("op_11042_cast_fp16")]; + tensor variance_371_axes_0 = const()[name = string("variance_371_axes_0"), val = tensor([1])]; + bool variance_371_keep_dims_0 = const()[name = string("variance_371_keep_dims_0"), val = bool(true)]; + tensor variance_371_cast_fp16 = reduce_mean(axes = variance_371_axes_0, keep_dims = variance_371_keep_dims_0, x = var_11042_cast_fp16)[name = string("variance_371_cast_fp16")]; + fp16 var_11045_to_fp16 = const()[name = string("op_11045_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11046_cast_fp16 = add(x = variance_371_cast_fp16, y = var_11045_to_fp16)[name = string("op_11046_cast_fp16")]; + fp32 var_11047_epsilon_0 = const()[name = string("op_11047_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11047_cast_fp16 = rsqrt(epsilon = var_11047_epsilon_0, x = var_11046_cast_fp16)[name = string("op_11047_cast_fp16")]; + tensor var_11048_cast_fp16 = mul(x = x_339_cast_fp16, y = var_11047_cast_fp16)[name = string("op_11048_cast_fp16")]; + tensor k_267_cast_fp16 = mul(x = var_11048_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_267_cast_fp16")]; + tensor var_11050 = const()[name = string("op_11050"), val = tensor([1, 16, 128, 1])]; + tensor z_177_cast_fp16 = reshape(shape = var_11050, x = q_267_cast_fp16)[name = string("z_177_cast_fp16")]; + tensor var_11052 = const()[name = string("op_11052"), val = tensor([1, 8, 128, 1])]; + tensor z_179_cast_fp16 = reshape(shape = var_11052, x = k_267_cast_fp16)[name = string("z_179_cast_fp16")]; + tensor z1_177_begin_0 = const()[name = string("z1_177_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_177_end_0 = const()[name = string("z1_177_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_177_end_mask_0 = const()[name = string("z1_177_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_177_cast_fp16 = slice_by_index(begin = z1_177_begin_0, end = z1_177_end_0, end_mask = z1_177_end_mask_0, x = z_177_cast_fp16)[name = string("z1_177_cast_fp16")]; + tensor z2_177_begin_0 = const()[name = string("z2_177_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_177_end_0 = const()[name = string("z2_177_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_177_end_mask_0 = const()[name = string("z2_177_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_177_cast_fp16 = slice_by_index(begin = z2_177_begin_0, end = z2_177_end_0, end_mask = z2_177_end_mask_0, x = z_177_cast_fp16)[name = string("z2_177_cast_fp16")]; + tensor var_11060_cast_fp16 = mul(x = z_177_cast_fp16, y = cos_81_to_fp16)[name = string("op_11060_cast_fp16")]; + fp16 const_97_promoted_to_fp16 = const()[name = string("const_97_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11061_cast_fp16 = mul(x = z2_177_cast_fp16, y = const_97_promoted_to_fp16)[name = string("op_11061_cast_fp16")]; + bool var_11063_interleave_0 = const()[name = string("op_11063_interleave_0"), val = bool(false)]; + tensor var_11063_cast_fp16 = concat(axis = var_10969, interleave = var_11063_interleave_0, values = (var_11061_cast_fp16, z1_177_cast_fp16))[name = string("op_11063_cast_fp16")]; + tensor var_11064_cast_fp16 = mul(x = var_11063_cast_fp16, y = sin_81_to_fp16)[name = string("op_11064_cast_fp16")]; + tensor q_269_cast_fp16 = add(x = var_11060_cast_fp16, y = var_11064_cast_fp16)[name = string("q_269_cast_fp16")]; + tensor z1_179_begin_0 = const()[name = string("z1_179_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_179_end_0 = const()[name = string("z1_179_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_179_end_mask_0 = const()[name = string("z1_179_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_179_cast_fp16 = slice_by_index(begin = z1_179_begin_0, end = z1_179_end_0, end_mask = z1_179_end_mask_0, x = z_179_cast_fp16)[name = string("z1_179_cast_fp16")]; + tensor z2_179_begin_0 = const()[name = string("z2_179_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_179_end_0 = const()[name = string("z2_179_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_179_end_mask_0 = const()[name = string("z2_179_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_179_cast_fp16 = slice_by_index(begin = z2_179_begin_0, end = z2_179_end_0, end_mask = z2_179_end_mask_0, x = z_179_cast_fp16)[name = string("z2_179_cast_fp16")]; + tensor var_11072_cast_fp16 = mul(x = z_179_cast_fp16, y = cos_81_to_fp16)[name = string("op_11072_cast_fp16")]; + fp16 const_98_promoted_to_fp16 = const()[name = string("const_98_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11073_cast_fp16 = mul(x = z2_179_cast_fp16, y = const_98_promoted_to_fp16)[name = string("op_11073_cast_fp16")]; + bool var_11075_interleave_0 = const()[name = string("op_11075_interleave_0"), val = bool(false)]; + tensor var_11075_cast_fp16 = concat(axis = var_10969, interleave = var_11075_interleave_0, values = (var_11073_cast_fp16, z1_179_cast_fp16))[name = string("op_11075_cast_fp16")]; + tensor var_11076_cast_fp16 = mul(x = var_11075_cast_fp16, y = sin_81_to_fp16)[name = string("op_11076_cast_fp16")]; + tensor k_269_cast_fp16 = add(x = var_11072_cast_fp16, y = var_11076_cast_fp16)[name = string("k_269_cast_fp16")]; + tensor var_11078 = const()[name = string("op_11078"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_89_cast_fp16 = reshape(shape = var_11078, x = k_269_cast_fp16)[name = string("cur_key_89_cast_fp16")]; + tensor var_11080_to_fp16 = const()[name = string("op_11080_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640192)))]; + tensor var_11081_cast_fp16 = mul(x = key_cache_89_cast_fp16, y = var_11080_to_fp16)[name = string("op_11081_cast_fp16")]; + tensor upd_89_to_fp16 = const()[name = string("upd_89_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640320)))]; + tensor var_11082_cast_fp16 = mul(x = cur_key_89_cast_fp16, y = upd_89_to_fp16)[name = string("op_11082_cast_fp16")]; + tensor key_89_cast_fp16 = add(x = var_11081_cast_fp16, y = var_11082_cast_fp16)[name = string("key_89_cast_fp16")]; + tensor var_11084_to_fp16 = const()[name = string("op_11084_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640192)))]; + tensor var_11085_cast_fp16 = mul(x = value_cache_89_cast_fp16, y = var_11084_to_fp16)[name = string("op_11085_cast_fp16")]; + tensor var_11086_cast_fp16 = mul(x = v_89_cast_fp16, y = upd_89_to_fp16)[name = string("op_11086_cast_fp16")]; + tensor value_89_cast_fp16 = add(x = var_11085_cast_fp16, y = var_11086_cast_fp16)[name = string("value_89_cast_fp16")]; + tensor var_11088 = const()[name = string("op_11088"), val = tensor([1, 8, 128, 16])]; + tensor kh_177_cast_fp16 = reshape(shape = var_11088, x = key_89_cast_fp16)[name = string("kh_177_cast_fp16")]; + tensor var_11090 = const()[name = string("op_11090"), val = tensor([1, 8, 128, 16])]; + tensor vh_177_cast_fp16 = reshape(shape = var_11090, x = value_89_cast_fp16)[name = string("vh_177_cast_fp16")]; + tensor transpose_176_perm_0 = const()[name = string("transpose_176_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_88_reps_0 = const()[name = string("tile_88_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_176_cast_fp16 = transpose(perm = transpose_176_perm_0, x = kh_177_cast_fp16)[name = string("transpose_215")]; + tensor tile_88_cast_fp16 = tile(reps = tile_88_reps_0, x = transpose_176_cast_fp16)[name = string("tile_88_cast_fp16")]; + tensor concat_219 = const()[name = string("concat_219"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_176_cast_fp16 = reshape(shape = concat_219, x = tile_88_cast_fp16)[name = string("reshape_176_cast_fp16")]; + tensor transpose_177_perm_0 = const()[name = string("transpose_177_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_220 = const()[name = string("concat_220"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_177_cast_fp16 = transpose(perm = transpose_177_perm_0, x = reshape_176_cast_fp16)[name = string("transpose_214")]; + tensor reshape_177_cast_fp16 = reshape(shape = concat_220, x = transpose_177_cast_fp16)[name = string("reshape_177_cast_fp16")]; + tensor transpose_178_perm_0 = const()[name = string("transpose_178_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_89_reps_0 = const()[name = string("tile_89_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_178_cast_fp16 = transpose(perm = transpose_178_perm_0, x = vh_177_cast_fp16)[name = string("transpose_213")]; + tensor tile_89_cast_fp16 = tile(reps = tile_89_reps_0, x = transpose_178_cast_fp16)[name = string("tile_89_cast_fp16")]; + tensor concat_221 = const()[name = string("concat_221"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_178_cast_fp16 = reshape(shape = concat_221, x = tile_89_cast_fp16)[name = string("reshape_178_cast_fp16")]; + tensor transpose_179_perm_0 = const()[name = string("transpose_179_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_222 = const()[name = string("concat_222"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_179_cast_fp16 = transpose(perm = transpose_179_perm_0, x = reshape_178_cast_fp16)[name = string("transpose_212")]; + tensor reshape_179_cast_fp16 = reshape(shape = concat_222, x = transpose_179_cast_fp16)[name = string("reshape_179_cast_fp16")]; + fp16 var_11094_to_fp16 = const()[name = string("op_11094_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_11095_cast_fp16 = mul(x = q_269_cast_fp16, y = var_11094_to_fp16)[name = string("op_11095_cast_fp16")]; + tensor transpose_493_perm_0 = const()[name = string("transpose_493_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_191_transpose_x_1 = const()[name = string("w_191_transpose_x_1"), val = bool(true)]; + bool w_191_transpose_y_1 = const()[name = string("w_191_transpose_y_1"), val = bool(false)]; + tensor transpose_493_cast_fp16 = transpose(perm = transpose_493_perm_0, x = reshape_177_cast_fp16)[name = string("transpose_211")]; + tensor w_191_cast_fp16 = matmul(transpose_x = w_191_transpose_x_1, transpose_y = w_191_transpose_y_1, x = var_11095_cast_fp16, y = transpose_493_cast_fp16)[name = string("w_191_cast_fp16")]; + tensor pad_89_to_fp16 = const()[name = string("pad_89_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640448)))]; + tensor var_11098_cast_fp16 = add(x = w_191_cast_fp16, y = pad_89_to_fp16)[name = string("op_11098_cast_fp16")]; + tensor w_193_cast_fp16 = softmax(axis = var_10973, x = var_11098_cast_fp16)[name = string("w_193_cast_fp16")]; + tensor transpose_494_perm_0 = const()[name = string("transpose_494_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_89_transpose_x_1 = const()[name = string("attn_89_transpose_x_1"), val = bool(false)]; + bool attn_89_transpose_y_1 = const()[name = string("attn_89_transpose_y_1"), val = bool(true)]; + tensor transpose_494_cast_fp16 = transpose(perm = transpose_494_perm_0, x = reshape_179_cast_fp16)[name = string("transpose_210")]; + tensor attn_89_cast_fp16 = matmul(transpose_x = attn_89_transpose_x_1, transpose_y = attn_89_transpose_y_1, x = transpose_494_cast_fp16, y = w_193_cast_fp16)[name = string("attn_89_cast_fp16")]; + tensor var_11102 = const()[name = string("op_11102"), val = tensor([1, 2048, 1, 1])]; + tensor input_471_cast_fp16 = reshape(shape = var_11102, x = attn_89_cast_fp16)[name = string("input_471_cast_fp16")]; + string attn_output_89_pad_type_0 = const()[name = string("attn_output_89_pad_type_0"), val = string("valid")]; + tensor attn_output_89_strides_0 = const()[name = string("attn_output_89_strides_0"), val = tensor([1, 1])]; + tensor attn_output_89_pad_0 = const()[name = string("attn_output_89_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_89_dilations_0 = const()[name = string("attn_output_89_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_89_groups_0 = const()[name = string("attn_output_89_groups_0"), val = int32(1)]; + tensor attn_output_89_cast_fp16 = conv(dilations = attn_output_89_dilations_0, groups = attn_output_89_groups_0, pad = attn_output_89_pad_0, pad_type = attn_output_89_pad_type_0, strides = attn_output_89_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_471_cast_fp16)[name = string("attn_output_89_cast_fp16")]; + tensor x_341_cast_fp16 = add(x = x_335_cast_fp16, y = attn_output_89_cast_fp16)[name = string("x_341_cast_fp16")]; + tensor var_11116_cast_fp16 = mul(x = x_341_cast_fp16, y = x_341_cast_fp16)[name = string("op_11116_cast_fp16")]; + tensor variance_373_axes_0 = const()[name = string("variance_373_axes_0"), val = tensor([1])]; + bool variance_373_keep_dims_0 = const()[name = string("variance_373_keep_dims_0"), val = bool(true)]; + tensor variance_373_cast_fp16 = reduce_mean(axes = variance_373_axes_0, keep_dims = variance_373_keep_dims_0, x = var_11116_cast_fp16)[name = string("variance_373_cast_fp16")]; + fp16 var_11119_to_fp16 = const()[name = string("op_11119_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11120_cast_fp16 = add(x = variance_373_cast_fp16, y = var_11119_to_fp16)[name = string("op_11120_cast_fp16")]; + fp32 var_11121_epsilon_0 = const()[name = string("op_11121_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11121_cast_fp16 = rsqrt(epsilon = var_11121_epsilon_0, x = var_11120_cast_fp16)[name = string("op_11121_cast_fp16")]; + tensor var_11122_cast_fp16 = mul(x = x_341_cast_fp16, y = var_11121_cast_fp16)[name = string("op_11122_cast_fp16")]; + tensor input_473_cast_fp16 = mul(x = var_11122_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_473_cast_fp16")]; + string input_475_pad_type_0 = const()[name = string("input_475_pad_type_0"), val = string("valid")]; + tensor input_475_strides_0 = const()[name = string("input_475_strides_0"), val = tensor([1, 1])]; + tensor input_475_pad_0 = const()[name = string("input_475_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_475_dilations_0 = const()[name = string("input_475_dilations_0"), val = tensor([1, 1])]; + int32 input_475_groups_0 = const()[name = string("input_475_groups_0"), val = int32(1)]; + tensor input_475_cast_fp16 = conv(dilations = input_475_dilations_0, groups = input_475_groups_0, pad = input_475_pad_0, pad_type = input_475_pad_type_0, strides = input_475_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_473_cast_fp16)[name = string("input_475_cast_fp16")]; + tensor var_11130_cast_fp16 = silu(x = input_475_cast_fp16)[name = string("op_11130_cast_fp16")]; + string var_11136_pad_type_0 = const()[name = string("op_11136_pad_type_0"), val = string("valid")]; + tensor var_11136_strides_0 = const()[name = string("op_11136_strides_0"), val = tensor([1, 1])]; + tensor var_11136_pad_0 = const()[name = string("op_11136_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11136_dilations_0 = const()[name = string("op_11136_dilations_0"), val = tensor([1, 1])]; + int32 var_11136_groups_0 = const()[name = string("op_11136_groups_0"), val = int32(1)]; + tensor var_11136_cast_fp16 = conv(dilations = var_11136_dilations_0, groups = var_11136_groups_0, pad = var_11136_pad_0, pad_type = var_11136_pad_type_0, strides = var_11136_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_473_cast_fp16)[name = string("op_11136_cast_fp16")]; + tensor input_477_cast_fp16 = mul(x = var_11130_cast_fp16, y = var_11136_cast_fp16)[name = string("input_477_cast_fp16")]; + string h_89_pad_type_0 = const()[name = string("h_89_pad_type_0"), val = string("valid")]; + tensor h_89_strides_0 = const()[name = string("h_89_strides_0"), val = tensor([1, 1])]; + tensor h_89_pad_0 = const()[name = string("h_89_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_89_dilations_0 = const()[name = string("h_89_dilations_0"), val = tensor([1, 1])]; + int32 h_89_groups_0 = const()[name = string("h_89_groups_0"), val = int32(1)]; + tensor h_89_cast_fp16 = conv(dilations = h_89_dilations_0, groups = h_89_groups_0, pad = h_89_pad_0, pad_type = h_89_pad_type_0, strides = h_89_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_477_cast_fp16)[name = string("h_89_cast_fp16")]; + tensor inputs_15_cast_fp16 = add(x = x_341_cast_fp16, y = h_89_cast_fp16)[name = string("inputs_15_cast_fp16")]; + int32 var_11164 = const()[name = string("op_11164"), val = int32(1)]; + bool layer_key_caches_19_interleave_0 = const()[name = string("layer_key_caches_19_interleave_0"), val = bool(false)]; + tensor layer_key_caches_19_cast_fp16 = concat(axis = var_11164, interleave = layer_key_caches_19_interleave_0, values = (key_81_cast_fp16, key_83_cast_fp16, key_85_cast_fp16, key_87_cast_fp16, key_89_cast_fp16))[name = string("layer_key_caches_19_cast_fp16")]; + int32 var_11167 = const()[name = string("op_11167"), val = int32(1)]; + bool layer_value_caches_19_interleave_0 = const()[name = string("layer_value_caches_19_interleave_0"), val = bool(false)]; + tensor layer_value_caches_19_cast_fp16 = concat(axis = var_11167, interleave = layer_value_caches_19_interleave_0, values = (value_81_cast_fp16, value_83_cast_fp16, value_85_cast_fp16, value_87_cast_fp16, value_89_cast_fp16))[name = string("layer_value_caches_19_cast_fp16")]; + tensor inputs_sq_15_cast_fp16 = mul(x = inputs_15_cast_fp16, y = inputs_15_cast_fp16)[name = string("inputs_sq_15_cast_fp16")]; + tensor variance_375_axes_0 = const()[name = string("variance_375_axes_0"), val = tensor([1])]; + bool variance_375_keep_dims_0 = const()[name = string("variance_375_keep_dims_0"), val = bool(true)]; + tensor variance_375_cast_fp16 = reduce_mean(axes = variance_375_axes_0, keep_dims = variance_375_keep_dims_0, x = inputs_sq_15_cast_fp16)[name = string("variance_375_cast_fp16")]; + fp16 var_11177_to_fp16 = const()[name = string("op_11177_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11178_cast_fp16 = add(x = variance_375_cast_fp16, y = var_11177_to_fp16)[name = string("op_11178_cast_fp16")]; + fp32 var_11179_epsilon_0 = const()[name = string("op_11179_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11179_cast_fp16 = rsqrt(epsilon = var_11179_epsilon_0, x = var_11178_cast_fp16)[name = string("op_11179_cast_fp16")]; + tensor hidden_states_15_cast_fp16 = mul(x = inputs_15_cast_fp16, y = var_11179_cast_fp16)[name = string("hidden_states_15_cast_fp16")]; + tensor input_479_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_15_cast_fp16)[name = string("input_479_cast_fp16")]; + string logits_29_pad_type_0 = const()[name = string("logits_29_pad_type_0"), val = string("valid")]; + tensor logits_29_strides_0 = const()[name = string("logits_29_strides_0"), val = tensor([1, 1])]; + tensor logits_29_pad_0 = const()[name = string("logits_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_29_dilations_0 = const()[name = string("logits_29_dilations_0"), val = tensor([1, 1])]; + int32 logits_29_groups_0 = const()[name = string("logits_29_groups_0"), val = int32(1)]; + tensor lm_heads_7_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93391232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95488448))))[name = string("lm_heads_7_weight_to_fp16_palettized")]; + tensor logits_29_cast_fp16 = conv(dilations = logits_29_dilations_0, groups = logits_29_groups_0, pad = logits_29_pad_0, pad_type = logits_29_pad_type_0, strides = logits_29_strides_0, weight = lm_heads_7_weight_to_fp16_palettized, x = input_479_cast_fp16)[name = string("logits_29_cast_fp16")]; + tensor var_11197 = const()[name = string("op_11197"), val = tensor([1, 2048])]; + tensor logits_31_cast_fp16 = reshape(shape = var_11197, x = logits_29_cast_fp16)[name = string("logits_31_cast_fp16")]; + tensor scaled_logits_15_cast_fp16 = real_div(x = logits_31_cast_fp16, y = temperature)[name = string("scaled_logits_15_cast_fp16")]; + int32 var_11207 = const()[name = string("op_11207"), val = int32(100)]; + int32 top_values_15_axis_0 = const()[name = string("top_values_15_axis_0"), val = int32(1)]; + bool top_values_15_ascending_0 = const()[name = string("top_values_15_ascending_0"), val = bool(false)]; + bool top_values_15_sort_0 = const()[name = string("top_values_15_sort_0"), val = bool(true)]; + bool top_values_15_return_indices_0 = const()[name = string("top_values_15_return_indices_0"), val = bool(true)]; + string top_values_15_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_15_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_15_cast_fp16_cast_uint16_0, tensor top_values_15_cast_fp16_cast_uint16_1 = topk(ascending = top_values_15_ascending_0, axis = top_values_15_axis_0, k = var_11207, output_indices_dtype = top_values_15_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_15_return_indices_0, sort = top_values_15_sort_0, x = scaled_logits_15_cast_fp16)[name = string("top_values_15_cast_fp16_cast_uint16")]; + tensor var_11213_cast_fp16 = mul(x = top_values_15_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_11213_cast_fp16")]; + tensor var_11217_cast_fp16 = add(x = var_11213_cast_fp16, y = var_2438_cast_fp16)[name = string("op_11217_cast_fp16")]; + tensor reduce_min_7_axes_0 = const()[name = string("reduce_min_7_axes_0"), val = tensor([1])]; + bool reduce_min_7_keep_dims_0 = const()[name = string("reduce_min_7_keep_dims_0"), val = bool(true)]; + tensor reduce_min_7_cast_fp16 = reduce_min(axes = reduce_min_7_axes_0, keep_dims = reduce_min_7_keep_dims_0, x = var_11217_cast_fp16)[name = string("reduce_min_7_cast_fp16")]; + tensor var_11220_cast_fp16 = greater_equal(x = scaled_logits_15_cast_fp16, y = reduce_min_7_cast_fp16)[name = string("op_11220_cast_fp16")]; + fp16 var_11221_value_0_to_fp16 = const()[name = string("op_11221_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_11221_cast_fp16 = fill_like(ref_tensor = scaled_logits_15_cast_fp16, value = var_11221_value_0_to_fp16)[name = string("op_11221_cast_fp16")]; + tensor masked_logits_15_cast_fp16 = select(a = scaled_logits_15_cast_fp16, b = var_11221_cast_fp16, cond = var_11220_cast_fp16)[name = string("masked_logits_15_cast_fp16")]; + tensor var_11225_begin_0 = const()[name = string("op_11225_begin_0"), val = tensor([7, 0])]; + tensor var_11225_end_0 = const()[name = string("op_11225_end_0"), val = tensor([8, 2048])]; + tensor var_11225_end_mask_0 = const()[name = string("op_11225_end_mask_0"), val = tensor([false, true])]; + tensor var_11225_squeeze_mask_0 = const()[name = string("op_11225_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_11225_cast_fp16 = slice_by_index(begin = var_11225_begin_0, end = var_11225_end_0, end_mask = var_11225_end_mask_0, squeeze_mask = var_11225_squeeze_mask_0, x = gumbel)[name = string("op_11225_cast_fp16")]; + tensor var_11228 = const()[name = string("op_11228"), val = tensor([1, 2048])]; + tensor var_11229_cast_fp16 = reshape(shape = var_11228, x = var_11225_cast_fp16)[name = string("op_11229_cast_fp16")]; + tensor noisy_logits_15_cast_fp16 = add(x = masked_logits_15_cast_fp16, y = var_11229_cast_fp16)[name = string("noisy_logits_15_cast_fp16")]; + int32 code_15_axis_0 = const()[name = string("code_15_axis_0"), val = int32(1)]; + bool code_15_keep_dims_0 = const()[name = string("code_15_keep_dims_0"), val = bool(false)]; + string code_15_output_dtype_0 = const()[name = string("code_15_output_dtype_0"), val = string("int32")]; + tensor code_15_cast_fp16 = reduce_argmax(axis = code_15_axis_0, keep_dims = code_15_keep_dims_0, output_dtype = code_15_output_dtype_0, x = noisy_logits_15_cast_fp16)[name = string("code_15_cast_fp16")]; + int32 var_11240 = const()[name = string("op_11240"), val = int32(14336)]; + tensor input_481 = add(x = code_15_cast_fp16, y = var_11240)[name = string("input_481")]; + int32 code_embed_29_axis_0 = const()[name = string("code_embed_29_axis_0"), val = int32(0)]; + int32 code_embed_29_batch_dims_0 = const()[name = string("code_embed_29_batch_dims_0"), val = int32(0)]; + bool code_embed_29_validate_indices_0 = const()[name = string("code_embed_29_validate_indices_0"), val = bool(false)]; + string input_481_to_uint16_dtype_0 = const()[name = string("input_481_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_481_to_uint16 = cast(dtype = input_481_to_uint16_dtype_0, x = input_481)[name = string("cast_7")]; + tensor code_embed_29_cast_fp16_cast_uint16 = gather(axis = code_embed_29_axis_0, batch_dims = code_embed_29_batch_dims_0, indices = input_481_to_uint16, validate_indices = code_embed_29_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_29_cast_fp16_cast_uint16")]; + tensor var_11244 = const()[name = string("op_11244"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_31_cast_fp16 = reshape(shape = var_11244, x = code_embed_29_cast_fp16_cast_uint16)[name = string("code_embed_31_cast_fp16")]; + tensor embed_sum_17_cast_fp16 = add(x = embed_sum_15_cast_fp16, y = code_embed_31_cast_fp16)[name = string("embed_sum_17_cast_fp16")]; + tensor key_cache_91_begin_0 = const()[name = string("key_cache_91_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_91_end_0 = const()[name = string("key_cache_91_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_91_end_mask_0 = const()[name = string("key_cache_91_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_91_cast_fp16 = slice_by_index(begin = key_cache_91_begin_0, end = key_cache_91_end_0, end_mask = key_cache_91_end_mask_0, x = layer_key_caches_19_cast_fp16)[name = string("key_cache_91_cast_fp16")]; + tensor value_cache_91_begin_0 = const()[name = string("value_cache_91_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_91_end_0 = const()[name = string("value_cache_91_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_91_end_mask_0 = const()[name = string("value_cache_91_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_91_cast_fp16 = slice_by_index(begin = value_cache_91_begin_0, end = value_cache_91_end_0, end_mask = value_cache_91_end_mask_0, x = layer_value_caches_19_cast_fp16)[name = string("value_cache_91_cast_fp16")]; + int32 var_11343 = const()[name = string("op_11343"), val = int32(2)]; + int32 var_11347 = const()[name = string("op_11347"), val = int32(3)]; + tensor var_11362_cast_fp16 = mul(x = code_embed_31_cast_fp16, y = code_embed_31_cast_fp16)[name = string("op_11362_cast_fp16")]; + tensor variance_377_axes_0 = const()[name = string("variance_377_axes_0"), val = tensor([1])]; + bool variance_377_keep_dims_0 = const()[name = string("variance_377_keep_dims_0"), val = bool(true)]; + tensor variance_377_cast_fp16 = reduce_mean(axes = variance_377_axes_0, keep_dims = variance_377_keep_dims_0, x = var_11362_cast_fp16)[name = string("variance_377_cast_fp16")]; + fp16 var_11365_to_fp16 = const()[name = string("op_11365_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11366_cast_fp16 = add(x = variance_377_cast_fp16, y = var_11365_to_fp16)[name = string("op_11366_cast_fp16")]; + fp32 var_11367_epsilon_0 = const()[name = string("op_11367_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11367_cast_fp16 = rsqrt(epsilon = var_11367_epsilon_0, x = var_11366_cast_fp16)[name = string("op_11367_cast_fp16")]; + tensor var_11368_cast_fp16 = mul(x = code_embed_31_cast_fp16, y = var_11367_cast_fp16)[name = string("op_11368_cast_fp16")]; + tensor input_483_cast_fp16 = mul(x = var_11368_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_483_cast_fp16")]; + string q_271_pad_type_0 = const()[name = string("q_271_pad_type_0"), val = string("valid")]; + tensor q_271_strides_0 = const()[name = string("q_271_strides_0"), val = tensor([1, 1])]; + tensor q_271_pad_0 = const()[name = string("q_271_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_271_dilations_0 = const()[name = string("q_271_dilations_0"), val = tensor([1, 1])]; + int32 q_271_groups_0 = const()[name = string("q_271_groups_0"), val = int32(1)]; + tensor q_271_cast_fp16 = conv(dilations = q_271_dilations_0, groups = q_271_groups_0, pad = q_271_pad_0, pad_type = q_271_pad_type_0, strides = q_271_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_483_cast_fp16)[name = string("q_271_cast_fp16")]; + string k_271_pad_type_0 = const()[name = string("k_271_pad_type_0"), val = string("valid")]; + tensor k_271_strides_0 = const()[name = string("k_271_strides_0"), val = tensor([1, 1])]; + tensor k_271_pad_0 = const()[name = string("k_271_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_271_dilations_0 = const()[name = string("k_271_dilations_0"), val = tensor([1, 1])]; + int32 k_271_groups_0 = const()[name = string("k_271_groups_0"), val = int32(1)]; + tensor k_271_cast_fp16 = conv(dilations = k_271_dilations_0, groups = k_271_groups_0, pad = k_271_pad_0, pad_type = k_271_pad_type_0, strides = k_271_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_483_cast_fp16)[name = string("k_271_cast_fp16")]; + string v_91_pad_type_0 = const()[name = string("v_91_pad_type_0"), val = string("valid")]; + tensor v_91_strides_0 = const()[name = string("v_91_strides_0"), val = tensor([1, 1])]; + tensor v_91_pad_0 = const()[name = string("v_91_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_91_dilations_0 = const()[name = string("v_91_dilations_0"), val = tensor([1, 1])]; + int32 v_91_groups_0 = const()[name = string("v_91_groups_0"), val = int32(1)]; + tensor v_91_cast_fp16 = conv(dilations = v_91_dilations_0, groups = v_91_groups_0, pad = v_91_pad_0, pad_type = v_91_pad_type_0, strides = v_91_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_483_cast_fp16)[name = string("v_91_cast_fp16")]; + tensor var_11402 = const()[name = string("op_11402"), val = tensor([16, 128, 1, 1])]; + tensor x_343_cast_fp16 = reshape(shape = var_11402, x = q_271_cast_fp16)[name = string("x_343_cast_fp16")]; + tensor var_11405_cast_fp16 = mul(x = x_343_cast_fp16, y = x_343_cast_fp16)[name = string("op_11405_cast_fp16")]; + tensor variance_379_axes_0 = const()[name = string("variance_379_axes_0"), val = tensor([1])]; + bool variance_379_keep_dims_0 = const()[name = string("variance_379_keep_dims_0"), val = bool(true)]; + tensor variance_379_cast_fp16 = reduce_mean(axes = variance_379_axes_0, keep_dims = variance_379_keep_dims_0, x = var_11405_cast_fp16)[name = string("variance_379_cast_fp16")]; + fp16 var_11408_to_fp16 = const()[name = string("op_11408_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11409_cast_fp16 = add(x = variance_379_cast_fp16, y = var_11408_to_fp16)[name = string("op_11409_cast_fp16")]; + fp32 var_11410_epsilon_0 = const()[name = string("op_11410_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11410_cast_fp16 = rsqrt(epsilon = var_11410_epsilon_0, x = var_11409_cast_fp16)[name = string("op_11410_cast_fp16")]; + tensor var_11411_cast_fp16 = mul(x = x_343_cast_fp16, y = var_11410_cast_fp16)[name = string("op_11411_cast_fp16")]; + tensor q_273_cast_fp16 = mul(x = var_11411_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_273_cast_fp16")]; + tensor var_11413 = const()[name = string("op_11413"), val = tensor([8, 128, 1, 1])]; + tensor x_345_cast_fp16 = reshape(shape = var_11413, x = k_271_cast_fp16)[name = string("x_345_cast_fp16")]; + tensor var_11416_cast_fp16 = mul(x = x_345_cast_fp16, y = x_345_cast_fp16)[name = string("op_11416_cast_fp16")]; + tensor variance_381_axes_0 = const()[name = string("variance_381_axes_0"), val = tensor([1])]; + bool variance_381_keep_dims_0 = const()[name = string("variance_381_keep_dims_0"), val = bool(true)]; + tensor variance_381_cast_fp16 = reduce_mean(axes = variance_381_axes_0, keep_dims = variance_381_keep_dims_0, x = var_11416_cast_fp16)[name = string("variance_381_cast_fp16")]; + fp16 var_11419_to_fp16 = const()[name = string("op_11419_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11420_cast_fp16 = add(x = variance_381_cast_fp16, y = var_11419_to_fp16)[name = string("op_11420_cast_fp16")]; + fp32 var_11421_epsilon_0 = const()[name = string("op_11421_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11421_cast_fp16 = rsqrt(epsilon = var_11421_epsilon_0, x = var_11420_cast_fp16)[name = string("op_11421_cast_fp16")]; + tensor var_11422_cast_fp16 = mul(x = x_345_cast_fp16, y = var_11421_cast_fp16)[name = string("op_11422_cast_fp16")]; + tensor k_273_cast_fp16 = mul(x = var_11422_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_273_cast_fp16")]; + tensor var_11424 = const()[name = string("op_11424"), val = tensor([1, 16, 128, 1])]; + tensor z_181_cast_fp16 = reshape(shape = var_11424, x = q_273_cast_fp16)[name = string("z_181_cast_fp16")]; + tensor var_11426 = const()[name = string("op_11426"), val = tensor([1, 8, 128, 1])]; + tensor z_183_cast_fp16 = reshape(shape = var_11426, x = k_273_cast_fp16)[name = string("z_183_cast_fp16")]; + tensor z1_181_begin_0 = const()[name = string("z1_181_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_181_end_0 = const()[name = string("z1_181_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_181_end_mask_0 = const()[name = string("z1_181_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_181_cast_fp16 = slice_by_index(begin = z1_181_begin_0, end = z1_181_end_0, end_mask = z1_181_end_mask_0, x = z_181_cast_fp16)[name = string("z1_181_cast_fp16")]; + tensor z2_181_begin_0 = const()[name = string("z2_181_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_181_end_0 = const()[name = string("z2_181_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_181_end_mask_0 = const()[name = string("z2_181_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_181_cast_fp16 = slice_by_index(begin = z2_181_begin_0, end = z2_181_end_0, end_mask = z2_181_end_mask_0, x = z_181_cast_fp16)[name = string("z2_181_cast_fp16")]; + tensor cos_91_to_fp16 = const()[name = string("cos_91_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640576)))]; + tensor var_11434_cast_fp16 = mul(x = z_181_cast_fp16, y = cos_91_to_fp16)[name = string("op_11434_cast_fp16")]; + fp16 const_100_promoted_to_fp16 = const()[name = string("const_100_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11435_cast_fp16 = mul(x = z2_181_cast_fp16, y = const_100_promoted_to_fp16)[name = string("op_11435_cast_fp16")]; + bool var_11437_interleave_0 = const()[name = string("op_11437_interleave_0"), val = bool(false)]; + tensor var_11437_cast_fp16 = concat(axis = var_11343, interleave = var_11437_interleave_0, values = (var_11435_cast_fp16, z1_181_cast_fp16))[name = string("op_11437_cast_fp16")]; + tensor sin_91_to_fp16 = const()[name = string("sin_91_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141640896)))]; + tensor var_11438_cast_fp16 = mul(x = var_11437_cast_fp16, y = sin_91_to_fp16)[name = string("op_11438_cast_fp16")]; + tensor q_275_cast_fp16 = add(x = var_11434_cast_fp16, y = var_11438_cast_fp16)[name = string("q_275_cast_fp16")]; + tensor z1_183_begin_0 = const()[name = string("z1_183_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_183_end_0 = const()[name = string("z1_183_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_183_end_mask_0 = const()[name = string("z1_183_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_183_cast_fp16 = slice_by_index(begin = z1_183_begin_0, end = z1_183_end_0, end_mask = z1_183_end_mask_0, x = z_183_cast_fp16)[name = string("z1_183_cast_fp16")]; + tensor z2_183_begin_0 = const()[name = string("z2_183_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_183_end_0 = const()[name = string("z2_183_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_183_end_mask_0 = const()[name = string("z2_183_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_183_cast_fp16 = slice_by_index(begin = z2_183_begin_0, end = z2_183_end_0, end_mask = z2_183_end_mask_0, x = z_183_cast_fp16)[name = string("z2_183_cast_fp16")]; + tensor var_11446_cast_fp16 = mul(x = z_183_cast_fp16, y = cos_91_to_fp16)[name = string("op_11446_cast_fp16")]; + fp16 const_101_promoted_to_fp16 = const()[name = string("const_101_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11447_cast_fp16 = mul(x = z2_183_cast_fp16, y = const_101_promoted_to_fp16)[name = string("op_11447_cast_fp16")]; + bool var_11449_interleave_0 = const()[name = string("op_11449_interleave_0"), val = bool(false)]; + tensor var_11449_cast_fp16 = concat(axis = var_11343, interleave = var_11449_interleave_0, values = (var_11447_cast_fp16, z1_183_cast_fp16))[name = string("op_11449_cast_fp16")]; + tensor var_11450_cast_fp16 = mul(x = var_11449_cast_fp16, y = sin_91_to_fp16)[name = string("op_11450_cast_fp16")]; + tensor k_275_cast_fp16 = add(x = var_11446_cast_fp16, y = var_11450_cast_fp16)[name = string("k_275_cast_fp16")]; + tensor var_11452 = const()[name = string("op_11452"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_91_cast_fp16 = reshape(shape = var_11452, x = k_275_cast_fp16)[name = string("cur_key_91_cast_fp16")]; + tensor var_11454_to_fp16 = const()[name = string("op_11454_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641216)))]; + tensor var_11455_cast_fp16 = mul(x = key_cache_91_cast_fp16, y = var_11454_to_fp16)[name = string("op_11455_cast_fp16")]; + tensor upd_91_to_fp16 = const()[name = string("upd_91_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641344)))]; + tensor var_11456_cast_fp16 = mul(x = cur_key_91_cast_fp16, y = upd_91_to_fp16)[name = string("op_11456_cast_fp16")]; + tensor key_91_cast_fp16 = add(x = var_11455_cast_fp16, y = var_11456_cast_fp16)[name = string("key_91_cast_fp16")]; + tensor var_11458_to_fp16 = const()[name = string("op_11458_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641216)))]; + tensor var_11459_cast_fp16 = mul(x = value_cache_91_cast_fp16, y = var_11458_to_fp16)[name = string("op_11459_cast_fp16")]; + tensor var_11460_cast_fp16 = mul(x = v_91_cast_fp16, y = upd_91_to_fp16)[name = string("op_11460_cast_fp16")]; + tensor value_91_cast_fp16 = add(x = var_11459_cast_fp16, y = var_11460_cast_fp16)[name = string("value_91_cast_fp16")]; + tensor var_11462 = const()[name = string("op_11462"), val = tensor([1, 8, 128, 16])]; + tensor kh_181_cast_fp16 = reshape(shape = var_11462, x = key_91_cast_fp16)[name = string("kh_181_cast_fp16")]; + tensor var_11464 = const()[name = string("op_11464"), val = tensor([1, 8, 128, 16])]; + tensor vh_181_cast_fp16 = reshape(shape = var_11464, x = value_91_cast_fp16)[name = string("vh_181_cast_fp16")]; + tensor transpose_180_perm_0 = const()[name = string("transpose_180_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_90_reps_0 = const()[name = string("tile_90_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_180_cast_fp16 = transpose(perm = transpose_180_perm_0, x = kh_181_cast_fp16)[name = string("transpose_209")]; + tensor tile_90_cast_fp16 = tile(reps = tile_90_reps_0, x = transpose_180_cast_fp16)[name = string("tile_90_cast_fp16")]; + tensor concat_228 = const()[name = string("concat_228"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_180_cast_fp16 = reshape(shape = concat_228, x = tile_90_cast_fp16)[name = string("reshape_180_cast_fp16")]; + tensor transpose_181_perm_0 = const()[name = string("transpose_181_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_229 = const()[name = string("concat_229"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_181_cast_fp16 = transpose(perm = transpose_181_perm_0, x = reshape_180_cast_fp16)[name = string("transpose_208")]; + tensor reshape_181_cast_fp16 = reshape(shape = concat_229, x = transpose_181_cast_fp16)[name = string("reshape_181_cast_fp16")]; + tensor transpose_182_perm_0 = const()[name = string("transpose_182_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_91_reps_0 = const()[name = string("tile_91_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_182_cast_fp16 = transpose(perm = transpose_182_perm_0, x = vh_181_cast_fp16)[name = string("transpose_207")]; + tensor tile_91_cast_fp16 = tile(reps = tile_91_reps_0, x = transpose_182_cast_fp16)[name = string("tile_91_cast_fp16")]; + tensor concat_230 = const()[name = string("concat_230"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_182_cast_fp16 = reshape(shape = concat_230, x = tile_91_cast_fp16)[name = string("reshape_182_cast_fp16")]; + tensor transpose_183_perm_0 = const()[name = string("transpose_183_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_231 = const()[name = string("concat_231"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_183_cast_fp16 = transpose(perm = transpose_183_perm_0, x = reshape_182_cast_fp16)[name = string("transpose_206")]; + tensor reshape_183_cast_fp16 = reshape(shape = concat_231, x = transpose_183_cast_fp16)[name = string("reshape_183_cast_fp16")]; + fp16 var_11468_to_fp16 = const()[name = string("op_11468_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_11469_cast_fp16 = mul(x = q_275_cast_fp16, y = var_11468_to_fp16)[name = string("op_11469_cast_fp16")]; + tensor transpose_497_perm_0 = const()[name = string("transpose_497_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_197_transpose_x_1 = const()[name = string("w_197_transpose_x_1"), val = bool(true)]; + bool w_197_transpose_y_1 = const()[name = string("w_197_transpose_y_1"), val = bool(false)]; + tensor transpose_497_cast_fp16 = transpose(perm = transpose_497_perm_0, x = reshape_181_cast_fp16)[name = string("transpose_205")]; + tensor w_197_cast_fp16 = matmul(transpose_x = w_197_transpose_x_1, transpose_y = w_197_transpose_y_1, x = var_11469_cast_fp16, y = transpose_497_cast_fp16)[name = string("w_197_cast_fp16")]; + tensor pad_91_to_fp16 = const()[name = string("pad_91_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641472)))]; + tensor var_11472_cast_fp16 = add(x = w_197_cast_fp16, y = pad_91_to_fp16)[name = string("op_11472_cast_fp16")]; + tensor w_199_cast_fp16 = softmax(axis = var_11347, x = var_11472_cast_fp16)[name = string("w_199_cast_fp16")]; + tensor transpose_498_perm_0 = const()[name = string("transpose_498_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_91_transpose_x_1 = const()[name = string("attn_91_transpose_x_1"), val = bool(false)]; + bool attn_91_transpose_y_1 = const()[name = string("attn_91_transpose_y_1"), val = bool(true)]; + tensor transpose_498_cast_fp16 = transpose(perm = transpose_498_perm_0, x = reshape_183_cast_fp16)[name = string("transpose_204")]; + tensor attn_91_cast_fp16 = matmul(transpose_x = attn_91_transpose_x_1, transpose_y = attn_91_transpose_y_1, x = transpose_498_cast_fp16, y = w_199_cast_fp16)[name = string("attn_91_cast_fp16")]; + tensor var_11476 = const()[name = string("op_11476"), val = tensor([1, 2048, 1, 1])]; + tensor input_485_cast_fp16 = reshape(shape = var_11476, x = attn_91_cast_fp16)[name = string("input_485_cast_fp16")]; + string attn_output_91_pad_type_0 = const()[name = string("attn_output_91_pad_type_0"), val = string("valid")]; + tensor attn_output_91_strides_0 = const()[name = string("attn_output_91_strides_0"), val = tensor([1, 1])]; + tensor attn_output_91_pad_0 = const()[name = string("attn_output_91_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_91_dilations_0 = const()[name = string("attn_output_91_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_91_groups_0 = const()[name = string("attn_output_91_groups_0"), val = int32(1)]; + tensor attn_output_91_cast_fp16 = conv(dilations = attn_output_91_dilations_0, groups = attn_output_91_groups_0, pad = attn_output_91_pad_0, pad_type = attn_output_91_pad_type_0, strides = attn_output_91_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_485_cast_fp16)[name = string("attn_output_91_cast_fp16")]; + tensor x_347_cast_fp16 = add(x = code_embed_31_cast_fp16, y = attn_output_91_cast_fp16)[name = string("x_347_cast_fp16")]; + tensor var_11490_cast_fp16 = mul(x = x_347_cast_fp16, y = x_347_cast_fp16)[name = string("op_11490_cast_fp16")]; + tensor variance_383_axes_0 = const()[name = string("variance_383_axes_0"), val = tensor([1])]; + bool variance_383_keep_dims_0 = const()[name = string("variance_383_keep_dims_0"), val = bool(true)]; + tensor variance_383_cast_fp16 = reduce_mean(axes = variance_383_axes_0, keep_dims = variance_383_keep_dims_0, x = var_11490_cast_fp16)[name = string("variance_383_cast_fp16")]; + fp16 var_11493_to_fp16 = const()[name = string("op_11493_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11494_cast_fp16 = add(x = variance_383_cast_fp16, y = var_11493_to_fp16)[name = string("op_11494_cast_fp16")]; + fp32 var_11495_epsilon_0 = const()[name = string("op_11495_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11495_cast_fp16 = rsqrt(epsilon = var_11495_epsilon_0, x = var_11494_cast_fp16)[name = string("op_11495_cast_fp16")]; + tensor var_11496_cast_fp16 = mul(x = x_347_cast_fp16, y = var_11495_cast_fp16)[name = string("op_11496_cast_fp16")]; + tensor input_487_cast_fp16 = mul(x = var_11496_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_487_cast_fp16")]; + string input_489_pad_type_0 = const()[name = string("input_489_pad_type_0"), val = string("valid")]; + tensor input_489_strides_0 = const()[name = string("input_489_strides_0"), val = tensor([1, 1])]; + tensor input_489_pad_0 = const()[name = string("input_489_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_489_dilations_0 = const()[name = string("input_489_dilations_0"), val = tensor([1, 1])]; + int32 input_489_groups_0 = const()[name = string("input_489_groups_0"), val = int32(1)]; + tensor input_489_cast_fp16 = conv(dilations = input_489_dilations_0, groups = input_489_groups_0, pad = input_489_pad_0, pad_type = input_489_pad_type_0, strides = input_489_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_487_cast_fp16)[name = string("input_489_cast_fp16")]; + tensor var_11504_cast_fp16 = silu(x = input_489_cast_fp16)[name = string("op_11504_cast_fp16")]; + string var_11510_pad_type_0 = const()[name = string("op_11510_pad_type_0"), val = string("valid")]; + tensor var_11510_strides_0 = const()[name = string("op_11510_strides_0"), val = tensor([1, 1])]; + tensor var_11510_pad_0 = const()[name = string("op_11510_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11510_dilations_0 = const()[name = string("op_11510_dilations_0"), val = tensor([1, 1])]; + int32 var_11510_groups_0 = const()[name = string("op_11510_groups_0"), val = int32(1)]; + tensor var_11510_cast_fp16 = conv(dilations = var_11510_dilations_0, groups = var_11510_groups_0, pad = var_11510_pad_0, pad_type = var_11510_pad_type_0, strides = var_11510_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_487_cast_fp16)[name = string("op_11510_cast_fp16")]; + tensor input_491_cast_fp16 = mul(x = var_11504_cast_fp16, y = var_11510_cast_fp16)[name = string("input_491_cast_fp16")]; + string h_91_pad_type_0 = const()[name = string("h_91_pad_type_0"), val = string("valid")]; + tensor h_91_strides_0 = const()[name = string("h_91_strides_0"), val = tensor([1, 1])]; + tensor h_91_pad_0 = const()[name = string("h_91_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_91_dilations_0 = const()[name = string("h_91_dilations_0"), val = tensor([1, 1])]; + int32 h_91_groups_0 = const()[name = string("h_91_groups_0"), val = int32(1)]; + tensor h_91_cast_fp16 = conv(dilations = h_91_dilations_0, groups = h_91_groups_0, pad = h_91_pad_0, pad_type = h_91_pad_type_0, strides = h_91_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_491_cast_fp16)[name = string("h_91_cast_fp16")]; + tensor x_349_cast_fp16 = add(x = x_347_cast_fp16, y = h_91_cast_fp16)[name = string("x_349_cast_fp16")]; + tensor key_cache_93_begin_0 = const()[name = string("key_cache_93_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_93_end_0 = const()[name = string("key_cache_93_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_93_end_mask_0 = const()[name = string("key_cache_93_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_93_cast_fp16 = slice_by_index(begin = key_cache_93_begin_0, end = key_cache_93_end_0, end_mask = key_cache_93_end_mask_0, x = layer_key_caches_19_cast_fp16)[name = string("key_cache_93_cast_fp16")]; + tensor value_cache_93_begin_0 = const()[name = string("value_cache_93_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_93_end_0 = const()[name = string("value_cache_93_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_93_end_mask_0 = const()[name = string("value_cache_93_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_93_cast_fp16 = slice_by_index(begin = value_cache_93_begin_0, end = value_cache_93_end_0, end_mask = value_cache_93_end_mask_0, x = layer_value_caches_19_cast_fp16)[name = string("value_cache_93_cast_fp16")]; + int32 var_11563 = const()[name = string("op_11563"), val = int32(2)]; + int32 var_11567 = const()[name = string("op_11567"), val = int32(3)]; + tensor var_11582_cast_fp16 = mul(x = x_349_cast_fp16, y = x_349_cast_fp16)[name = string("op_11582_cast_fp16")]; + tensor variance_385_axes_0 = const()[name = string("variance_385_axes_0"), val = tensor([1])]; + bool variance_385_keep_dims_0 = const()[name = string("variance_385_keep_dims_0"), val = bool(true)]; + tensor variance_385_cast_fp16 = reduce_mean(axes = variance_385_axes_0, keep_dims = variance_385_keep_dims_0, x = var_11582_cast_fp16)[name = string("variance_385_cast_fp16")]; + fp16 var_11585_to_fp16 = const()[name = string("op_11585_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11586_cast_fp16 = add(x = variance_385_cast_fp16, y = var_11585_to_fp16)[name = string("op_11586_cast_fp16")]; + fp32 var_11587_epsilon_0 = const()[name = string("op_11587_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11587_cast_fp16 = rsqrt(epsilon = var_11587_epsilon_0, x = var_11586_cast_fp16)[name = string("op_11587_cast_fp16")]; + tensor var_11588_cast_fp16 = mul(x = x_349_cast_fp16, y = var_11587_cast_fp16)[name = string("op_11588_cast_fp16")]; + tensor input_493_cast_fp16 = mul(x = var_11588_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_493_cast_fp16")]; + string q_277_pad_type_0 = const()[name = string("q_277_pad_type_0"), val = string("valid")]; + tensor q_277_strides_0 = const()[name = string("q_277_strides_0"), val = tensor([1, 1])]; + tensor q_277_pad_0 = const()[name = string("q_277_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_277_dilations_0 = const()[name = string("q_277_dilations_0"), val = tensor([1, 1])]; + int32 q_277_groups_0 = const()[name = string("q_277_groups_0"), val = int32(1)]; + tensor q_277_cast_fp16 = conv(dilations = q_277_dilations_0, groups = q_277_groups_0, pad = q_277_pad_0, pad_type = q_277_pad_type_0, strides = q_277_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_493_cast_fp16)[name = string("q_277_cast_fp16")]; + string k_277_pad_type_0 = const()[name = string("k_277_pad_type_0"), val = string("valid")]; + tensor k_277_strides_0 = const()[name = string("k_277_strides_0"), val = tensor([1, 1])]; + tensor k_277_pad_0 = const()[name = string("k_277_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_277_dilations_0 = const()[name = string("k_277_dilations_0"), val = tensor([1, 1])]; + int32 k_277_groups_0 = const()[name = string("k_277_groups_0"), val = int32(1)]; + tensor k_277_cast_fp16 = conv(dilations = k_277_dilations_0, groups = k_277_groups_0, pad = k_277_pad_0, pad_type = k_277_pad_type_0, strides = k_277_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_493_cast_fp16)[name = string("k_277_cast_fp16")]; + string v_93_pad_type_0 = const()[name = string("v_93_pad_type_0"), val = string("valid")]; + tensor v_93_strides_0 = const()[name = string("v_93_strides_0"), val = tensor([1, 1])]; + tensor v_93_pad_0 = const()[name = string("v_93_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_93_dilations_0 = const()[name = string("v_93_dilations_0"), val = tensor([1, 1])]; + int32 v_93_groups_0 = const()[name = string("v_93_groups_0"), val = int32(1)]; + tensor v_93_cast_fp16 = conv(dilations = v_93_dilations_0, groups = v_93_groups_0, pad = v_93_pad_0, pad_type = v_93_pad_type_0, strides = v_93_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_493_cast_fp16)[name = string("v_93_cast_fp16")]; + tensor var_11622 = const()[name = string("op_11622"), val = tensor([16, 128, 1, 1])]; + tensor x_351_cast_fp16 = reshape(shape = var_11622, x = q_277_cast_fp16)[name = string("x_351_cast_fp16")]; + tensor var_11625_cast_fp16 = mul(x = x_351_cast_fp16, y = x_351_cast_fp16)[name = string("op_11625_cast_fp16")]; + tensor variance_387_axes_0 = const()[name = string("variance_387_axes_0"), val = tensor([1])]; + bool variance_387_keep_dims_0 = const()[name = string("variance_387_keep_dims_0"), val = bool(true)]; + tensor variance_387_cast_fp16 = reduce_mean(axes = variance_387_axes_0, keep_dims = variance_387_keep_dims_0, x = var_11625_cast_fp16)[name = string("variance_387_cast_fp16")]; + fp16 var_11628_to_fp16 = const()[name = string("op_11628_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11629_cast_fp16 = add(x = variance_387_cast_fp16, y = var_11628_to_fp16)[name = string("op_11629_cast_fp16")]; + fp32 var_11630_epsilon_0 = const()[name = string("op_11630_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11630_cast_fp16 = rsqrt(epsilon = var_11630_epsilon_0, x = var_11629_cast_fp16)[name = string("op_11630_cast_fp16")]; + tensor var_11631_cast_fp16 = mul(x = x_351_cast_fp16, y = var_11630_cast_fp16)[name = string("op_11631_cast_fp16")]; + tensor q_279_cast_fp16 = mul(x = var_11631_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_279_cast_fp16")]; + tensor var_11633 = const()[name = string("op_11633"), val = tensor([8, 128, 1, 1])]; + tensor x_353_cast_fp16 = reshape(shape = var_11633, x = k_277_cast_fp16)[name = string("x_353_cast_fp16")]; + tensor var_11636_cast_fp16 = mul(x = x_353_cast_fp16, y = x_353_cast_fp16)[name = string("op_11636_cast_fp16")]; + tensor variance_389_axes_0 = const()[name = string("variance_389_axes_0"), val = tensor([1])]; + bool variance_389_keep_dims_0 = const()[name = string("variance_389_keep_dims_0"), val = bool(true)]; + tensor variance_389_cast_fp16 = reduce_mean(axes = variance_389_axes_0, keep_dims = variance_389_keep_dims_0, x = var_11636_cast_fp16)[name = string("variance_389_cast_fp16")]; + fp16 var_11639_to_fp16 = const()[name = string("op_11639_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11640_cast_fp16 = add(x = variance_389_cast_fp16, y = var_11639_to_fp16)[name = string("op_11640_cast_fp16")]; + fp32 var_11641_epsilon_0 = const()[name = string("op_11641_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11641_cast_fp16 = rsqrt(epsilon = var_11641_epsilon_0, x = var_11640_cast_fp16)[name = string("op_11641_cast_fp16")]; + tensor var_11642_cast_fp16 = mul(x = x_353_cast_fp16, y = var_11641_cast_fp16)[name = string("op_11642_cast_fp16")]; + tensor k_279_cast_fp16 = mul(x = var_11642_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_279_cast_fp16")]; + tensor var_11644 = const()[name = string("op_11644"), val = tensor([1, 16, 128, 1])]; + tensor z_185_cast_fp16 = reshape(shape = var_11644, x = q_279_cast_fp16)[name = string("z_185_cast_fp16")]; + tensor var_11646 = const()[name = string("op_11646"), val = tensor([1, 8, 128, 1])]; + tensor z_187_cast_fp16 = reshape(shape = var_11646, x = k_279_cast_fp16)[name = string("z_187_cast_fp16")]; + tensor z1_185_begin_0 = const()[name = string("z1_185_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_185_end_0 = const()[name = string("z1_185_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_185_end_mask_0 = const()[name = string("z1_185_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_185_cast_fp16 = slice_by_index(begin = z1_185_begin_0, end = z1_185_end_0, end_mask = z1_185_end_mask_0, x = z_185_cast_fp16)[name = string("z1_185_cast_fp16")]; + tensor z2_185_begin_0 = const()[name = string("z2_185_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_185_end_0 = const()[name = string("z2_185_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_185_end_mask_0 = const()[name = string("z2_185_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_185_cast_fp16 = slice_by_index(begin = z2_185_begin_0, end = z2_185_end_0, end_mask = z2_185_end_mask_0, x = z_185_cast_fp16)[name = string("z2_185_cast_fp16")]; + tensor var_11654_cast_fp16 = mul(x = z_185_cast_fp16, y = cos_91_to_fp16)[name = string("op_11654_cast_fp16")]; + fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11655_cast_fp16 = mul(x = z2_185_cast_fp16, y = const_102_promoted_to_fp16)[name = string("op_11655_cast_fp16")]; + bool var_11657_interleave_0 = const()[name = string("op_11657_interleave_0"), val = bool(false)]; + tensor var_11657_cast_fp16 = concat(axis = var_11563, interleave = var_11657_interleave_0, values = (var_11655_cast_fp16, z1_185_cast_fp16))[name = string("op_11657_cast_fp16")]; + tensor var_11658_cast_fp16 = mul(x = var_11657_cast_fp16, y = sin_91_to_fp16)[name = string("op_11658_cast_fp16")]; + tensor q_281_cast_fp16 = add(x = var_11654_cast_fp16, y = var_11658_cast_fp16)[name = string("q_281_cast_fp16")]; + tensor z1_187_begin_0 = const()[name = string("z1_187_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_187_end_0 = const()[name = string("z1_187_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_187_end_mask_0 = const()[name = string("z1_187_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_187_cast_fp16 = slice_by_index(begin = z1_187_begin_0, end = z1_187_end_0, end_mask = z1_187_end_mask_0, x = z_187_cast_fp16)[name = string("z1_187_cast_fp16")]; + tensor z2_187_begin_0 = const()[name = string("z2_187_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_187_end_0 = const()[name = string("z2_187_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_187_end_mask_0 = const()[name = string("z2_187_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_187_cast_fp16 = slice_by_index(begin = z2_187_begin_0, end = z2_187_end_0, end_mask = z2_187_end_mask_0, x = z_187_cast_fp16)[name = string("z2_187_cast_fp16")]; + tensor var_11666_cast_fp16 = mul(x = z_187_cast_fp16, y = cos_91_to_fp16)[name = string("op_11666_cast_fp16")]; + fp16 const_103_promoted_to_fp16 = const()[name = string("const_103_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11667_cast_fp16 = mul(x = z2_187_cast_fp16, y = const_103_promoted_to_fp16)[name = string("op_11667_cast_fp16")]; + bool var_11669_interleave_0 = const()[name = string("op_11669_interleave_0"), val = bool(false)]; + tensor var_11669_cast_fp16 = concat(axis = var_11563, interleave = var_11669_interleave_0, values = (var_11667_cast_fp16, z1_187_cast_fp16))[name = string("op_11669_cast_fp16")]; + tensor var_11670_cast_fp16 = mul(x = var_11669_cast_fp16, y = sin_91_to_fp16)[name = string("op_11670_cast_fp16")]; + tensor k_281_cast_fp16 = add(x = var_11666_cast_fp16, y = var_11670_cast_fp16)[name = string("k_281_cast_fp16")]; + tensor var_11672 = const()[name = string("op_11672"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_93_cast_fp16 = reshape(shape = var_11672, x = k_281_cast_fp16)[name = string("cur_key_93_cast_fp16")]; + tensor var_11674_to_fp16 = const()[name = string("op_11674_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641216)))]; + tensor var_11675_cast_fp16 = mul(x = key_cache_93_cast_fp16, y = var_11674_to_fp16)[name = string("op_11675_cast_fp16")]; + tensor upd_93_to_fp16 = const()[name = string("upd_93_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641344)))]; + tensor var_11676_cast_fp16 = mul(x = cur_key_93_cast_fp16, y = upd_93_to_fp16)[name = string("op_11676_cast_fp16")]; + tensor key_93_cast_fp16 = add(x = var_11675_cast_fp16, y = var_11676_cast_fp16)[name = string("key_93_cast_fp16")]; + tensor var_11678_to_fp16 = const()[name = string("op_11678_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641216)))]; + tensor var_11679_cast_fp16 = mul(x = value_cache_93_cast_fp16, y = var_11678_to_fp16)[name = string("op_11679_cast_fp16")]; + tensor var_11680_cast_fp16 = mul(x = v_93_cast_fp16, y = upd_93_to_fp16)[name = string("op_11680_cast_fp16")]; + tensor value_93_cast_fp16 = add(x = var_11679_cast_fp16, y = var_11680_cast_fp16)[name = string("value_93_cast_fp16")]; + tensor var_11682 = const()[name = string("op_11682"), val = tensor([1, 8, 128, 16])]; + tensor kh_185_cast_fp16 = reshape(shape = var_11682, x = key_93_cast_fp16)[name = string("kh_185_cast_fp16")]; + tensor var_11684 = const()[name = string("op_11684"), val = tensor([1, 8, 128, 16])]; + tensor vh_185_cast_fp16 = reshape(shape = var_11684, x = value_93_cast_fp16)[name = string("vh_185_cast_fp16")]; + tensor transpose_184_perm_0 = const()[name = string("transpose_184_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_92_reps_0 = const()[name = string("tile_92_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_184_cast_fp16 = transpose(perm = transpose_184_perm_0, x = kh_185_cast_fp16)[name = string("transpose_203")]; + tensor tile_92_cast_fp16 = tile(reps = tile_92_reps_0, x = transpose_184_cast_fp16)[name = string("tile_92_cast_fp16")]; + tensor concat_232 = const()[name = string("concat_232"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_184_cast_fp16 = reshape(shape = concat_232, x = tile_92_cast_fp16)[name = string("reshape_184_cast_fp16")]; + tensor transpose_185_perm_0 = const()[name = string("transpose_185_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_233 = const()[name = string("concat_233"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_185_cast_fp16 = transpose(perm = transpose_185_perm_0, x = reshape_184_cast_fp16)[name = string("transpose_202")]; + tensor reshape_185_cast_fp16 = reshape(shape = concat_233, x = transpose_185_cast_fp16)[name = string("reshape_185_cast_fp16")]; + tensor transpose_186_perm_0 = const()[name = string("transpose_186_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_93_reps_0 = const()[name = string("tile_93_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_186_cast_fp16 = transpose(perm = transpose_186_perm_0, x = vh_185_cast_fp16)[name = string("transpose_201")]; + tensor tile_93_cast_fp16 = tile(reps = tile_93_reps_0, x = transpose_186_cast_fp16)[name = string("tile_93_cast_fp16")]; + tensor concat_234 = const()[name = string("concat_234"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_186_cast_fp16 = reshape(shape = concat_234, x = tile_93_cast_fp16)[name = string("reshape_186_cast_fp16")]; + tensor transpose_187_perm_0 = const()[name = string("transpose_187_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_235 = const()[name = string("concat_235"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_187_cast_fp16 = transpose(perm = transpose_187_perm_0, x = reshape_186_cast_fp16)[name = string("transpose_200")]; + tensor reshape_187_cast_fp16 = reshape(shape = concat_235, x = transpose_187_cast_fp16)[name = string("reshape_187_cast_fp16")]; + fp16 var_11688_to_fp16 = const()[name = string("op_11688_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_11689_cast_fp16 = mul(x = q_281_cast_fp16, y = var_11688_to_fp16)[name = string("op_11689_cast_fp16")]; + tensor transpose_501_perm_0 = const()[name = string("transpose_501_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_201_transpose_x_1 = const()[name = string("w_201_transpose_x_1"), val = bool(true)]; + bool w_201_transpose_y_1 = const()[name = string("w_201_transpose_y_1"), val = bool(false)]; + tensor transpose_501_cast_fp16 = transpose(perm = transpose_501_perm_0, x = reshape_185_cast_fp16)[name = string("transpose_199")]; + tensor w_201_cast_fp16 = matmul(transpose_x = w_201_transpose_x_1, transpose_y = w_201_transpose_y_1, x = var_11689_cast_fp16, y = transpose_501_cast_fp16)[name = string("w_201_cast_fp16")]; + tensor pad_93_to_fp16 = const()[name = string("pad_93_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641472)))]; + tensor var_11692_cast_fp16 = add(x = w_201_cast_fp16, y = pad_93_to_fp16)[name = string("op_11692_cast_fp16")]; + tensor w_203_cast_fp16 = softmax(axis = var_11567, x = var_11692_cast_fp16)[name = string("w_203_cast_fp16")]; + tensor transpose_502_perm_0 = const()[name = string("transpose_502_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_93_transpose_x_1 = const()[name = string("attn_93_transpose_x_1"), val = bool(false)]; + bool attn_93_transpose_y_1 = const()[name = string("attn_93_transpose_y_1"), val = bool(true)]; + tensor transpose_502_cast_fp16 = transpose(perm = transpose_502_perm_0, x = reshape_187_cast_fp16)[name = string("transpose_198")]; + tensor attn_93_cast_fp16 = matmul(transpose_x = attn_93_transpose_x_1, transpose_y = attn_93_transpose_y_1, x = transpose_502_cast_fp16, y = w_203_cast_fp16)[name = string("attn_93_cast_fp16")]; + tensor var_11696 = const()[name = string("op_11696"), val = tensor([1, 2048, 1, 1])]; + tensor input_495_cast_fp16 = reshape(shape = var_11696, x = attn_93_cast_fp16)[name = string("input_495_cast_fp16")]; + string attn_output_93_pad_type_0 = const()[name = string("attn_output_93_pad_type_0"), val = string("valid")]; + tensor attn_output_93_strides_0 = const()[name = string("attn_output_93_strides_0"), val = tensor([1, 1])]; + tensor attn_output_93_pad_0 = const()[name = string("attn_output_93_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_93_dilations_0 = const()[name = string("attn_output_93_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_93_groups_0 = const()[name = string("attn_output_93_groups_0"), val = int32(1)]; + tensor attn_output_93_cast_fp16 = conv(dilations = attn_output_93_dilations_0, groups = attn_output_93_groups_0, pad = attn_output_93_pad_0, pad_type = attn_output_93_pad_type_0, strides = attn_output_93_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_495_cast_fp16)[name = string("attn_output_93_cast_fp16")]; + tensor x_355_cast_fp16 = add(x = x_349_cast_fp16, y = attn_output_93_cast_fp16)[name = string("x_355_cast_fp16")]; + tensor var_11710_cast_fp16 = mul(x = x_355_cast_fp16, y = x_355_cast_fp16)[name = string("op_11710_cast_fp16")]; + tensor variance_391_axes_0 = const()[name = string("variance_391_axes_0"), val = tensor([1])]; + bool variance_391_keep_dims_0 = const()[name = string("variance_391_keep_dims_0"), val = bool(true)]; + tensor variance_391_cast_fp16 = reduce_mean(axes = variance_391_axes_0, keep_dims = variance_391_keep_dims_0, x = var_11710_cast_fp16)[name = string("variance_391_cast_fp16")]; + fp16 var_11713_to_fp16 = const()[name = string("op_11713_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11714_cast_fp16 = add(x = variance_391_cast_fp16, y = var_11713_to_fp16)[name = string("op_11714_cast_fp16")]; + fp32 var_11715_epsilon_0 = const()[name = string("op_11715_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11715_cast_fp16 = rsqrt(epsilon = var_11715_epsilon_0, x = var_11714_cast_fp16)[name = string("op_11715_cast_fp16")]; + tensor var_11716_cast_fp16 = mul(x = x_355_cast_fp16, y = var_11715_cast_fp16)[name = string("op_11716_cast_fp16")]; + tensor input_497_cast_fp16 = mul(x = var_11716_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_497_cast_fp16")]; + string input_499_pad_type_0 = const()[name = string("input_499_pad_type_0"), val = string("valid")]; + tensor input_499_strides_0 = const()[name = string("input_499_strides_0"), val = tensor([1, 1])]; + tensor input_499_pad_0 = const()[name = string("input_499_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_499_dilations_0 = const()[name = string("input_499_dilations_0"), val = tensor([1, 1])]; + int32 input_499_groups_0 = const()[name = string("input_499_groups_0"), val = int32(1)]; + tensor input_499_cast_fp16 = conv(dilations = input_499_dilations_0, groups = input_499_groups_0, pad = input_499_pad_0, pad_type = input_499_pad_type_0, strides = input_499_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_497_cast_fp16)[name = string("input_499_cast_fp16")]; + tensor var_11724_cast_fp16 = silu(x = input_499_cast_fp16)[name = string("op_11724_cast_fp16")]; + string var_11730_pad_type_0 = const()[name = string("op_11730_pad_type_0"), val = string("valid")]; + tensor var_11730_strides_0 = const()[name = string("op_11730_strides_0"), val = tensor([1, 1])]; + tensor var_11730_pad_0 = const()[name = string("op_11730_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11730_dilations_0 = const()[name = string("op_11730_dilations_0"), val = tensor([1, 1])]; + int32 var_11730_groups_0 = const()[name = string("op_11730_groups_0"), val = int32(1)]; + tensor var_11730_cast_fp16 = conv(dilations = var_11730_dilations_0, groups = var_11730_groups_0, pad = var_11730_pad_0, pad_type = var_11730_pad_type_0, strides = var_11730_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_497_cast_fp16)[name = string("op_11730_cast_fp16")]; + tensor input_501_cast_fp16 = mul(x = var_11724_cast_fp16, y = var_11730_cast_fp16)[name = string("input_501_cast_fp16")]; + string h_93_pad_type_0 = const()[name = string("h_93_pad_type_0"), val = string("valid")]; + tensor h_93_strides_0 = const()[name = string("h_93_strides_0"), val = tensor([1, 1])]; + tensor h_93_pad_0 = const()[name = string("h_93_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_93_dilations_0 = const()[name = string("h_93_dilations_0"), val = tensor([1, 1])]; + int32 h_93_groups_0 = const()[name = string("h_93_groups_0"), val = int32(1)]; + tensor h_93_cast_fp16 = conv(dilations = h_93_dilations_0, groups = h_93_groups_0, pad = h_93_pad_0, pad_type = h_93_pad_type_0, strides = h_93_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_501_cast_fp16)[name = string("h_93_cast_fp16")]; + tensor x_357_cast_fp16 = add(x = x_355_cast_fp16, y = h_93_cast_fp16)[name = string("x_357_cast_fp16")]; + tensor key_cache_95_begin_0 = const()[name = string("key_cache_95_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_95_end_0 = const()[name = string("key_cache_95_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_95_end_mask_0 = const()[name = string("key_cache_95_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_95_cast_fp16 = slice_by_index(begin = key_cache_95_begin_0, end = key_cache_95_end_0, end_mask = key_cache_95_end_mask_0, x = layer_key_caches_19_cast_fp16)[name = string("key_cache_95_cast_fp16")]; + tensor value_cache_95_begin_0 = const()[name = string("value_cache_95_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_95_end_0 = const()[name = string("value_cache_95_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_95_end_mask_0 = const()[name = string("value_cache_95_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_95_cast_fp16 = slice_by_index(begin = value_cache_95_begin_0, end = value_cache_95_end_0, end_mask = value_cache_95_end_mask_0, x = layer_value_caches_19_cast_fp16)[name = string("value_cache_95_cast_fp16")]; + int32 var_11783 = const()[name = string("op_11783"), val = int32(2)]; + int32 var_11787 = const()[name = string("op_11787"), val = int32(3)]; + tensor var_11802_cast_fp16 = mul(x = x_357_cast_fp16, y = x_357_cast_fp16)[name = string("op_11802_cast_fp16")]; + tensor variance_393_axes_0 = const()[name = string("variance_393_axes_0"), val = tensor([1])]; + bool variance_393_keep_dims_0 = const()[name = string("variance_393_keep_dims_0"), val = bool(true)]; + tensor variance_393_cast_fp16 = reduce_mean(axes = variance_393_axes_0, keep_dims = variance_393_keep_dims_0, x = var_11802_cast_fp16)[name = string("variance_393_cast_fp16")]; + fp16 var_11805_to_fp16 = const()[name = string("op_11805_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11806_cast_fp16 = add(x = variance_393_cast_fp16, y = var_11805_to_fp16)[name = string("op_11806_cast_fp16")]; + fp32 var_11807_epsilon_0 = const()[name = string("op_11807_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11807_cast_fp16 = rsqrt(epsilon = var_11807_epsilon_0, x = var_11806_cast_fp16)[name = string("op_11807_cast_fp16")]; + tensor var_11808_cast_fp16 = mul(x = x_357_cast_fp16, y = var_11807_cast_fp16)[name = string("op_11808_cast_fp16")]; + tensor input_503_cast_fp16 = mul(x = var_11808_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_503_cast_fp16")]; + string q_283_pad_type_0 = const()[name = string("q_283_pad_type_0"), val = string("valid")]; + tensor q_283_strides_0 = const()[name = string("q_283_strides_0"), val = tensor([1, 1])]; + tensor q_283_pad_0 = const()[name = string("q_283_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_283_dilations_0 = const()[name = string("q_283_dilations_0"), val = tensor([1, 1])]; + int32 q_283_groups_0 = const()[name = string("q_283_groups_0"), val = int32(1)]; + tensor q_283_cast_fp16 = conv(dilations = q_283_dilations_0, groups = q_283_groups_0, pad = q_283_pad_0, pad_type = q_283_pad_type_0, strides = q_283_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_503_cast_fp16)[name = string("q_283_cast_fp16")]; + string k_283_pad_type_0 = const()[name = string("k_283_pad_type_0"), val = string("valid")]; + tensor k_283_strides_0 = const()[name = string("k_283_strides_0"), val = tensor([1, 1])]; + tensor k_283_pad_0 = const()[name = string("k_283_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_283_dilations_0 = const()[name = string("k_283_dilations_0"), val = tensor([1, 1])]; + int32 k_283_groups_0 = const()[name = string("k_283_groups_0"), val = int32(1)]; + tensor k_283_cast_fp16 = conv(dilations = k_283_dilations_0, groups = k_283_groups_0, pad = k_283_pad_0, pad_type = k_283_pad_type_0, strides = k_283_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_503_cast_fp16)[name = string("k_283_cast_fp16")]; + string v_95_pad_type_0 = const()[name = string("v_95_pad_type_0"), val = string("valid")]; + tensor v_95_strides_0 = const()[name = string("v_95_strides_0"), val = tensor([1, 1])]; + tensor v_95_pad_0 = const()[name = string("v_95_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_95_dilations_0 = const()[name = string("v_95_dilations_0"), val = tensor([1, 1])]; + int32 v_95_groups_0 = const()[name = string("v_95_groups_0"), val = int32(1)]; + tensor v_95_cast_fp16 = conv(dilations = v_95_dilations_0, groups = v_95_groups_0, pad = v_95_pad_0, pad_type = v_95_pad_type_0, strides = v_95_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_503_cast_fp16)[name = string("v_95_cast_fp16")]; + tensor var_11842 = const()[name = string("op_11842"), val = tensor([16, 128, 1, 1])]; + tensor x_359_cast_fp16 = reshape(shape = var_11842, x = q_283_cast_fp16)[name = string("x_359_cast_fp16")]; + tensor var_11845_cast_fp16 = mul(x = x_359_cast_fp16, y = x_359_cast_fp16)[name = string("op_11845_cast_fp16")]; + tensor variance_395_axes_0 = const()[name = string("variance_395_axes_0"), val = tensor([1])]; + bool variance_395_keep_dims_0 = const()[name = string("variance_395_keep_dims_0"), val = bool(true)]; + tensor variance_395_cast_fp16 = reduce_mean(axes = variance_395_axes_0, keep_dims = variance_395_keep_dims_0, x = var_11845_cast_fp16)[name = string("variance_395_cast_fp16")]; + fp16 var_11848_to_fp16 = const()[name = string("op_11848_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11849_cast_fp16 = add(x = variance_395_cast_fp16, y = var_11848_to_fp16)[name = string("op_11849_cast_fp16")]; + fp32 var_11850_epsilon_0 = const()[name = string("op_11850_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11850_cast_fp16 = rsqrt(epsilon = var_11850_epsilon_0, x = var_11849_cast_fp16)[name = string("op_11850_cast_fp16")]; + tensor var_11851_cast_fp16 = mul(x = x_359_cast_fp16, y = var_11850_cast_fp16)[name = string("op_11851_cast_fp16")]; + tensor q_285_cast_fp16 = mul(x = var_11851_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_285_cast_fp16")]; + tensor var_11853 = const()[name = string("op_11853"), val = tensor([8, 128, 1, 1])]; + tensor x_361_cast_fp16 = reshape(shape = var_11853, x = k_283_cast_fp16)[name = string("x_361_cast_fp16")]; + tensor var_11856_cast_fp16 = mul(x = x_361_cast_fp16, y = x_361_cast_fp16)[name = string("op_11856_cast_fp16")]; + tensor variance_397_axes_0 = const()[name = string("variance_397_axes_0"), val = tensor([1])]; + bool variance_397_keep_dims_0 = const()[name = string("variance_397_keep_dims_0"), val = bool(true)]; + tensor variance_397_cast_fp16 = reduce_mean(axes = variance_397_axes_0, keep_dims = variance_397_keep_dims_0, x = var_11856_cast_fp16)[name = string("variance_397_cast_fp16")]; + fp16 var_11859_to_fp16 = const()[name = string("op_11859_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11860_cast_fp16 = add(x = variance_397_cast_fp16, y = var_11859_to_fp16)[name = string("op_11860_cast_fp16")]; + fp32 var_11861_epsilon_0 = const()[name = string("op_11861_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11861_cast_fp16 = rsqrt(epsilon = var_11861_epsilon_0, x = var_11860_cast_fp16)[name = string("op_11861_cast_fp16")]; + tensor var_11862_cast_fp16 = mul(x = x_361_cast_fp16, y = var_11861_cast_fp16)[name = string("op_11862_cast_fp16")]; + tensor k_285_cast_fp16 = mul(x = var_11862_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_285_cast_fp16")]; + tensor var_11864 = const()[name = string("op_11864"), val = tensor([1, 16, 128, 1])]; + tensor z_189_cast_fp16 = reshape(shape = var_11864, x = q_285_cast_fp16)[name = string("z_189_cast_fp16")]; + tensor var_11866 = const()[name = string("op_11866"), val = tensor([1, 8, 128, 1])]; + tensor z_191_cast_fp16 = reshape(shape = var_11866, x = k_285_cast_fp16)[name = string("z_191_cast_fp16")]; + tensor z1_189_begin_0 = const()[name = string("z1_189_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_189_end_0 = const()[name = string("z1_189_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_189_end_mask_0 = const()[name = string("z1_189_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_189_cast_fp16 = slice_by_index(begin = z1_189_begin_0, end = z1_189_end_0, end_mask = z1_189_end_mask_0, x = z_189_cast_fp16)[name = string("z1_189_cast_fp16")]; + tensor z2_189_begin_0 = const()[name = string("z2_189_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_189_end_0 = const()[name = string("z2_189_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_189_end_mask_0 = const()[name = string("z2_189_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_189_cast_fp16 = slice_by_index(begin = z2_189_begin_0, end = z2_189_end_0, end_mask = z2_189_end_mask_0, x = z_189_cast_fp16)[name = string("z2_189_cast_fp16")]; + tensor var_11874_cast_fp16 = mul(x = z_189_cast_fp16, y = cos_91_to_fp16)[name = string("op_11874_cast_fp16")]; + fp16 const_104_promoted_to_fp16 = const()[name = string("const_104_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11875_cast_fp16 = mul(x = z2_189_cast_fp16, y = const_104_promoted_to_fp16)[name = string("op_11875_cast_fp16")]; + bool var_11877_interleave_0 = const()[name = string("op_11877_interleave_0"), val = bool(false)]; + tensor var_11877_cast_fp16 = concat(axis = var_11783, interleave = var_11877_interleave_0, values = (var_11875_cast_fp16, z1_189_cast_fp16))[name = string("op_11877_cast_fp16")]; + tensor var_11878_cast_fp16 = mul(x = var_11877_cast_fp16, y = sin_91_to_fp16)[name = string("op_11878_cast_fp16")]; + tensor q_287_cast_fp16 = add(x = var_11874_cast_fp16, y = var_11878_cast_fp16)[name = string("q_287_cast_fp16")]; + tensor z1_191_begin_0 = const()[name = string("z1_191_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_191_end_0 = const()[name = string("z1_191_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_191_end_mask_0 = const()[name = string("z1_191_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_191_cast_fp16 = slice_by_index(begin = z1_191_begin_0, end = z1_191_end_0, end_mask = z1_191_end_mask_0, x = z_191_cast_fp16)[name = string("z1_191_cast_fp16")]; + tensor z2_191_begin_0 = const()[name = string("z2_191_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_191_end_0 = const()[name = string("z2_191_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_191_end_mask_0 = const()[name = string("z2_191_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_191_cast_fp16 = slice_by_index(begin = z2_191_begin_0, end = z2_191_end_0, end_mask = z2_191_end_mask_0, x = z_191_cast_fp16)[name = string("z2_191_cast_fp16")]; + tensor var_11886_cast_fp16 = mul(x = z_191_cast_fp16, y = cos_91_to_fp16)[name = string("op_11886_cast_fp16")]; + fp16 const_105_promoted_to_fp16 = const()[name = string("const_105_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11887_cast_fp16 = mul(x = z2_191_cast_fp16, y = const_105_promoted_to_fp16)[name = string("op_11887_cast_fp16")]; + bool var_11889_interleave_0 = const()[name = string("op_11889_interleave_0"), val = bool(false)]; + tensor var_11889_cast_fp16 = concat(axis = var_11783, interleave = var_11889_interleave_0, values = (var_11887_cast_fp16, z1_191_cast_fp16))[name = string("op_11889_cast_fp16")]; + tensor var_11890_cast_fp16 = mul(x = var_11889_cast_fp16, y = sin_91_to_fp16)[name = string("op_11890_cast_fp16")]; + tensor k_287_cast_fp16 = add(x = var_11886_cast_fp16, y = var_11890_cast_fp16)[name = string("k_287_cast_fp16")]; + tensor var_11892 = const()[name = string("op_11892"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_95_cast_fp16 = reshape(shape = var_11892, x = k_287_cast_fp16)[name = string("cur_key_95_cast_fp16")]; + tensor var_11894_to_fp16 = const()[name = string("op_11894_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641216)))]; + tensor var_11895_cast_fp16 = mul(x = key_cache_95_cast_fp16, y = var_11894_to_fp16)[name = string("op_11895_cast_fp16")]; + tensor upd_95_to_fp16 = const()[name = string("upd_95_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641344)))]; + tensor var_11896_cast_fp16 = mul(x = cur_key_95_cast_fp16, y = upd_95_to_fp16)[name = string("op_11896_cast_fp16")]; + tensor key_95_cast_fp16 = add(x = var_11895_cast_fp16, y = var_11896_cast_fp16)[name = string("key_95_cast_fp16")]; + tensor var_11898_to_fp16 = const()[name = string("op_11898_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641216)))]; + tensor var_11899_cast_fp16 = mul(x = value_cache_95_cast_fp16, y = var_11898_to_fp16)[name = string("op_11899_cast_fp16")]; + tensor var_11900_cast_fp16 = mul(x = v_95_cast_fp16, y = upd_95_to_fp16)[name = string("op_11900_cast_fp16")]; + tensor value_95_cast_fp16 = add(x = var_11899_cast_fp16, y = var_11900_cast_fp16)[name = string("value_95_cast_fp16")]; + tensor var_11902 = const()[name = string("op_11902"), val = tensor([1, 8, 128, 16])]; + tensor kh_189_cast_fp16 = reshape(shape = var_11902, x = key_95_cast_fp16)[name = string("kh_189_cast_fp16")]; + tensor var_11904 = const()[name = string("op_11904"), val = tensor([1, 8, 128, 16])]; + tensor vh_189_cast_fp16 = reshape(shape = var_11904, x = value_95_cast_fp16)[name = string("vh_189_cast_fp16")]; + tensor transpose_188_perm_0 = const()[name = string("transpose_188_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_94_reps_0 = const()[name = string("tile_94_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_188_cast_fp16 = transpose(perm = transpose_188_perm_0, x = kh_189_cast_fp16)[name = string("transpose_197")]; + tensor tile_94_cast_fp16 = tile(reps = tile_94_reps_0, x = transpose_188_cast_fp16)[name = string("tile_94_cast_fp16")]; + tensor concat_236 = const()[name = string("concat_236"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_188_cast_fp16 = reshape(shape = concat_236, x = tile_94_cast_fp16)[name = string("reshape_188_cast_fp16")]; + tensor transpose_189_perm_0 = const()[name = string("transpose_189_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_237 = const()[name = string("concat_237"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_189_cast_fp16 = transpose(perm = transpose_189_perm_0, x = reshape_188_cast_fp16)[name = string("transpose_196")]; + tensor reshape_189_cast_fp16 = reshape(shape = concat_237, x = transpose_189_cast_fp16)[name = string("reshape_189_cast_fp16")]; + tensor transpose_190_perm_0 = const()[name = string("transpose_190_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_95_reps_0 = const()[name = string("tile_95_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_190_cast_fp16 = transpose(perm = transpose_190_perm_0, x = vh_189_cast_fp16)[name = string("transpose_195")]; + tensor tile_95_cast_fp16 = tile(reps = tile_95_reps_0, x = transpose_190_cast_fp16)[name = string("tile_95_cast_fp16")]; + tensor concat_238 = const()[name = string("concat_238"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_190_cast_fp16 = reshape(shape = concat_238, x = tile_95_cast_fp16)[name = string("reshape_190_cast_fp16")]; + tensor transpose_191_perm_0 = const()[name = string("transpose_191_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_239 = const()[name = string("concat_239"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_191_cast_fp16 = transpose(perm = transpose_191_perm_0, x = reshape_190_cast_fp16)[name = string("transpose_194")]; + tensor reshape_191_cast_fp16 = reshape(shape = concat_239, x = transpose_191_cast_fp16)[name = string("reshape_191_cast_fp16")]; + fp16 var_11908_to_fp16 = const()[name = string("op_11908_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_11909_cast_fp16 = mul(x = q_287_cast_fp16, y = var_11908_to_fp16)[name = string("op_11909_cast_fp16")]; + tensor transpose_505_perm_0 = const()[name = string("transpose_505_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_205_transpose_x_1 = const()[name = string("w_205_transpose_x_1"), val = bool(true)]; + bool w_205_transpose_y_1 = const()[name = string("w_205_transpose_y_1"), val = bool(false)]; + tensor transpose_505_cast_fp16 = transpose(perm = transpose_505_perm_0, x = reshape_189_cast_fp16)[name = string("transpose_193")]; + tensor w_205_cast_fp16 = matmul(transpose_x = w_205_transpose_x_1, transpose_y = w_205_transpose_y_1, x = var_11909_cast_fp16, y = transpose_505_cast_fp16)[name = string("w_205_cast_fp16")]; + tensor pad_95_to_fp16 = const()[name = string("pad_95_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641472)))]; + tensor var_11912_cast_fp16 = add(x = w_205_cast_fp16, y = pad_95_to_fp16)[name = string("op_11912_cast_fp16")]; + tensor w_207_cast_fp16 = softmax(axis = var_11787, x = var_11912_cast_fp16)[name = string("w_207_cast_fp16")]; + tensor transpose_506_perm_0 = const()[name = string("transpose_506_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_95_transpose_x_1 = const()[name = string("attn_95_transpose_x_1"), val = bool(false)]; + bool attn_95_transpose_y_1 = const()[name = string("attn_95_transpose_y_1"), val = bool(true)]; + tensor transpose_506_cast_fp16 = transpose(perm = transpose_506_perm_0, x = reshape_191_cast_fp16)[name = string("transpose_192")]; + tensor attn_95_cast_fp16 = matmul(transpose_x = attn_95_transpose_x_1, transpose_y = attn_95_transpose_y_1, x = transpose_506_cast_fp16, y = w_207_cast_fp16)[name = string("attn_95_cast_fp16")]; + tensor var_11916 = const()[name = string("op_11916"), val = tensor([1, 2048, 1, 1])]; + tensor input_505_cast_fp16 = reshape(shape = var_11916, x = attn_95_cast_fp16)[name = string("input_505_cast_fp16")]; + string attn_output_95_pad_type_0 = const()[name = string("attn_output_95_pad_type_0"), val = string("valid")]; + tensor attn_output_95_strides_0 = const()[name = string("attn_output_95_strides_0"), val = tensor([1, 1])]; + tensor attn_output_95_pad_0 = const()[name = string("attn_output_95_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_95_dilations_0 = const()[name = string("attn_output_95_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_95_groups_0 = const()[name = string("attn_output_95_groups_0"), val = int32(1)]; + tensor attn_output_95_cast_fp16 = conv(dilations = attn_output_95_dilations_0, groups = attn_output_95_groups_0, pad = attn_output_95_pad_0, pad_type = attn_output_95_pad_type_0, strides = attn_output_95_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_505_cast_fp16)[name = string("attn_output_95_cast_fp16")]; + tensor x_363_cast_fp16 = add(x = x_357_cast_fp16, y = attn_output_95_cast_fp16)[name = string("x_363_cast_fp16")]; + tensor var_11930_cast_fp16 = mul(x = x_363_cast_fp16, y = x_363_cast_fp16)[name = string("op_11930_cast_fp16")]; + tensor variance_399_axes_0 = const()[name = string("variance_399_axes_0"), val = tensor([1])]; + bool variance_399_keep_dims_0 = const()[name = string("variance_399_keep_dims_0"), val = bool(true)]; + tensor variance_399_cast_fp16 = reduce_mean(axes = variance_399_axes_0, keep_dims = variance_399_keep_dims_0, x = var_11930_cast_fp16)[name = string("variance_399_cast_fp16")]; + fp16 var_11933_to_fp16 = const()[name = string("op_11933_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11934_cast_fp16 = add(x = variance_399_cast_fp16, y = var_11933_to_fp16)[name = string("op_11934_cast_fp16")]; + fp32 var_11935_epsilon_0 = const()[name = string("op_11935_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11935_cast_fp16 = rsqrt(epsilon = var_11935_epsilon_0, x = var_11934_cast_fp16)[name = string("op_11935_cast_fp16")]; + tensor var_11936_cast_fp16 = mul(x = x_363_cast_fp16, y = var_11935_cast_fp16)[name = string("op_11936_cast_fp16")]; + tensor input_507_cast_fp16 = mul(x = var_11936_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_507_cast_fp16")]; + string input_509_pad_type_0 = const()[name = string("input_509_pad_type_0"), val = string("valid")]; + tensor input_509_strides_0 = const()[name = string("input_509_strides_0"), val = tensor([1, 1])]; + tensor input_509_pad_0 = const()[name = string("input_509_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_509_dilations_0 = const()[name = string("input_509_dilations_0"), val = tensor([1, 1])]; + int32 input_509_groups_0 = const()[name = string("input_509_groups_0"), val = int32(1)]; + tensor input_509_cast_fp16 = conv(dilations = input_509_dilations_0, groups = input_509_groups_0, pad = input_509_pad_0, pad_type = input_509_pad_type_0, strides = input_509_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_507_cast_fp16)[name = string("input_509_cast_fp16")]; + tensor var_11944_cast_fp16 = silu(x = input_509_cast_fp16)[name = string("op_11944_cast_fp16")]; + string var_11950_pad_type_0 = const()[name = string("op_11950_pad_type_0"), val = string("valid")]; + tensor var_11950_strides_0 = const()[name = string("op_11950_strides_0"), val = tensor([1, 1])]; + tensor var_11950_pad_0 = const()[name = string("op_11950_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11950_dilations_0 = const()[name = string("op_11950_dilations_0"), val = tensor([1, 1])]; + int32 var_11950_groups_0 = const()[name = string("op_11950_groups_0"), val = int32(1)]; + tensor var_11950_cast_fp16 = conv(dilations = var_11950_dilations_0, groups = var_11950_groups_0, pad = var_11950_pad_0, pad_type = var_11950_pad_type_0, strides = var_11950_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_507_cast_fp16)[name = string("op_11950_cast_fp16")]; + tensor input_511_cast_fp16 = mul(x = var_11944_cast_fp16, y = var_11950_cast_fp16)[name = string("input_511_cast_fp16")]; + string h_95_pad_type_0 = const()[name = string("h_95_pad_type_0"), val = string("valid")]; + tensor h_95_strides_0 = const()[name = string("h_95_strides_0"), val = tensor([1, 1])]; + tensor h_95_pad_0 = const()[name = string("h_95_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_95_dilations_0 = const()[name = string("h_95_dilations_0"), val = tensor([1, 1])]; + int32 h_95_groups_0 = const()[name = string("h_95_groups_0"), val = int32(1)]; + tensor h_95_cast_fp16 = conv(dilations = h_95_dilations_0, groups = h_95_groups_0, pad = h_95_pad_0, pad_type = h_95_pad_type_0, strides = h_95_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_511_cast_fp16)[name = string("h_95_cast_fp16")]; + tensor x_365_cast_fp16 = add(x = x_363_cast_fp16, y = h_95_cast_fp16)[name = string("x_365_cast_fp16")]; + tensor key_cache_97_begin_0 = const()[name = string("key_cache_97_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_97_end_0 = const()[name = string("key_cache_97_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_97_end_mask_0 = const()[name = string("key_cache_97_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_97_cast_fp16 = slice_by_index(begin = key_cache_97_begin_0, end = key_cache_97_end_0, end_mask = key_cache_97_end_mask_0, x = layer_key_caches_19_cast_fp16)[name = string("key_cache_97_cast_fp16")]; + tensor value_cache_97_begin_0 = const()[name = string("value_cache_97_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_97_end_0 = const()[name = string("value_cache_97_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_97_end_mask_0 = const()[name = string("value_cache_97_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_97_cast_fp16 = slice_by_index(begin = value_cache_97_begin_0, end = value_cache_97_end_0, end_mask = value_cache_97_end_mask_0, x = layer_value_caches_19_cast_fp16)[name = string("value_cache_97_cast_fp16")]; + int32 var_12003 = const()[name = string("op_12003"), val = int32(2)]; + int32 var_12007 = const()[name = string("op_12007"), val = int32(3)]; + tensor var_12022_cast_fp16 = mul(x = x_365_cast_fp16, y = x_365_cast_fp16)[name = string("op_12022_cast_fp16")]; + tensor variance_401_axes_0 = const()[name = string("variance_401_axes_0"), val = tensor([1])]; + bool variance_401_keep_dims_0 = const()[name = string("variance_401_keep_dims_0"), val = bool(true)]; + tensor variance_401_cast_fp16 = reduce_mean(axes = variance_401_axes_0, keep_dims = variance_401_keep_dims_0, x = var_12022_cast_fp16)[name = string("variance_401_cast_fp16")]; + fp16 var_12025_to_fp16 = const()[name = string("op_12025_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12026_cast_fp16 = add(x = variance_401_cast_fp16, y = var_12025_to_fp16)[name = string("op_12026_cast_fp16")]; + fp32 var_12027_epsilon_0 = const()[name = string("op_12027_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12027_cast_fp16 = rsqrt(epsilon = var_12027_epsilon_0, x = var_12026_cast_fp16)[name = string("op_12027_cast_fp16")]; + tensor var_12028_cast_fp16 = mul(x = x_365_cast_fp16, y = var_12027_cast_fp16)[name = string("op_12028_cast_fp16")]; + tensor input_513_cast_fp16 = mul(x = var_12028_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_513_cast_fp16")]; + string q_289_pad_type_0 = const()[name = string("q_289_pad_type_0"), val = string("valid")]; + tensor q_289_strides_0 = const()[name = string("q_289_strides_0"), val = tensor([1, 1])]; + tensor q_289_pad_0 = const()[name = string("q_289_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_289_dilations_0 = const()[name = string("q_289_dilations_0"), val = tensor([1, 1])]; + int32 q_289_groups_0 = const()[name = string("q_289_groups_0"), val = int32(1)]; + tensor q_289_cast_fp16 = conv(dilations = q_289_dilations_0, groups = q_289_groups_0, pad = q_289_pad_0, pad_type = q_289_pad_type_0, strides = q_289_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_513_cast_fp16)[name = string("q_289_cast_fp16")]; + string k_289_pad_type_0 = const()[name = string("k_289_pad_type_0"), val = string("valid")]; + tensor k_289_strides_0 = const()[name = string("k_289_strides_0"), val = tensor([1, 1])]; + tensor k_289_pad_0 = const()[name = string("k_289_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_289_dilations_0 = const()[name = string("k_289_dilations_0"), val = tensor([1, 1])]; + int32 k_289_groups_0 = const()[name = string("k_289_groups_0"), val = int32(1)]; + tensor k_289_cast_fp16 = conv(dilations = k_289_dilations_0, groups = k_289_groups_0, pad = k_289_pad_0, pad_type = k_289_pad_type_0, strides = k_289_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_513_cast_fp16)[name = string("k_289_cast_fp16")]; + string v_97_pad_type_0 = const()[name = string("v_97_pad_type_0"), val = string("valid")]; + tensor v_97_strides_0 = const()[name = string("v_97_strides_0"), val = tensor([1, 1])]; + tensor v_97_pad_0 = const()[name = string("v_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_97_dilations_0 = const()[name = string("v_97_dilations_0"), val = tensor([1, 1])]; + int32 v_97_groups_0 = const()[name = string("v_97_groups_0"), val = int32(1)]; + tensor v_97_cast_fp16 = conv(dilations = v_97_dilations_0, groups = v_97_groups_0, pad = v_97_pad_0, pad_type = v_97_pad_type_0, strides = v_97_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_513_cast_fp16)[name = string("v_97_cast_fp16")]; + tensor var_12062 = const()[name = string("op_12062"), val = tensor([16, 128, 1, 1])]; + tensor x_367_cast_fp16 = reshape(shape = var_12062, x = q_289_cast_fp16)[name = string("x_367_cast_fp16")]; + tensor var_12065_cast_fp16 = mul(x = x_367_cast_fp16, y = x_367_cast_fp16)[name = string("op_12065_cast_fp16")]; + tensor variance_403_axes_0 = const()[name = string("variance_403_axes_0"), val = tensor([1])]; + bool variance_403_keep_dims_0 = const()[name = string("variance_403_keep_dims_0"), val = bool(true)]; + tensor variance_403_cast_fp16 = reduce_mean(axes = variance_403_axes_0, keep_dims = variance_403_keep_dims_0, x = var_12065_cast_fp16)[name = string("variance_403_cast_fp16")]; + fp16 var_12068_to_fp16 = const()[name = string("op_12068_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12069_cast_fp16 = add(x = variance_403_cast_fp16, y = var_12068_to_fp16)[name = string("op_12069_cast_fp16")]; + fp32 var_12070_epsilon_0 = const()[name = string("op_12070_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12070_cast_fp16 = rsqrt(epsilon = var_12070_epsilon_0, x = var_12069_cast_fp16)[name = string("op_12070_cast_fp16")]; + tensor var_12071_cast_fp16 = mul(x = x_367_cast_fp16, y = var_12070_cast_fp16)[name = string("op_12071_cast_fp16")]; + tensor q_291_cast_fp16 = mul(x = var_12071_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_291_cast_fp16")]; + tensor var_12073 = const()[name = string("op_12073"), val = tensor([8, 128, 1, 1])]; + tensor x_369_cast_fp16 = reshape(shape = var_12073, x = k_289_cast_fp16)[name = string("x_369_cast_fp16")]; + tensor var_12076_cast_fp16 = mul(x = x_369_cast_fp16, y = x_369_cast_fp16)[name = string("op_12076_cast_fp16")]; + tensor variance_405_axes_0 = const()[name = string("variance_405_axes_0"), val = tensor([1])]; + bool variance_405_keep_dims_0 = const()[name = string("variance_405_keep_dims_0"), val = bool(true)]; + tensor variance_405_cast_fp16 = reduce_mean(axes = variance_405_axes_0, keep_dims = variance_405_keep_dims_0, x = var_12076_cast_fp16)[name = string("variance_405_cast_fp16")]; + fp16 var_12079_to_fp16 = const()[name = string("op_12079_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12080_cast_fp16 = add(x = variance_405_cast_fp16, y = var_12079_to_fp16)[name = string("op_12080_cast_fp16")]; + fp32 var_12081_epsilon_0 = const()[name = string("op_12081_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12081_cast_fp16 = rsqrt(epsilon = var_12081_epsilon_0, x = var_12080_cast_fp16)[name = string("op_12081_cast_fp16")]; + tensor var_12082_cast_fp16 = mul(x = x_369_cast_fp16, y = var_12081_cast_fp16)[name = string("op_12082_cast_fp16")]; + tensor k_291_cast_fp16 = mul(x = var_12082_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_291_cast_fp16")]; + tensor var_12084 = const()[name = string("op_12084"), val = tensor([1, 16, 128, 1])]; + tensor z_193_cast_fp16 = reshape(shape = var_12084, x = q_291_cast_fp16)[name = string("z_193_cast_fp16")]; + tensor var_12086 = const()[name = string("op_12086"), val = tensor([1, 8, 128, 1])]; + tensor z_195_cast_fp16 = reshape(shape = var_12086, x = k_291_cast_fp16)[name = string("z_195_cast_fp16")]; + tensor z1_193_begin_0 = const()[name = string("z1_193_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_193_end_0 = const()[name = string("z1_193_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_193_end_mask_0 = const()[name = string("z1_193_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_193_cast_fp16 = slice_by_index(begin = z1_193_begin_0, end = z1_193_end_0, end_mask = z1_193_end_mask_0, x = z_193_cast_fp16)[name = string("z1_193_cast_fp16")]; + tensor z2_193_begin_0 = const()[name = string("z2_193_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_193_end_0 = const()[name = string("z2_193_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_193_end_mask_0 = const()[name = string("z2_193_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_193_cast_fp16 = slice_by_index(begin = z2_193_begin_0, end = z2_193_end_0, end_mask = z2_193_end_mask_0, x = z_193_cast_fp16)[name = string("z2_193_cast_fp16")]; + tensor var_12094_cast_fp16 = mul(x = z_193_cast_fp16, y = cos_91_to_fp16)[name = string("op_12094_cast_fp16")]; + fp16 const_106_promoted_to_fp16 = const()[name = string("const_106_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12095_cast_fp16 = mul(x = z2_193_cast_fp16, y = const_106_promoted_to_fp16)[name = string("op_12095_cast_fp16")]; + bool var_12097_interleave_0 = const()[name = string("op_12097_interleave_0"), val = bool(false)]; + tensor var_12097_cast_fp16 = concat(axis = var_12003, interleave = var_12097_interleave_0, values = (var_12095_cast_fp16, z1_193_cast_fp16))[name = string("op_12097_cast_fp16")]; + tensor var_12098_cast_fp16 = mul(x = var_12097_cast_fp16, y = sin_91_to_fp16)[name = string("op_12098_cast_fp16")]; + tensor q_293_cast_fp16 = add(x = var_12094_cast_fp16, y = var_12098_cast_fp16)[name = string("q_293_cast_fp16")]; + tensor z1_195_begin_0 = const()[name = string("z1_195_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_195_end_0 = const()[name = string("z1_195_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_195_end_mask_0 = const()[name = string("z1_195_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_195_cast_fp16 = slice_by_index(begin = z1_195_begin_0, end = z1_195_end_0, end_mask = z1_195_end_mask_0, x = z_195_cast_fp16)[name = string("z1_195_cast_fp16")]; + tensor z2_195_begin_0 = const()[name = string("z2_195_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_195_end_0 = const()[name = string("z2_195_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_195_end_mask_0 = const()[name = string("z2_195_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_195_cast_fp16 = slice_by_index(begin = z2_195_begin_0, end = z2_195_end_0, end_mask = z2_195_end_mask_0, x = z_195_cast_fp16)[name = string("z2_195_cast_fp16")]; + tensor var_12106_cast_fp16 = mul(x = z_195_cast_fp16, y = cos_91_to_fp16)[name = string("op_12106_cast_fp16")]; + fp16 const_107_promoted_to_fp16 = const()[name = string("const_107_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12107_cast_fp16 = mul(x = z2_195_cast_fp16, y = const_107_promoted_to_fp16)[name = string("op_12107_cast_fp16")]; + bool var_12109_interleave_0 = const()[name = string("op_12109_interleave_0"), val = bool(false)]; + tensor var_12109_cast_fp16 = concat(axis = var_12003, interleave = var_12109_interleave_0, values = (var_12107_cast_fp16, z1_195_cast_fp16))[name = string("op_12109_cast_fp16")]; + tensor var_12110_cast_fp16 = mul(x = var_12109_cast_fp16, y = sin_91_to_fp16)[name = string("op_12110_cast_fp16")]; + tensor k_293_cast_fp16 = add(x = var_12106_cast_fp16, y = var_12110_cast_fp16)[name = string("k_293_cast_fp16")]; + tensor var_12112 = const()[name = string("op_12112"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_97_cast_fp16 = reshape(shape = var_12112, x = k_293_cast_fp16)[name = string("cur_key_97_cast_fp16")]; + tensor var_12114_to_fp16 = const()[name = string("op_12114_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641216)))]; + tensor var_12115_cast_fp16 = mul(x = key_cache_97_cast_fp16, y = var_12114_to_fp16)[name = string("op_12115_cast_fp16")]; + tensor upd_97_to_fp16 = const()[name = string("upd_97_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641344)))]; + tensor var_12116_cast_fp16 = mul(x = cur_key_97_cast_fp16, y = upd_97_to_fp16)[name = string("op_12116_cast_fp16")]; + tensor key_97_cast_fp16 = add(x = var_12115_cast_fp16, y = var_12116_cast_fp16)[name = string("key_97_cast_fp16")]; + tensor var_12118_to_fp16 = const()[name = string("op_12118_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641216)))]; + tensor var_12119_cast_fp16 = mul(x = value_cache_97_cast_fp16, y = var_12118_to_fp16)[name = string("op_12119_cast_fp16")]; + tensor var_12120_cast_fp16 = mul(x = v_97_cast_fp16, y = upd_97_to_fp16)[name = string("op_12120_cast_fp16")]; + tensor value_97_cast_fp16 = add(x = var_12119_cast_fp16, y = var_12120_cast_fp16)[name = string("value_97_cast_fp16")]; + tensor var_12122 = const()[name = string("op_12122"), val = tensor([1, 8, 128, 16])]; + tensor kh_193_cast_fp16 = reshape(shape = var_12122, x = key_97_cast_fp16)[name = string("kh_193_cast_fp16")]; + tensor var_12124 = const()[name = string("op_12124"), val = tensor([1, 8, 128, 16])]; + tensor vh_193_cast_fp16 = reshape(shape = var_12124, x = value_97_cast_fp16)[name = string("vh_193_cast_fp16")]; + tensor transpose_192_perm_0 = const()[name = string("transpose_192_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_96_reps_0 = const()[name = string("tile_96_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_192_cast_fp16 = transpose(perm = transpose_192_perm_0, x = kh_193_cast_fp16)[name = string("transpose_191")]; + tensor tile_96_cast_fp16 = tile(reps = tile_96_reps_0, x = transpose_192_cast_fp16)[name = string("tile_96_cast_fp16")]; + tensor concat_240 = const()[name = string("concat_240"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_192_cast_fp16 = reshape(shape = concat_240, x = tile_96_cast_fp16)[name = string("reshape_192_cast_fp16")]; + tensor transpose_193_perm_0 = const()[name = string("transpose_193_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_241 = const()[name = string("concat_241"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_193_cast_fp16 = transpose(perm = transpose_193_perm_0, x = reshape_192_cast_fp16)[name = string("transpose_190")]; + tensor reshape_193_cast_fp16 = reshape(shape = concat_241, x = transpose_193_cast_fp16)[name = string("reshape_193_cast_fp16")]; + tensor transpose_194_perm_0 = const()[name = string("transpose_194_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_97_reps_0 = const()[name = string("tile_97_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_194_cast_fp16 = transpose(perm = transpose_194_perm_0, x = vh_193_cast_fp16)[name = string("transpose_189")]; + tensor tile_97_cast_fp16 = tile(reps = tile_97_reps_0, x = transpose_194_cast_fp16)[name = string("tile_97_cast_fp16")]; + tensor concat_242 = const()[name = string("concat_242"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_194_cast_fp16 = reshape(shape = concat_242, x = tile_97_cast_fp16)[name = string("reshape_194_cast_fp16")]; + tensor transpose_195_perm_0 = const()[name = string("transpose_195_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_243 = const()[name = string("concat_243"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_195_cast_fp16 = transpose(perm = transpose_195_perm_0, x = reshape_194_cast_fp16)[name = string("transpose_188")]; + tensor reshape_195_cast_fp16 = reshape(shape = concat_243, x = transpose_195_cast_fp16)[name = string("reshape_195_cast_fp16")]; + fp16 var_12128_to_fp16 = const()[name = string("op_12128_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_12129_cast_fp16 = mul(x = q_293_cast_fp16, y = var_12128_to_fp16)[name = string("op_12129_cast_fp16")]; + tensor transpose_509_perm_0 = const()[name = string("transpose_509_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_209_transpose_x_1 = const()[name = string("w_209_transpose_x_1"), val = bool(true)]; + bool w_209_transpose_y_1 = const()[name = string("w_209_transpose_y_1"), val = bool(false)]; + tensor transpose_509_cast_fp16 = transpose(perm = transpose_509_perm_0, x = reshape_193_cast_fp16)[name = string("transpose_187")]; + tensor w_209_cast_fp16 = matmul(transpose_x = w_209_transpose_x_1, transpose_y = w_209_transpose_y_1, x = var_12129_cast_fp16, y = transpose_509_cast_fp16)[name = string("w_209_cast_fp16")]; + tensor pad_97_to_fp16 = const()[name = string("pad_97_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641472)))]; + tensor var_12132_cast_fp16 = add(x = w_209_cast_fp16, y = pad_97_to_fp16)[name = string("op_12132_cast_fp16")]; + tensor w_211_cast_fp16 = softmax(axis = var_12007, x = var_12132_cast_fp16)[name = string("w_211_cast_fp16")]; + tensor transpose_510_perm_0 = const()[name = string("transpose_510_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_97_transpose_x_1 = const()[name = string("attn_97_transpose_x_1"), val = bool(false)]; + bool attn_97_transpose_y_1 = const()[name = string("attn_97_transpose_y_1"), val = bool(true)]; + tensor transpose_510_cast_fp16 = transpose(perm = transpose_510_perm_0, x = reshape_195_cast_fp16)[name = string("transpose_186")]; + tensor attn_97_cast_fp16 = matmul(transpose_x = attn_97_transpose_x_1, transpose_y = attn_97_transpose_y_1, x = transpose_510_cast_fp16, y = w_211_cast_fp16)[name = string("attn_97_cast_fp16")]; + tensor var_12136 = const()[name = string("op_12136"), val = tensor([1, 2048, 1, 1])]; + tensor input_515_cast_fp16 = reshape(shape = var_12136, x = attn_97_cast_fp16)[name = string("input_515_cast_fp16")]; + string attn_output_97_pad_type_0 = const()[name = string("attn_output_97_pad_type_0"), val = string("valid")]; + tensor attn_output_97_strides_0 = const()[name = string("attn_output_97_strides_0"), val = tensor([1, 1])]; + tensor attn_output_97_pad_0 = const()[name = string("attn_output_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_97_dilations_0 = const()[name = string("attn_output_97_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_97_groups_0 = const()[name = string("attn_output_97_groups_0"), val = int32(1)]; + tensor attn_output_97_cast_fp16 = conv(dilations = attn_output_97_dilations_0, groups = attn_output_97_groups_0, pad = attn_output_97_pad_0, pad_type = attn_output_97_pad_type_0, strides = attn_output_97_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_515_cast_fp16)[name = string("attn_output_97_cast_fp16")]; + tensor x_371_cast_fp16 = add(x = x_365_cast_fp16, y = attn_output_97_cast_fp16)[name = string("x_371_cast_fp16")]; + tensor var_12150_cast_fp16 = mul(x = x_371_cast_fp16, y = x_371_cast_fp16)[name = string("op_12150_cast_fp16")]; + tensor variance_407_axes_0 = const()[name = string("variance_407_axes_0"), val = tensor([1])]; + bool variance_407_keep_dims_0 = const()[name = string("variance_407_keep_dims_0"), val = bool(true)]; + tensor variance_407_cast_fp16 = reduce_mean(axes = variance_407_axes_0, keep_dims = variance_407_keep_dims_0, x = var_12150_cast_fp16)[name = string("variance_407_cast_fp16")]; + fp16 var_12153_to_fp16 = const()[name = string("op_12153_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12154_cast_fp16 = add(x = variance_407_cast_fp16, y = var_12153_to_fp16)[name = string("op_12154_cast_fp16")]; + fp32 var_12155_epsilon_0 = const()[name = string("op_12155_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12155_cast_fp16 = rsqrt(epsilon = var_12155_epsilon_0, x = var_12154_cast_fp16)[name = string("op_12155_cast_fp16")]; + tensor var_12156_cast_fp16 = mul(x = x_371_cast_fp16, y = var_12155_cast_fp16)[name = string("op_12156_cast_fp16")]; + tensor input_517_cast_fp16 = mul(x = var_12156_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_517_cast_fp16")]; + string input_519_pad_type_0 = const()[name = string("input_519_pad_type_0"), val = string("valid")]; + tensor input_519_strides_0 = const()[name = string("input_519_strides_0"), val = tensor([1, 1])]; + tensor input_519_pad_0 = const()[name = string("input_519_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_519_dilations_0 = const()[name = string("input_519_dilations_0"), val = tensor([1, 1])]; + int32 input_519_groups_0 = const()[name = string("input_519_groups_0"), val = int32(1)]; + tensor input_519_cast_fp16 = conv(dilations = input_519_dilations_0, groups = input_519_groups_0, pad = input_519_pad_0, pad_type = input_519_pad_type_0, strides = input_519_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_517_cast_fp16)[name = string("input_519_cast_fp16")]; + tensor var_12164_cast_fp16 = silu(x = input_519_cast_fp16)[name = string("op_12164_cast_fp16")]; + string var_12170_pad_type_0 = const()[name = string("op_12170_pad_type_0"), val = string("valid")]; + tensor var_12170_strides_0 = const()[name = string("op_12170_strides_0"), val = tensor([1, 1])]; + tensor var_12170_pad_0 = const()[name = string("op_12170_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12170_dilations_0 = const()[name = string("op_12170_dilations_0"), val = tensor([1, 1])]; + int32 var_12170_groups_0 = const()[name = string("op_12170_groups_0"), val = int32(1)]; + tensor var_12170_cast_fp16 = conv(dilations = var_12170_dilations_0, groups = var_12170_groups_0, pad = var_12170_pad_0, pad_type = var_12170_pad_type_0, strides = var_12170_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_517_cast_fp16)[name = string("op_12170_cast_fp16")]; + tensor input_521_cast_fp16 = mul(x = var_12164_cast_fp16, y = var_12170_cast_fp16)[name = string("input_521_cast_fp16")]; + string h_97_pad_type_0 = const()[name = string("h_97_pad_type_0"), val = string("valid")]; + tensor h_97_strides_0 = const()[name = string("h_97_strides_0"), val = tensor([1, 1])]; + tensor h_97_pad_0 = const()[name = string("h_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_97_dilations_0 = const()[name = string("h_97_dilations_0"), val = tensor([1, 1])]; + int32 h_97_groups_0 = const()[name = string("h_97_groups_0"), val = int32(1)]; + tensor h_97_cast_fp16 = conv(dilations = h_97_dilations_0, groups = h_97_groups_0, pad = h_97_pad_0, pad_type = h_97_pad_type_0, strides = h_97_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_521_cast_fp16)[name = string("h_97_cast_fp16")]; + tensor x_373_cast_fp16 = add(x = x_371_cast_fp16, y = h_97_cast_fp16)[name = string("x_373_cast_fp16")]; + tensor key_cache_99_begin_0 = const()[name = string("key_cache_99_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_99_end_0 = const()[name = string("key_cache_99_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_99_end_mask_0 = const()[name = string("key_cache_99_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_99_cast_fp16 = slice_by_index(begin = key_cache_99_begin_0, end = key_cache_99_end_0, end_mask = key_cache_99_end_mask_0, x = layer_key_caches_19_cast_fp16)[name = string("key_cache_99_cast_fp16")]; + tensor value_cache_99_begin_0 = const()[name = string("value_cache_99_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_99_end_0 = const()[name = string("value_cache_99_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_99_end_mask_0 = const()[name = string("value_cache_99_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_99_cast_fp16 = slice_by_index(begin = value_cache_99_begin_0, end = value_cache_99_end_0, end_mask = value_cache_99_end_mask_0, x = layer_value_caches_19_cast_fp16)[name = string("value_cache_99_cast_fp16")]; + int32 var_12223 = const()[name = string("op_12223"), val = int32(2)]; + int32 var_12227 = const()[name = string("op_12227"), val = int32(3)]; + tensor var_12242_cast_fp16 = mul(x = x_373_cast_fp16, y = x_373_cast_fp16)[name = string("op_12242_cast_fp16")]; + tensor variance_409_axes_0 = const()[name = string("variance_409_axes_0"), val = tensor([1])]; + bool variance_409_keep_dims_0 = const()[name = string("variance_409_keep_dims_0"), val = bool(true)]; + tensor variance_409_cast_fp16 = reduce_mean(axes = variance_409_axes_0, keep_dims = variance_409_keep_dims_0, x = var_12242_cast_fp16)[name = string("variance_409_cast_fp16")]; + fp16 var_12245_to_fp16 = const()[name = string("op_12245_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12246_cast_fp16 = add(x = variance_409_cast_fp16, y = var_12245_to_fp16)[name = string("op_12246_cast_fp16")]; + fp32 var_12247_epsilon_0 = const()[name = string("op_12247_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12247_cast_fp16 = rsqrt(epsilon = var_12247_epsilon_0, x = var_12246_cast_fp16)[name = string("op_12247_cast_fp16")]; + tensor var_12248_cast_fp16 = mul(x = x_373_cast_fp16, y = var_12247_cast_fp16)[name = string("op_12248_cast_fp16")]; + tensor input_523_cast_fp16 = mul(x = var_12248_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_523_cast_fp16")]; + string q_295_pad_type_0 = const()[name = string("q_295_pad_type_0"), val = string("valid")]; + tensor q_295_strides_0 = const()[name = string("q_295_strides_0"), val = tensor([1, 1])]; + tensor q_295_pad_0 = const()[name = string("q_295_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_295_dilations_0 = const()[name = string("q_295_dilations_0"), val = tensor([1, 1])]; + int32 q_295_groups_0 = const()[name = string("q_295_groups_0"), val = int32(1)]; + tensor q_295_cast_fp16 = conv(dilations = q_295_dilations_0, groups = q_295_groups_0, pad = q_295_pad_0, pad_type = q_295_pad_type_0, strides = q_295_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_523_cast_fp16)[name = string("q_295_cast_fp16")]; + string k_295_pad_type_0 = const()[name = string("k_295_pad_type_0"), val = string("valid")]; + tensor k_295_strides_0 = const()[name = string("k_295_strides_0"), val = tensor([1, 1])]; + tensor k_295_pad_0 = const()[name = string("k_295_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_295_dilations_0 = const()[name = string("k_295_dilations_0"), val = tensor([1, 1])]; + int32 k_295_groups_0 = const()[name = string("k_295_groups_0"), val = int32(1)]; + tensor k_295_cast_fp16 = conv(dilations = k_295_dilations_0, groups = k_295_groups_0, pad = k_295_pad_0, pad_type = k_295_pad_type_0, strides = k_295_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_523_cast_fp16)[name = string("k_295_cast_fp16")]; + string v_99_pad_type_0 = const()[name = string("v_99_pad_type_0"), val = string("valid")]; + tensor v_99_strides_0 = const()[name = string("v_99_strides_0"), val = tensor([1, 1])]; + tensor v_99_pad_0 = const()[name = string("v_99_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_99_dilations_0 = const()[name = string("v_99_dilations_0"), val = tensor([1, 1])]; + int32 v_99_groups_0 = const()[name = string("v_99_groups_0"), val = int32(1)]; + tensor v_99_cast_fp16 = conv(dilations = v_99_dilations_0, groups = v_99_groups_0, pad = v_99_pad_0, pad_type = v_99_pad_type_0, strides = v_99_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_523_cast_fp16)[name = string("v_99_cast_fp16")]; + tensor var_12282 = const()[name = string("op_12282"), val = tensor([16, 128, 1, 1])]; + tensor x_375_cast_fp16 = reshape(shape = var_12282, x = q_295_cast_fp16)[name = string("x_375_cast_fp16")]; + tensor var_12285_cast_fp16 = mul(x = x_375_cast_fp16, y = x_375_cast_fp16)[name = string("op_12285_cast_fp16")]; + tensor variance_411_axes_0 = const()[name = string("variance_411_axes_0"), val = tensor([1])]; + bool variance_411_keep_dims_0 = const()[name = string("variance_411_keep_dims_0"), val = bool(true)]; + tensor variance_411_cast_fp16 = reduce_mean(axes = variance_411_axes_0, keep_dims = variance_411_keep_dims_0, x = var_12285_cast_fp16)[name = string("variance_411_cast_fp16")]; + fp16 var_12288_to_fp16 = const()[name = string("op_12288_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12289_cast_fp16 = add(x = variance_411_cast_fp16, y = var_12288_to_fp16)[name = string("op_12289_cast_fp16")]; + fp32 var_12290_epsilon_0 = const()[name = string("op_12290_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12290_cast_fp16 = rsqrt(epsilon = var_12290_epsilon_0, x = var_12289_cast_fp16)[name = string("op_12290_cast_fp16")]; + tensor var_12291_cast_fp16 = mul(x = x_375_cast_fp16, y = var_12290_cast_fp16)[name = string("op_12291_cast_fp16")]; + tensor q_297_cast_fp16 = mul(x = var_12291_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_297_cast_fp16")]; + tensor var_12293 = const()[name = string("op_12293"), val = tensor([8, 128, 1, 1])]; + tensor x_377_cast_fp16 = reshape(shape = var_12293, x = k_295_cast_fp16)[name = string("x_377_cast_fp16")]; + tensor var_12296_cast_fp16 = mul(x = x_377_cast_fp16, y = x_377_cast_fp16)[name = string("op_12296_cast_fp16")]; + tensor variance_413_axes_0 = const()[name = string("variance_413_axes_0"), val = tensor([1])]; + bool variance_413_keep_dims_0 = const()[name = string("variance_413_keep_dims_0"), val = bool(true)]; + tensor variance_413_cast_fp16 = reduce_mean(axes = variance_413_axes_0, keep_dims = variance_413_keep_dims_0, x = var_12296_cast_fp16)[name = string("variance_413_cast_fp16")]; + fp16 var_12299_to_fp16 = const()[name = string("op_12299_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12300_cast_fp16 = add(x = variance_413_cast_fp16, y = var_12299_to_fp16)[name = string("op_12300_cast_fp16")]; + fp32 var_12301_epsilon_0 = const()[name = string("op_12301_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12301_cast_fp16 = rsqrt(epsilon = var_12301_epsilon_0, x = var_12300_cast_fp16)[name = string("op_12301_cast_fp16")]; + tensor var_12302_cast_fp16 = mul(x = x_377_cast_fp16, y = var_12301_cast_fp16)[name = string("op_12302_cast_fp16")]; + tensor k_297_cast_fp16 = mul(x = var_12302_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_297_cast_fp16")]; + tensor var_12304 = const()[name = string("op_12304"), val = tensor([1, 16, 128, 1])]; + tensor z_197_cast_fp16 = reshape(shape = var_12304, x = q_297_cast_fp16)[name = string("z_197_cast_fp16")]; + tensor var_12306 = const()[name = string("op_12306"), val = tensor([1, 8, 128, 1])]; + tensor z_199_cast_fp16 = reshape(shape = var_12306, x = k_297_cast_fp16)[name = string("z_199_cast_fp16")]; + tensor z1_197_begin_0 = const()[name = string("z1_197_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_197_end_0 = const()[name = string("z1_197_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_197_end_mask_0 = const()[name = string("z1_197_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_197_cast_fp16 = slice_by_index(begin = z1_197_begin_0, end = z1_197_end_0, end_mask = z1_197_end_mask_0, x = z_197_cast_fp16)[name = string("z1_197_cast_fp16")]; + tensor z2_197_begin_0 = const()[name = string("z2_197_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_197_end_0 = const()[name = string("z2_197_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_197_end_mask_0 = const()[name = string("z2_197_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_197_cast_fp16 = slice_by_index(begin = z2_197_begin_0, end = z2_197_end_0, end_mask = z2_197_end_mask_0, x = z_197_cast_fp16)[name = string("z2_197_cast_fp16")]; + tensor var_12314_cast_fp16 = mul(x = z_197_cast_fp16, y = cos_91_to_fp16)[name = string("op_12314_cast_fp16")]; + fp16 const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12315_cast_fp16 = mul(x = z2_197_cast_fp16, y = const_108_promoted_to_fp16)[name = string("op_12315_cast_fp16")]; + bool var_12317_interleave_0 = const()[name = string("op_12317_interleave_0"), val = bool(false)]; + tensor var_12317_cast_fp16 = concat(axis = var_12223, interleave = var_12317_interleave_0, values = (var_12315_cast_fp16, z1_197_cast_fp16))[name = string("op_12317_cast_fp16")]; + tensor var_12318_cast_fp16 = mul(x = var_12317_cast_fp16, y = sin_91_to_fp16)[name = string("op_12318_cast_fp16")]; + tensor q_299_cast_fp16 = add(x = var_12314_cast_fp16, y = var_12318_cast_fp16)[name = string("q_299_cast_fp16")]; + tensor z1_199_begin_0 = const()[name = string("z1_199_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_199_end_0 = const()[name = string("z1_199_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_199_end_mask_0 = const()[name = string("z1_199_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_199_cast_fp16 = slice_by_index(begin = z1_199_begin_0, end = z1_199_end_0, end_mask = z1_199_end_mask_0, x = z_199_cast_fp16)[name = string("z1_199_cast_fp16")]; + tensor z2_199_begin_0 = const()[name = string("z2_199_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_199_end_0 = const()[name = string("z2_199_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_199_end_mask_0 = const()[name = string("z2_199_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_199_cast_fp16 = slice_by_index(begin = z2_199_begin_0, end = z2_199_end_0, end_mask = z2_199_end_mask_0, x = z_199_cast_fp16)[name = string("z2_199_cast_fp16")]; + tensor var_12326_cast_fp16 = mul(x = z_199_cast_fp16, y = cos_91_to_fp16)[name = string("op_12326_cast_fp16")]; + fp16 const_109_promoted_to_fp16 = const()[name = string("const_109_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12327_cast_fp16 = mul(x = z2_199_cast_fp16, y = const_109_promoted_to_fp16)[name = string("op_12327_cast_fp16")]; + bool var_12329_interleave_0 = const()[name = string("op_12329_interleave_0"), val = bool(false)]; + tensor var_12329_cast_fp16 = concat(axis = var_12223, interleave = var_12329_interleave_0, values = (var_12327_cast_fp16, z1_199_cast_fp16))[name = string("op_12329_cast_fp16")]; + tensor var_12330_cast_fp16 = mul(x = var_12329_cast_fp16, y = sin_91_to_fp16)[name = string("op_12330_cast_fp16")]; + tensor k_299_cast_fp16 = add(x = var_12326_cast_fp16, y = var_12330_cast_fp16)[name = string("k_299_cast_fp16")]; + tensor var_12332 = const()[name = string("op_12332"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_99_cast_fp16 = reshape(shape = var_12332, x = k_299_cast_fp16)[name = string("cur_key_99_cast_fp16")]; + tensor var_12334_to_fp16 = const()[name = string("op_12334_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641216)))]; + tensor var_12335_cast_fp16 = mul(x = key_cache_99_cast_fp16, y = var_12334_to_fp16)[name = string("op_12335_cast_fp16")]; + tensor upd_99_to_fp16 = const()[name = string("upd_99_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641344)))]; + tensor var_12336_cast_fp16 = mul(x = cur_key_99_cast_fp16, y = upd_99_to_fp16)[name = string("op_12336_cast_fp16")]; + tensor key_99_cast_fp16 = add(x = var_12335_cast_fp16, y = var_12336_cast_fp16)[name = string("key_99_cast_fp16")]; + tensor var_12338_to_fp16 = const()[name = string("op_12338_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641216)))]; + tensor var_12339_cast_fp16 = mul(x = value_cache_99_cast_fp16, y = var_12338_to_fp16)[name = string("op_12339_cast_fp16")]; + tensor var_12340_cast_fp16 = mul(x = v_99_cast_fp16, y = upd_99_to_fp16)[name = string("op_12340_cast_fp16")]; + tensor value_99_cast_fp16 = add(x = var_12339_cast_fp16, y = var_12340_cast_fp16)[name = string("value_99_cast_fp16")]; + tensor var_12342 = const()[name = string("op_12342"), val = tensor([1, 8, 128, 16])]; + tensor kh_197_cast_fp16 = reshape(shape = var_12342, x = key_99_cast_fp16)[name = string("kh_197_cast_fp16")]; + tensor var_12344 = const()[name = string("op_12344"), val = tensor([1, 8, 128, 16])]; + tensor vh_197_cast_fp16 = reshape(shape = var_12344, x = value_99_cast_fp16)[name = string("vh_197_cast_fp16")]; + tensor transpose_196_perm_0 = const()[name = string("transpose_196_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_98_reps_0 = const()[name = string("tile_98_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_196_cast_fp16 = transpose(perm = transpose_196_perm_0, x = kh_197_cast_fp16)[name = string("transpose_185")]; + tensor tile_98_cast_fp16 = tile(reps = tile_98_reps_0, x = transpose_196_cast_fp16)[name = string("tile_98_cast_fp16")]; + tensor concat_244 = const()[name = string("concat_244"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_196_cast_fp16 = reshape(shape = concat_244, x = tile_98_cast_fp16)[name = string("reshape_196_cast_fp16")]; + tensor transpose_197_perm_0 = const()[name = string("transpose_197_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_245 = const()[name = string("concat_245"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_197_cast_fp16 = transpose(perm = transpose_197_perm_0, x = reshape_196_cast_fp16)[name = string("transpose_184")]; + tensor reshape_197_cast_fp16 = reshape(shape = concat_245, x = transpose_197_cast_fp16)[name = string("reshape_197_cast_fp16")]; + tensor transpose_198_perm_0 = const()[name = string("transpose_198_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_99_reps_0 = const()[name = string("tile_99_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_198_cast_fp16 = transpose(perm = transpose_198_perm_0, x = vh_197_cast_fp16)[name = string("transpose_183")]; + tensor tile_99_cast_fp16 = tile(reps = tile_99_reps_0, x = transpose_198_cast_fp16)[name = string("tile_99_cast_fp16")]; + tensor concat_246 = const()[name = string("concat_246"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_198_cast_fp16 = reshape(shape = concat_246, x = tile_99_cast_fp16)[name = string("reshape_198_cast_fp16")]; + tensor transpose_199_perm_0 = const()[name = string("transpose_199_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_247 = const()[name = string("concat_247"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_199_cast_fp16 = transpose(perm = transpose_199_perm_0, x = reshape_198_cast_fp16)[name = string("transpose_182")]; + tensor reshape_199_cast_fp16 = reshape(shape = concat_247, x = transpose_199_cast_fp16)[name = string("reshape_199_cast_fp16")]; + fp16 var_12348_to_fp16 = const()[name = string("op_12348_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_12349_cast_fp16 = mul(x = q_299_cast_fp16, y = var_12348_to_fp16)[name = string("op_12349_cast_fp16")]; + tensor transpose_513_perm_0 = const()[name = string("transpose_513_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_213_transpose_x_1 = const()[name = string("w_213_transpose_x_1"), val = bool(true)]; + bool w_213_transpose_y_1 = const()[name = string("w_213_transpose_y_1"), val = bool(false)]; + tensor transpose_513_cast_fp16 = transpose(perm = transpose_513_perm_0, x = reshape_197_cast_fp16)[name = string("transpose_181")]; + tensor w_213_cast_fp16 = matmul(transpose_x = w_213_transpose_x_1, transpose_y = w_213_transpose_y_1, x = var_12349_cast_fp16, y = transpose_513_cast_fp16)[name = string("w_213_cast_fp16")]; + tensor pad_99_to_fp16 = const()[name = string("pad_99_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641472)))]; + tensor var_12352_cast_fp16 = add(x = w_213_cast_fp16, y = pad_99_to_fp16)[name = string("op_12352_cast_fp16")]; + tensor w_215_cast_fp16 = softmax(axis = var_12227, x = var_12352_cast_fp16)[name = string("w_215_cast_fp16")]; + tensor transpose_514_perm_0 = const()[name = string("transpose_514_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_99_transpose_x_1 = const()[name = string("attn_99_transpose_x_1"), val = bool(false)]; + bool attn_99_transpose_y_1 = const()[name = string("attn_99_transpose_y_1"), val = bool(true)]; + tensor transpose_514_cast_fp16 = transpose(perm = transpose_514_perm_0, x = reshape_199_cast_fp16)[name = string("transpose_180")]; + tensor attn_99_cast_fp16 = matmul(transpose_x = attn_99_transpose_x_1, transpose_y = attn_99_transpose_y_1, x = transpose_514_cast_fp16, y = w_215_cast_fp16)[name = string("attn_99_cast_fp16")]; + tensor var_12356 = const()[name = string("op_12356"), val = tensor([1, 2048, 1, 1])]; + tensor input_525_cast_fp16 = reshape(shape = var_12356, x = attn_99_cast_fp16)[name = string("input_525_cast_fp16")]; + string attn_output_99_pad_type_0 = const()[name = string("attn_output_99_pad_type_0"), val = string("valid")]; + tensor attn_output_99_strides_0 = const()[name = string("attn_output_99_strides_0"), val = tensor([1, 1])]; + tensor attn_output_99_pad_0 = const()[name = string("attn_output_99_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_99_dilations_0 = const()[name = string("attn_output_99_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_99_groups_0 = const()[name = string("attn_output_99_groups_0"), val = int32(1)]; + tensor attn_output_99_cast_fp16 = conv(dilations = attn_output_99_dilations_0, groups = attn_output_99_groups_0, pad = attn_output_99_pad_0, pad_type = attn_output_99_pad_type_0, strides = attn_output_99_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_525_cast_fp16)[name = string("attn_output_99_cast_fp16")]; + tensor x_379_cast_fp16 = add(x = x_373_cast_fp16, y = attn_output_99_cast_fp16)[name = string("x_379_cast_fp16")]; + tensor var_12370_cast_fp16 = mul(x = x_379_cast_fp16, y = x_379_cast_fp16)[name = string("op_12370_cast_fp16")]; + tensor variance_415_axes_0 = const()[name = string("variance_415_axes_0"), val = tensor([1])]; + bool variance_415_keep_dims_0 = const()[name = string("variance_415_keep_dims_0"), val = bool(true)]; + tensor variance_415_cast_fp16 = reduce_mean(axes = variance_415_axes_0, keep_dims = variance_415_keep_dims_0, x = var_12370_cast_fp16)[name = string("variance_415_cast_fp16")]; + fp16 var_12373_to_fp16 = const()[name = string("op_12373_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12374_cast_fp16 = add(x = variance_415_cast_fp16, y = var_12373_to_fp16)[name = string("op_12374_cast_fp16")]; + fp32 var_12375_epsilon_0 = const()[name = string("op_12375_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12375_cast_fp16 = rsqrt(epsilon = var_12375_epsilon_0, x = var_12374_cast_fp16)[name = string("op_12375_cast_fp16")]; + tensor var_12376_cast_fp16 = mul(x = x_379_cast_fp16, y = var_12375_cast_fp16)[name = string("op_12376_cast_fp16")]; + tensor input_527_cast_fp16 = mul(x = var_12376_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_527_cast_fp16")]; + string input_529_pad_type_0 = const()[name = string("input_529_pad_type_0"), val = string("valid")]; + tensor input_529_strides_0 = const()[name = string("input_529_strides_0"), val = tensor([1, 1])]; + tensor input_529_pad_0 = const()[name = string("input_529_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_529_dilations_0 = const()[name = string("input_529_dilations_0"), val = tensor([1, 1])]; + int32 input_529_groups_0 = const()[name = string("input_529_groups_0"), val = int32(1)]; + tensor input_529_cast_fp16 = conv(dilations = input_529_dilations_0, groups = input_529_groups_0, pad = input_529_pad_0, pad_type = input_529_pad_type_0, strides = input_529_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_527_cast_fp16)[name = string("input_529_cast_fp16")]; + tensor var_12384_cast_fp16 = silu(x = input_529_cast_fp16)[name = string("op_12384_cast_fp16")]; + string var_12390_pad_type_0 = const()[name = string("op_12390_pad_type_0"), val = string("valid")]; + tensor var_12390_strides_0 = const()[name = string("op_12390_strides_0"), val = tensor([1, 1])]; + tensor var_12390_pad_0 = const()[name = string("op_12390_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12390_dilations_0 = const()[name = string("op_12390_dilations_0"), val = tensor([1, 1])]; + int32 var_12390_groups_0 = const()[name = string("op_12390_groups_0"), val = int32(1)]; + tensor var_12390_cast_fp16 = conv(dilations = var_12390_dilations_0, groups = var_12390_groups_0, pad = var_12390_pad_0, pad_type = var_12390_pad_type_0, strides = var_12390_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_527_cast_fp16)[name = string("op_12390_cast_fp16")]; + tensor input_531_cast_fp16 = mul(x = var_12384_cast_fp16, y = var_12390_cast_fp16)[name = string("input_531_cast_fp16")]; + string h_99_pad_type_0 = const()[name = string("h_99_pad_type_0"), val = string("valid")]; + tensor h_99_strides_0 = const()[name = string("h_99_strides_0"), val = tensor([1, 1])]; + tensor h_99_pad_0 = const()[name = string("h_99_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_99_dilations_0 = const()[name = string("h_99_dilations_0"), val = tensor([1, 1])]; + int32 h_99_groups_0 = const()[name = string("h_99_groups_0"), val = int32(1)]; + tensor h_99_cast_fp16 = conv(dilations = h_99_dilations_0, groups = h_99_groups_0, pad = h_99_pad_0, pad_type = h_99_pad_type_0, strides = h_99_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_531_cast_fp16)[name = string("h_99_cast_fp16")]; + tensor inputs_17_cast_fp16 = add(x = x_379_cast_fp16, y = h_99_cast_fp16)[name = string("inputs_17_cast_fp16")]; + int32 var_12418 = const()[name = string("op_12418"), val = int32(1)]; + bool layer_key_caches_21_interleave_0 = const()[name = string("layer_key_caches_21_interleave_0"), val = bool(false)]; + tensor layer_key_caches_21_cast_fp16 = concat(axis = var_12418, interleave = layer_key_caches_21_interleave_0, values = (key_91_cast_fp16, key_93_cast_fp16, key_95_cast_fp16, key_97_cast_fp16, key_99_cast_fp16))[name = string("layer_key_caches_21_cast_fp16")]; + int32 var_12421 = const()[name = string("op_12421"), val = int32(1)]; + bool layer_value_caches_21_interleave_0 = const()[name = string("layer_value_caches_21_interleave_0"), val = bool(false)]; + tensor layer_value_caches_21_cast_fp16 = concat(axis = var_12421, interleave = layer_value_caches_21_interleave_0, values = (value_91_cast_fp16, value_93_cast_fp16, value_95_cast_fp16, value_97_cast_fp16, value_99_cast_fp16))[name = string("layer_value_caches_21_cast_fp16")]; + tensor inputs_sq_17_cast_fp16 = mul(x = inputs_17_cast_fp16, y = inputs_17_cast_fp16)[name = string("inputs_sq_17_cast_fp16")]; + tensor variance_417_axes_0 = const()[name = string("variance_417_axes_0"), val = tensor([1])]; + bool variance_417_keep_dims_0 = const()[name = string("variance_417_keep_dims_0"), val = bool(true)]; + tensor variance_417_cast_fp16 = reduce_mean(axes = variance_417_axes_0, keep_dims = variance_417_keep_dims_0, x = inputs_sq_17_cast_fp16)[name = string("variance_417_cast_fp16")]; + fp16 var_12431_to_fp16 = const()[name = string("op_12431_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12432_cast_fp16 = add(x = variance_417_cast_fp16, y = var_12431_to_fp16)[name = string("op_12432_cast_fp16")]; + fp32 var_12433_epsilon_0 = const()[name = string("op_12433_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12433_cast_fp16 = rsqrt(epsilon = var_12433_epsilon_0, x = var_12432_cast_fp16)[name = string("op_12433_cast_fp16")]; + tensor hidden_states_17_cast_fp16 = mul(x = inputs_17_cast_fp16, y = var_12433_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; + tensor input_533_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_17_cast_fp16)[name = string("input_533_cast_fp16")]; + string logits_33_pad_type_0 = const()[name = string("logits_33_pad_type_0"), val = string("valid")]; + tensor logits_33_strides_0 = const()[name = string("logits_33_strides_0"), val = tensor([1, 1])]; + tensor logits_33_pad_0 = const()[name = string("logits_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_33_dilations_0 = const()[name = string("logits_33_dilations_0"), val = tensor([1, 1])]; + int32 logits_33_groups_0 = const()[name = string("logits_33_groups_0"), val = int32(1)]; + tensor lm_heads_8_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95489024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(97586240))))[name = string("lm_heads_8_weight_to_fp16_palettized")]; + tensor logits_33_cast_fp16 = conv(dilations = logits_33_dilations_0, groups = logits_33_groups_0, pad = logits_33_pad_0, pad_type = logits_33_pad_type_0, strides = logits_33_strides_0, weight = lm_heads_8_weight_to_fp16_palettized, x = input_533_cast_fp16)[name = string("logits_33_cast_fp16")]; + tensor var_12451 = const()[name = string("op_12451"), val = tensor([1, 2048])]; + tensor logits_35_cast_fp16 = reshape(shape = var_12451, x = logits_33_cast_fp16)[name = string("logits_35_cast_fp16")]; + tensor scaled_logits_17_cast_fp16 = real_div(x = logits_35_cast_fp16, y = temperature)[name = string("scaled_logits_17_cast_fp16")]; + int32 var_12461 = const()[name = string("op_12461"), val = int32(100)]; + int32 top_values_17_axis_0 = const()[name = string("top_values_17_axis_0"), val = int32(1)]; + bool top_values_17_ascending_0 = const()[name = string("top_values_17_ascending_0"), val = bool(false)]; + bool top_values_17_sort_0 = const()[name = string("top_values_17_sort_0"), val = bool(true)]; + bool top_values_17_return_indices_0 = const()[name = string("top_values_17_return_indices_0"), val = bool(true)]; + string top_values_17_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_17_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_17_cast_fp16_cast_uint16_0, tensor top_values_17_cast_fp16_cast_uint16_1 = topk(ascending = top_values_17_ascending_0, axis = top_values_17_axis_0, k = var_12461, output_indices_dtype = top_values_17_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_17_return_indices_0, sort = top_values_17_sort_0, x = scaled_logits_17_cast_fp16)[name = string("top_values_17_cast_fp16_cast_uint16")]; + tensor var_12467_cast_fp16 = mul(x = top_values_17_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_12467_cast_fp16")]; + tensor var_12471_cast_fp16 = add(x = var_12467_cast_fp16, y = var_2438_cast_fp16)[name = string("op_12471_cast_fp16")]; + tensor reduce_min_8_axes_0 = const()[name = string("reduce_min_8_axes_0"), val = tensor([1])]; + bool reduce_min_8_keep_dims_0 = const()[name = string("reduce_min_8_keep_dims_0"), val = bool(true)]; + tensor reduce_min_8_cast_fp16 = reduce_min(axes = reduce_min_8_axes_0, keep_dims = reduce_min_8_keep_dims_0, x = var_12471_cast_fp16)[name = string("reduce_min_8_cast_fp16")]; + tensor var_12474_cast_fp16 = greater_equal(x = scaled_logits_17_cast_fp16, y = reduce_min_8_cast_fp16)[name = string("op_12474_cast_fp16")]; + fp16 var_12475_value_0_to_fp16 = const()[name = string("op_12475_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_12475_cast_fp16 = fill_like(ref_tensor = scaled_logits_17_cast_fp16, value = var_12475_value_0_to_fp16)[name = string("op_12475_cast_fp16")]; + tensor masked_logits_17_cast_fp16 = select(a = scaled_logits_17_cast_fp16, b = var_12475_cast_fp16, cond = var_12474_cast_fp16)[name = string("masked_logits_17_cast_fp16")]; + tensor var_12479_begin_0 = const()[name = string("op_12479_begin_0"), val = tensor([8, 0])]; + tensor var_12479_end_0 = const()[name = string("op_12479_end_0"), val = tensor([9, 2048])]; + tensor var_12479_end_mask_0 = const()[name = string("op_12479_end_mask_0"), val = tensor([false, true])]; + tensor var_12479_squeeze_mask_0 = const()[name = string("op_12479_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_12479_cast_fp16 = slice_by_index(begin = var_12479_begin_0, end = var_12479_end_0, end_mask = var_12479_end_mask_0, squeeze_mask = var_12479_squeeze_mask_0, x = gumbel)[name = string("op_12479_cast_fp16")]; + tensor var_12482 = const()[name = string("op_12482"), val = tensor([1, 2048])]; + tensor var_12483_cast_fp16 = reshape(shape = var_12482, x = var_12479_cast_fp16)[name = string("op_12483_cast_fp16")]; + tensor noisy_logits_17_cast_fp16 = add(x = masked_logits_17_cast_fp16, y = var_12483_cast_fp16)[name = string("noisy_logits_17_cast_fp16")]; + int32 code_17_axis_0 = const()[name = string("code_17_axis_0"), val = int32(1)]; + bool code_17_keep_dims_0 = const()[name = string("code_17_keep_dims_0"), val = bool(false)]; + string code_17_output_dtype_0 = const()[name = string("code_17_output_dtype_0"), val = string("int32")]; + tensor code_17_cast_fp16 = reduce_argmax(axis = code_17_axis_0, keep_dims = code_17_keep_dims_0, output_dtype = code_17_output_dtype_0, x = noisy_logits_17_cast_fp16)[name = string("code_17_cast_fp16")]; + int32 var_12494 = const()[name = string("op_12494"), val = int32(16384)]; + tensor input_535 = add(x = code_17_cast_fp16, y = var_12494)[name = string("input_535")]; + int32 code_embed_33_axis_0 = const()[name = string("code_embed_33_axis_0"), val = int32(0)]; + int32 code_embed_33_batch_dims_0 = const()[name = string("code_embed_33_batch_dims_0"), val = int32(0)]; + bool code_embed_33_validate_indices_0 = const()[name = string("code_embed_33_validate_indices_0"), val = bool(false)]; + string input_535_to_uint16_dtype_0 = const()[name = string("input_535_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_535_to_uint16 = cast(dtype = input_535_to_uint16_dtype_0, x = input_535)[name = string("cast_6")]; + tensor code_embed_33_cast_fp16_cast_uint16 = gather(axis = code_embed_33_axis_0, batch_dims = code_embed_33_batch_dims_0, indices = input_535_to_uint16, validate_indices = code_embed_33_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_33_cast_fp16_cast_uint16")]; + tensor var_12498 = const()[name = string("op_12498"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_35_cast_fp16 = reshape(shape = var_12498, x = code_embed_33_cast_fp16_cast_uint16)[name = string("code_embed_35_cast_fp16")]; + tensor embed_sum_19_cast_fp16 = add(x = embed_sum_17_cast_fp16, y = code_embed_35_cast_fp16)[name = string("embed_sum_19_cast_fp16")]; + tensor key_cache_101_begin_0 = const()[name = string("key_cache_101_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_101_end_0 = const()[name = string("key_cache_101_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_101_end_mask_0 = const()[name = string("key_cache_101_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_101_cast_fp16 = slice_by_index(begin = key_cache_101_begin_0, end = key_cache_101_end_0, end_mask = key_cache_101_end_mask_0, x = layer_key_caches_21_cast_fp16)[name = string("key_cache_101_cast_fp16")]; + tensor value_cache_101_begin_0 = const()[name = string("value_cache_101_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_101_end_0 = const()[name = string("value_cache_101_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_101_end_mask_0 = const()[name = string("value_cache_101_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_101_cast_fp16 = slice_by_index(begin = value_cache_101_begin_0, end = value_cache_101_end_0, end_mask = value_cache_101_end_mask_0, x = layer_value_caches_21_cast_fp16)[name = string("value_cache_101_cast_fp16")]; + int32 var_12597 = const()[name = string("op_12597"), val = int32(2)]; + int32 var_12601 = const()[name = string("op_12601"), val = int32(3)]; + tensor var_12616_cast_fp16 = mul(x = code_embed_35_cast_fp16, y = code_embed_35_cast_fp16)[name = string("op_12616_cast_fp16")]; + tensor variance_419_axes_0 = const()[name = string("variance_419_axes_0"), val = tensor([1])]; + bool variance_419_keep_dims_0 = const()[name = string("variance_419_keep_dims_0"), val = bool(true)]; + tensor variance_419_cast_fp16 = reduce_mean(axes = variance_419_axes_0, keep_dims = variance_419_keep_dims_0, x = var_12616_cast_fp16)[name = string("variance_419_cast_fp16")]; + fp16 var_12619_to_fp16 = const()[name = string("op_12619_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12620_cast_fp16 = add(x = variance_419_cast_fp16, y = var_12619_to_fp16)[name = string("op_12620_cast_fp16")]; + fp32 var_12621_epsilon_0 = const()[name = string("op_12621_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12621_cast_fp16 = rsqrt(epsilon = var_12621_epsilon_0, x = var_12620_cast_fp16)[name = string("op_12621_cast_fp16")]; + tensor var_12622_cast_fp16 = mul(x = code_embed_35_cast_fp16, y = var_12621_cast_fp16)[name = string("op_12622_cast_fp16")]; + tensor input_537_cast_fp16 = mul(x = var_12622_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_537_cast_fp16")]; + string q_301_pad_type_0 = const()[name = string("q_301_pad_type_0"), val = string("valid")]; + tensor q_301_strides_0 = const()[name = string("q_301_strides_0"), val = tensor([1, 1])]; + tensor q_301_pad_0 = const()[name = string("q_301_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_301_dilations_0 = const()[name = string("q_301_dilations_0"), val = tensor([1, 1])]; + int32 q_301_groups_0 = const()[name = string("q_301_groups_0"), val = int32(1)]; + tensor q_301_cast_fp16 = conv(dilations = q_301_dilations_0, groups = q_301_groups_0, pad = q_301_pad_0, pad_type = q_301_pad_type_0, strides = q_301_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_537_cast_fp16)[name = string("q_301_cast_fp16")]; + string k_301_pad_type_0 = const()[name = string("k_301_pad_type_0"), val = string("valid")]; + tensor k_301_strides_0 = const()[name = string("k_301_strides_0"), val = tensor([1, 1])]; + tensor k_301_pad_0 = const()[name = string("k_301_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_301_dilations_0 = const()[name = string("k_301_dilations_0"), val = tensor([1, 1])]; + int32 k_301_groups_0 = const()[name = string("k_301_groups_0"), val = int32(1)]; + tensor k_301_cast_fp16 = conv(dilations = k_301_dilations_0, groups = k_301_groups_0, pad = k_301_pad_0, pad_type = k_301_pad_type_0, strides = k_301_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_537_cast_fp16)[name = string("k_301_cast_fp16")]; + string v_101_pad_type_0 = const()[name = string("v_101_pad_type_0"), val = string("valid")]; + tensor v_101_strides_0 = const()[name = string("v_101_strides_0"), val = tensor([1, 1])]; + tensor v_101_pad_0 = const()[name = string("v_101_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_101_dilations_0 = const()[name = string("v_101_dilations_0"), val = tensor([1, 1])]; + int32 v_101_groups_0 = const()[name = string("v_101_groups_0"), val = int32(1)]; + tensor v_101_cast_fp16 = conv(dilations = v_101_dilations_0, groups = v_101_groups_0, pad = v_101_pad_0, pad_type = v_101_pad_type_0, strides = v_101_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_537_cast_fp16)[name = string("v_101_cast_fp16")]; + tensor var_12656 = const()[name = string("op_12656"), val = tensor([16, 128, 1, 1])]; + tensor x_381_cast_fp16 = reshape(shape = var_12656, x = q_301_cast_fp16)[name = string("x_381_cast_fp16")]; + tensor var_12659_cast_fp16 = mul(x = x_381_cast_fp16, y = x_381_cast_fp16)[name = string("op_12659_cast_fp16")]; + tensor variance_421_axes_0 = const()[name = string("variance_421_axes_0"), val = tensor([1])]; + bool variance_421_keep_dims_0 = const()[name = string("variance_421_keep_dims_0"), val = bool(true)]; + tensor variance_421_cast_fp16 = reduce_mean(axes = variance_421_axes_0, keep_dims = variance_421_keep_dims_0, x = var_12659_cast_fp16)[name = string("variance_421_cast_fp16")]; + fp16 var_12662_to_fp16 = const()[name = string("op_12662_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12663_cast_fp16 = add(x = variance_421_cast_fp16, y = var_12662_to_fp16)[name = string("op_12663_cast_fp16")]; + fp32 var_12664_epsilon_0 = const()[name = string("op_12664_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12664_cast_fp16 = rsqrt(epsilon = var_12664_epsilon_0, x = var_12663_cast_fp16)[name = string("op_12664_cast_fp16")]; + tensor var_12665_cast_fp16 = mul(x = x_381_cast_fp16, y = var_12664_cast_fp16)[name = string("op_12665_cast_fp16")]; + tensor q_303_cast_fp16 = mul(x = var_12665_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_303_cast_fp16")]; + tensor var_12667 = const()[name = string("op_12667"), val = tensor([8, 128, 1, 1])]; + tensor x_383_cast_fp16 = reshape(shape = var_12667, x = k_301_cast_fp16)[name = string("x_383_cast_fp16")]; + tensor var_12670_cast_fp16 = mul(x = x_383_cast_fp16, y = x_383_cast_fp16)[name = string("op_12670_cast_fp16")]; + tensor variance_423_axes_0 = const()[name = string("variance_423_axes_0"), val = tensor([1])]; + bool variance_423_keep_dims_0 = const()[name = string("variance_423_keep_dims_0"), val = bool(true)]; + tensor variance_423_cast_fp16 = reduce_mean(axes = variance_423_axes_0, keep_dims = variance_423_keep_dims_0, x = var_12670_cast_fp16)[name = string("variance_423_cast_fp16")]; + fp16 var_12673_to_fp16 = const()[name = string("op_12673_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12674_cast_fp16 = add(x = variance_423_cast_fp16, y = var_12673_to_fp16)[name = string("op_12674_cast_fp16")]; + fp32 var_12675_epsilon_0 = const()[name = string("op_12675_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12675_cast_fp16 = rsqrt(epsilon = var_12675_epsilon_0, x = var_12674_cast_fp16)[name = string("op_12675_cast_fp16")]; + tensor var_12676_cast_fp16 = mul(x = x_383_cast_fp16, y = var_12675_cast_fp16)[name = string("op_12676_cast_fp16")]; + tensor k_303_cast_fp16 = mul(x = var_12676_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_303_cast_fp16")]; + tensor var_12678 = const()[name = string("op_12678"), val = tensor([1, 16, 128, 1])]; + tensor z_201_cast_fp16 = reshape(shape = var_12678, x = q_303_cast_fp16)[name = string("z_201_cast_fp16")]; + tensor var_12680 = const()[name = string("op_12680"), val = tensor([1, 8, 128, 1])]; + tensor z_203_cast_fp16 = reshape(shape = var_12680, x = k_303_cast_fp16)[name = string("z_203_cast_fp16")]; + tensor z1_201_begin_0 = const()[name = string("z1_201_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_201_end_0 = const()[name = string("z1_201_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_201_end_mask_0 = const()[name = string("z1_201_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_201_cast_fp16 = slice_by_index(begin = z1_201_begin_0, end = z1_201_end_0, end_mask = z1_201_end_mask_0, x = z_201_cast_fp16)[name = string("z1_201_cast_fp16")]; + tensor z2_201_begin_0 = const()[name = string("z2_201_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_201_end_0 = const()[name = string("z2_201_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_201_end_mask_0 = const()[name = string("z2_201_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_201_cast_fp16 = slice_by_index(begin = z2_201_begin_0, end = z2_201_end_0, end_mask = z2_201_end_mask_0, x = z_201_cast_fp16)[name = string("z2_201_cast_fp16")]; + tensor cos_101_to_fp16 = const()[name = string("cos_101_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641600)))]; + tensor var_12688_cast_fp16 = mul(x = z_201_cast_fp16, y = cos_101_to_fp16)[name = string("op_12688_cast_fp16")]; + fp16 const_111_promoted_to_fp16 = const()[name = string("const_111_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12689_cast_fp16 = mul(x = z2_201_cast_fp16, y = const_111_promoted_to_fp16)[name = string("op_12689_cast_fp16")]; + bool var_12691_interleave_0 = const()[name = string("op_12691_interleave_0"), val = bool(false)]; + tensor var_12691_cast_fp16 = concat(axis = var_12597, interleave = var_12691_interleave_0, values = (var_12689_cast_fp16, z1_201_cast_fp16))[name = string("op_12691_cast_fp16")]; + tensor sin_101_to_fp16 = const()[name = string("sin_101_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141641920)))]; + tensor var_12692_cast_fp16 = mul(x = var_12691_cast_fp16, y = sin_101_to_fp16)[name = string("op_12692_cast_fp16")]; + tensor q_305_cast_fp16 = add(x = var_12688_cast_fp16, y = var_12692_cast_fp16)[name = string("q_305_cast_fp16")]; + tensor z1_203_begin_0 = const()[name = string("z1_203_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_203_end_0 = const()[name = string("z1_203_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_203_end_mask_0 = const()[name = string("z1_203_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_203_cast_fp16 = slice_by_index(begin = z1_203_begin_0, end = z1_203_end_0, end_mask = z1_203_end_mask_0, x = z_203_cast_fp16)[name = string("z1_203_cast_fp16")]; + tensor z2_203_begin_0 = const()[name = string("z2_203_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_203_end_0 = const()[name = string("z2_203_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_203_end_mask_0 = const()[name = string("z2_203_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_203_cast_fp16 = slice_by_index(begin = z2_203_begin_0, end = z2_203_end_0, end_mask = z2_203_end_mask_0, x = z_203_cast_fp16)[name = string("z2_203_cast_fp16")]; + tensor var_12700_cast_fp16 = mul(x = z_203_cast_fp16, y = cos_101_to_fp16)[name = string("op_12700_cast_fp16")]; + fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12701_cast_fp16 = mul(x = z2_203_cast_fp16, y = const_112_promoted_to_fp16)[name = string("op_12701_cast_fp16")]; + bool var_12703_interleave_0 = const()[name = string("op_12703_interleave_0"), val = bool(false)]; + tensor var_12703_cast_fp16 = concat(axis = var_12597, interleave = var_12703_interleave_0, values = (var_12701_cast_fp16, z1_203_cast_fp16))[name = string("op_12703_cast_fp16")]; + tensor var_12704_cast_fp16 = mul(x = var_12703_cast_fp16, y = sin_101_to_fp16)[name = string("op_12704_cast_fp16")]; + tensor k_305_cast_fp16 = add(x = var_12700_cast_fp16, y = var_12704_cast_fp16)[name = string("k_305_cast_fp16")]; + tensor var_12706 = const()[name = string("op_12706"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_101_cast_fp16 = reshape(shape = var_12706, x = k_305_cast_fp16)[name = string("cur_key_101_cast_fp16")]; + tensor var_12708_to_fp16 = const()[name = string("op_12708_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642240)))]; + tensor var_12709_cast_fp16 = mul(x = key_cache_101_cast_fp16, y = var_12708_to_fp16)[name = string("op_12709_cast_fp16")]; + tensor upd_101_to_fp16 = const()[name = string("upd_101_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642368)))]; + tensor var_12710_cast_fp16 = mul(x = cur_key_101_cast_fp16, y = upd_101_to_fp16)[name = string("op_12710_cast_fp16")]; + tensor key_101_cast_fp16 = add(x = var_12709_cast_fp16, y = var_12710_cast_fp16)[name = string("key_101_cast_fp16")]; + tensor var_12712_to_fp16 = const()[name = string("op_12712_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642240)))]; + tensor var_12713_cast_fp16 = mul(x = value_cache_101_cast_fp16, y = var_12712_to_fp16)[name = string("op_12713_cast_fp16")]; + tensor var_12714_cast_fp16 = mul(x = v_101_cast_fp16, y = upd_101_to_fp16)[name = string("op_12714_cast_fp16")]; + tensor value_101_cast_fp16 = add(x = var_12713_cast_fp16, y = var_12714_cast_fp16)[name = string("value_101_cast_fp16")]; + tensor var_12716 = const()[name = string("op_12716"), val = tensor([1, 8, 128, 16])]; + tensor kh_201_cast_fp16 = reshape(shape = var_12716, x = key_101_cast_fp16)[name = string("kh_201_cast_fp16")]; + tensor var_12718 = const()[name = string("op_12718"), val = tensor([1, 8, 128, 16])]; + tensor vh_201_cast_fp16 = reshape(shape = var_12718, x = value_101_cast_fp16)[name = string("vh_201_cast_fp16")]; + tensor transpose_200_perm_0 = const()[name = string("transpose_200_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_100_reps_0 = const()[name = string("tile_100_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_200_cast_fp16 = transpose(perm = transpose_200_perm_0, x = kh_201_cast_fp16)[name = string("transpose_179")]; + tensor tile_100_cast_fp16 = tile(reps = tile_100_reps_0, x = transpose_200_cast_fp16)[name = string("tile_100_cast_fp16")]; + tensor concat_253 = const()[name = string("concat_253"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_200_cast_fp16 = reshape(shape = concat_253, x = tile_100_cast_fp16)[name = string("reshape_200_cast_fp16")]; + tensor transpose_201_perm_0 = const()[name = string("transpose_201_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_254 = const()[name = string("concat_254"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_201_cast_fp16 = transpose(perm = transpose_201_perm_0, x = reshape_200_cast_fp16)[name = string("transpose_178")]; + tensor reshape_201_cast_fp16 = reshape(shape = concat_254, x = transpose_201_cast_fp16)[name = string("reshape_201_cast_fp16")]; + tensor transpose_202_perm_0 = const()[name = string("transpose_202_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_101_reps_0 = const()[name = string("tile_101_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_202_cast_fp16 = transpose(perm = transpose_202_perm_0, x = vh_201_cast_fp16)[name = string("transpose_177")]; + tensor tile_101_cast_fp16 = tile(reps = tile_101_reps_0, x = transpose_202_cast_fp16)[name = string("tile_101_cast_fp16")]; + tensor concat_255 = const()[name = string("concat_255"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_202_cast_fp16 = reshape(shape = concat_255, x = tile_101_cast_fp16)[name = string("reshape_202_cast_fp16")]; + tensor transpose_203_perm_0 = const()[name = string("transpose_203_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_256 = const()[name = string("concat_256"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_203_cast_fp16 = transpose(perm = transpose_203_perm_0, x = reshape_202_cast_fp16)[name = string("transpose_176")]; + tensor reshape_203_cast_fp16 = reshape(shape = concat_256, x = transpose_203_cast_fp16)[name = string("reshape_203_cast_fp16")]; + fp16 var_12722_to_fp16 = const()[name = string("op_12722_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_12723_cast_fp16 = mul(x = q_305_cast_fp16, y = var_12722_to_fp16)[name = string("op_12723_cast_fp16")]; + tensor transpose_517_perm_0 = const()[name = string("transpose_517_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_219_transpose_x_1 = const()[name = string("w_219_transpose_x_1"), val = bool(true)]; + bool w_219_transpose_y_1 = const()[name = string("w_219_transpose_y_1"), val = bool(false)]; + tensor transpose_517_cast_fp16 = transpose(perm = transpose_517_perm_0, x = reshape_201_cast_fp16)[name = string("transpose_175")]; + tensor w_219_cast_fp16 = matmul(transpose_x = w_219_transpose_x_1, transpose_y = w_219_transpose_y_1, x = var_12723_cast_fp16, y = transpose_517_cast_fp16)[name = string("w_219_cast_fp16")]; + tensor pad_101_to_fp16 = const()[name = string("pad_101_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642496)))]; + tensor var_12726_cast_fp16 = add(x = w_219_cast_fp16, y = pad_101_to_fp16)[name = string("op_12726_cast_fp16")]; + tensor w_221_cast_fp16 = softmax(axis = var_12601, x = var_12726_cast_fp16)[name = string("w_221_cast_fp16")]; + tensor transpose_518_perm_0 = const()[name = string("transpose_518_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_101_transpose_x_1 = const()[name = string("attn_101_transpose_x_1"), val = bool(false)]; + bool attn_101_transpose_y_1 = const()[name = string("attn_101_transpose_y_1"), val = bool(true)]; + tensor transpose_518_cast_fp16 = transpose(perm = transpose_518_perm_0, x = reshape_203_cast_fp16)[name = string("transpose_174")]; + tensor attn_101_cast_fp16 = matmul(transpose_x = attn_101_transpose_x_1, transpose_y = attn_101_transpose_y_1, x = transpose_518_cast_fp16, y = w_221_cast_fp16)[name = string("attn_101_cast_fp16")]; + tensor var_12730 = const()[name = string("op_12730"), val = tensor([1, 2048, 1, 1])]; + tensor input_539_cast_fp16 = reshape(shape = var_12730, x = attn_101_cast_fp16)[name = string("input_539_cast_fp16")]; + string attn_output_101_pad_type_0 = const()[name = string("attn_output_101_pad_type_0"), val = string("valid")]; + tensor attn_output_101_strides_0 = const()[name = string("attn_output_101_strides_0"), val = tensor([1, 1])]; + tensor attn_output_101_pad_0 = const()[name = string("attn_output_101_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_101_dilations_0 = const()[name = string("attn_output_101_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_101_groups_0 = const()[name = string("attn_output_101_groups_0"), val = int32(1)]; + tensor attn_output_101_cast_fp16 = conv(dilations = attn_output_101_dilations_0, groups = attn_output_101_groups_0, pad = attn_output_101_pad_0, pad_type = attn_output_101_pad_type_0, strides = attn_output_101_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_539_cast_fp16)[name = string("attn_output_101_cast_fp16")]; + tensor x_385_cast_fp16 = add(x = code_embed_35_cast_fp16, y = attn_output_101_cast_fp16)[name = string("x_385_cast_fp16")]; + tensor var_12744_cast_fp16 = mul(x = x_385_cast_fp16, y = x_385_cast_fp16)[name = string("op_12744_cast_fp16")]; + tensor variance_425_axes_0 = const()[name = string("variance_425_axes_0"), val = tensor([1])]; + bool variance_425_keep_dims_0 = const()[name = string("variance_425_keep_dims_0"), val = bool(true)]; + tensor variance_425_cast_fp16 = reduce_mean(axes = variance_425_axes_0, keep_dims = variance_425_keep_dims_0, x = var_12744_cast_fp16)[name = string("variance_425_cast_fp16")]; + fp16 var_12747_to_fp16 = const()[name = string("op_12747_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12748_cast_fp16 = add(x = variance_425_cast_fp16, y = var_12747_to_fp16)[name = string("op_12748_cast_fp16")]; + fp32 var_12749_epsilon_0 = const()[name = string("op_12749_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12749_cast_fp16 = rsqrt(epsilon = var_12749_epsilon_0, x = var_12748_cast_fp16)[name = string("op_12749_cast_fp16")]; + tensor var_12750_cast_fp16 = mul(x = x_385_cast_fp16, y = var_12749_cast_fp16)[name = string("op_12750_cast_fp16")]; + tensor input_541_cast_fp16 = mul(x = var_12750_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_541_cast_fp16")]; + string input_543_pad_type_0 = const()[name = string("input_543_pad_type_0"), val = string("valid")]; + tensor input_543_strides_0 = const()[name = string("input_543_strides_0"), val = tensor([1, 1])]; + tensor input_543_pad_0 = const()[name = string("input_543_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_543_dilations_0 = const()[name = string("input_543_dilations_0"), val = tensor([1, 1])]; + int32 input_543_groups_0 = const()[name = string("input_543_groups_0"), val = int32(1)]; + tensor input_543_cast_fp16 = conv(dilations = input_543_dilations_0, groups = input_543_groups_0, pad = input_543_pad_0, pad_type = input_543_pad_type_0, strides = input_543_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_541_cast_fp16)[name = string("input_543_cast_fp16")]; + tensor var_12758_cast_fp16 = silu(x = input_543_cast_fp16)[name = string("op_12758_cast_fp16")]; + string var_12764_pad_type_0 = const()[name = string("op_12764_pad_type_0"), val = string("valid")]; + tensor var_12764_strides_0 = const()[name = string("op_12764_strides_0"), val = tensor([1, 1])]; + tensor var_12764_pad_0 = const()[name = string("op_12764_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12764_dilations_0 = const()[name = string("op_12764_dilations_0"), val = tensor([1, 1])]; + int32 var_12764_groups_0 = const()[name = string("op_12764_groups_0"), val = int32(1)]; + tensor var_12764_cast_fp16 = conv(dilations = var_12764_dilations_0, groups = var_12764_groups_0, pad = var_12764_pad_0, pad_type = var_12764_pad_type_0, strides = var_12764_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_541_cast_fp16)[name = string("op_12764_cast_fp16")]; + tensor input_545_cast_fp16 = mul(x = var_12758_cast_fp16, y = var_12764_cast_fp16)[name = string("input_545_cast_fp16")]; + string h_101_pad_type_0 = const()[name = string("h_101_pad_type_0"), val = string("valid")]; + tensor h_101_strides_0 = const()[name = string("h_101_strides_0"), val = tensor([1, 1])]; + tensor h_101_pad_0 = const()[name = string("h_101_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_101_dilations_0 = const()[name = string("h_101_dilations_0"), val = tensor([1, 1])]; + int32 h_101_groups_0 = const()[name = string("h_101_groups_0"), val = int32(1)]; + tensor h_101_cast_fp16 = conv(dilations = h_101_dilations_0, groups = h_101_groups_0, pad = h_101_pad_0, pad_type = h_101_pad_type_0, strides = h_101_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_545_cast_fp16)[name = string("h_101_cast_fp16")]; + tensor x_387_cast_fp16 = add(x = x_385_cast_fp16, y = h_101_cast_fp16)[name = string("x_387_cast_fp16")]; + tensor key_cache_103_begin_0 = const()[name = string("key_cache_103_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_103_end_0 = const()[name = string("key_cache_103_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_103_end_mask_0 = const()[name = string("key_cache_103_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_103_cast_fp16 = slice_by_index(begin = key_cache_103_begin_0, end = key_cache_103_end_0, end_mask = key_cache_103_end_mask_0, x = layer_key_caches_21_cast_fp16)[name = string("key_cache_103_cast_fp16")]; + tensor value_cache_103_begin_0 = const()[name = string("value_cache_103_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_103_end_0 = const()[name = string("value_cache_103_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_103_end_mask_0 = const()[name = string("value_cache_103_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_103_cast_fp16 = slice_by_index(begin = value_cache_103_begin_0, end = value_cache_103_end_0, end_mask = value_cache_103_end_mask_0, x = layer_value_caches_21_cast_fp16)[name = string("value_cache_103_cast_fp16")]; + int32 var_12817 = const()[name = string("op_12817"), val = int32(2)]; + int32 var_12821 = const()[name = string("op_12821"), val = int32(3)]; + tensor var_12836_cast_fp16 = mul(x = x_387_cast_fp16, y = x_387_cast_fp16)[name = string("op_12836_cast_fp16")]; + tensor variance_427_axes_0 = const()[name = string("variance_427_axes_0"), val = tensor([1])]; + bool variance_427_keep_dims_0 = const()[name = string("variance_427_keep_dims_0"), val = bool(true)]; + tensor variance_427_cast_fp16 = reduce_mean(axes = variance_427_axes_0, keep_dims = variance_427_keep_dims_0, x = var_12836_cast_fp16)[name = string("variance_427_cast_fp16")]; + fp16 var_12839_to_fp16 = const()[name = string("op_12839_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12840_cast_fp16 = add(x = variance_427_cast_fp16, y = var_12839_to_fp16)[name = string("op_12840_cast_fp16")]; + fp32 var_12841_epsilon_0 = const()[name = string("op_12841_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12841_cast_fp16 = rsqrt(epsilon = var_12841_epsilon_0, x = var_12840_cast_fp16)[name = string("op_12841_cast_fp16")]; + tensor var_12842_cast_fp16 = mul(x = x_387_cast_fp16, y = var_12841_cast_fp16)[name = string("op_12842_cast_fp16")]; + tensor input_547_cast_fp16 = mul(x = var_12842_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_547_cast_fp16")]; + string q_307_pad_type_0 = const()[name = string("q_307_pad_type_0"), val = string("valid")]; + tensor q_307_strides_0 = const()[name = string("q_307_strides_0"), val = tensor([1, 1])]; + tensor q_307_pad_0 = const()[name = string("q_307_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_307_dilations_0 = const()[name = string("q_307_dilations_0"), val = tensor([1, 1])]; + int32 q_307_groups_0 = const()[name = string("q_307_groups_0"), val = int32(1)]; + tensor q_307_cast_fp16 = conv(dilations = q_307_dilations_0, groups = q_307_groups_0, pad = q_307_pad_0, pad_type = q_307_pad_type_0, strides = q_307_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_547_cast_fp16)[name = string("q_307_cast_fp16")]; + string k_307_pad_type_0 = const()[name = string("k_307_pad_type_0"), val = string("valid")]; + tensor k_307_strides_0 = const()[name = string("k_307_strides_0"), val = tensor([1, 1])]; + tensor k_307_pad_0 = const()[name = string("k_307_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_307_dilations_0 = const()[name = string("k_307_dilations_0"), val = tensor([1, 1])]; + int32 k_307_groups_0 = const()[name = string("k_307_groups_0"), val = int32(1)]; + tensor k_307_cast_fp16 = conv(dilations = k_307_dilations_0, groups = k_307_groups_0, pad = k_307_pad_0, pad_type = k_307_pad_type_0, strides = k_307_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_547_cast_fp16)[name = string("k_307_cast_fp16")]; + string v_103_pad_type_0 = const()[name = string("v_103_pad_type_0"), val = string("valid")]; + tensor v_103_strides_0 = const()[name = string("v_103_strides_0"), val = tensor([1, 1])]; + tensor v_103_pad_0 = const()[name = string("v_103_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_103_dilations_0 = const()[name = string("v_103_dilations_0"), val = tensor([1, 1])]; + int32 v_103_groups_0 = const()[name = string("v_103_groups_0"), val = int32(1)]; + tensor v_103_cast_fp16 = conv(dilations = v_103_dilations_0, groups = v_103_groups_0, pad = v_103_pad_0, pad_type = v_103_pad_type_0, strides = v_103_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_547_cast_fp16)[name = string("v_103_cast_fp16")]; + tensor var_12876 = const()[name = string("op_12876"), val = tensor([16, 128, 1, 1])]; + tensor x_389_cast_fp16 = reshape(shape = var_12876, x = q_307_cast_fp16)[name = string("x_389_cast_fp16")]; + tensor var_12879_cast_fp16 = mul(x = x_389_cast_fp16, y = x_389_cast_fp16)[name = string("op_12879_cast_fp16")]; + tensor variance_429_axes_0 = const()[name = string("variance_429_axes_0"), val = tensor([1])]; + bool variance_429_keep_dims_0 = const()[name = string("variance_429_keep_dims_0"), val = bool(true)]; + tensor variance_429_cast_fp16 = reduce_mean(axes = variance_429_axes_0, keep_dims = variance_429_keep_dims_0, x = var_12879_cast_fp16)[name = string("variance_429_cast_fp16")]; + fp16 var_12882_to_fp16 = const()[name = string("op_12882_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12883_cast_fp16 = add(x = variance_429_cast_fp16, y = var_12882_to_fp16)[name = string("op_12883_cast_fp16")]; + fp32 var_12884_epsilon_0 = const()[name = string("op_12884_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12884_cast_fp16 = rsqrt(epsilon = var_12884_epsilon_0, x = var_12883_cast_fp16)[name = string("op_12884_cast_fp16")]; + tensor var_12885_cast_fp16 = mul(x = x_389_cast_fp16, y = var_12884_cast_fp16)[name = string("op_12885_cast_fp16")]; + tensor q_309_cast_fp16 = mul(x = var_12885_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_309_cast_fp16")]; + tensor var_12887 = const()[name = string("op_12887"), val = tensor([8, 128, 1, 1])]; + tensor x_391_cast_fp16 = reshape(shape = var_12887, x = k_307_cast_fp16)[name = string("x_391_cast_fp16")]; + tensor var_12890_cast_fp16 = mul(x = x_391_cast_fp16, y = x_391_cast_fp16)[name = string("op_12890_cast_fp16")]; + tensor variance_431_axes_0 = const()[name = string("variance_431_axes_0"), val = tensor([1])]; + bool variance_431_keep_dims_0 = const()[name = string("variance_431_keep_dims_0"), val = bool(true)]; + tensor variance_431_cast_fp16 = reduce_mean(axes = variance_431_axes_0, keep_dims = variance_431_keep_dims_0, x = var_12890_cast_fp16)[name = string("variance_431_cast_fp16")]; + fp16 var_12893_to_fp16 = const()[name = string("op_12893_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12894_cast_fp16 = add(x = variance_431_cast_fp16, y = var_12893_to_fp16)[name = string("op_12894_cast_fp16")]; + fp32 var_12895_epsilon_0 = const()[name = string("op_12895_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12895_cast_fp16 = rsqrt(epsilon = var_12895_epsilon_0, x = var_12894_cast_fp16)[name = string("op_12895_cast_fp16")]; + tensor var_12896_cast_fp16 = mul(x = x_391_cast_fp16, y = var_12895_cast_fp16)[name = string("op_12896_cast_fp16")]; + tensor k_309_cast_fp16 = mul(x = var_12896_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_309_cast_fp16")]; + tensor var_12898 = const()[name = string("op_12898"), val = tensor([1, 16, 128, 1])]; + tensor z_205_cast_fp16 = reshape(shape = var_12898, x = q_309_cast_fp16)[name = string("z_205_cast_fp16")]; + tensor var_12900 = const()[name = string("op_12900"), val = tensor([1, 8, 128, 1])]; + tensor z_207_cast_fp16 = reshape(shape = var_12900, x = k_309_cast_fp16)[name = string("z_207_cast_fp16")]; + tensor z1_205_begin_0 = const()[name = string("z1_205_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_205_end_0 = const()[name = string("z1_205_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_205_end_mask_0 = const()[name = string("z1_205_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_205_cast_fp16 = slice_by_index(begin = z1_205_begin_0, end = z1_205_end_0, end_mask = z1_205_end_mask_0, x = z_205_cast_fp16)[name = string("z1_205_cast_fp16")]; + tensor z2_205_begin_0 = const()[name = string("z2_205_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_205_end_0 = const()[name = string("z2_205_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_205_end_mask_0 = const()[name = string("z2_205_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_205_cast_fp16 = slice_by_index(begin = z2_205_begin_0, end = z2_205_end_0, end_mask = z2_205_end_mask_0, x = z_205_cast_fp16)[name = string("z2_205_cast_fp16")]; + tensor var_12908_cast_fp16 = mul(x = z_205_cast_fp16, y = cos_101_to_fp16)[name = string("op_12908_cast_fp16")]; + fp16 const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12909_cast_fp16 = mul(x = z2_205_cast_fp16, y = const_113_promoted_to_fp16)[name = string("op_12909_cast_fp16")]; + bool var_12911_interleave_0 = const()[name = string("op_12911_interleave_0"), val = bool(false)]; + tensor var_12911_cast_fp16 = concat(axis = var_12817, interleave = var_12911_interleave_0, values = (var_12909_cast_fp16, z1_205_cast_fp16))[name = string("op_12911_cast_fp16")]; + tensor var_12912_cast_fp16 = mul(x = var_12911_cast_fp16, y = sin_101_to_fp16)[name = string("op_12912_cast_fp16")]; + tensor q_311_cast_fp16 = add(x = var_12908_cast_fp16, y = var_12912_cast_fp16)[name = string("q_311_cast_fp16")]; + tensor z1_207_begin_0 = const()[name = string("z1_207_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_207_end_0 = const()[name = string("z1_207_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_207_end_mask_0 = const()[name = string("z1_207_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_207_cast_fp16 = slice_by_index(begin = z1_207_begin_0, end = z1_207_end_0, end_mask = z1_207_end_mask_0, x = z_207_cast_fp16)[name = string("z1_207_cast_fp16")]; + tensor z2_207_begin_0 = const()[name = string("z2_207_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_207_end_0 = const()[name = string("z2_207_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_207_end_mask_0 = const()[name = string("z2_207_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_207_cast_fp16 = slice_by_index(begin = z2_207_begin_0, end = z2_207_end_0, end_mask = z2_207_end_mask_0, x = z_207_cast_fp16)[name = string("z2_207_cast_fp16")]; + tensor var_12920_cast_fp16 = mul(x = z_207_cast_fp16, y = cos_101_to_fp16)[name = string("op_12920_cast_fp16")]; + fp16 const_114_promoted_to_fp16 = const()[name = string("const_114_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12921_cast_fp16 = mul(x = z2_207_cast_fp16, y = const_114_promoted_to_fp16)[name = string("op_12921_cast_fp16")]; + bool var_12923_interleave_0 = const()[name = string("op_12923_interleave_0"), val = bool(false)]; + tensor var_12923_cast_fp16 = concat(axis = var_12817, interleave = var_12923_interleave_0, values = (var_12921_cast_fp16, z1_207_cast_fp16))[name = string("op_12923_cast_fp16")]; + tensor var_12924_cast_fp16 = mul(x = var_12923_cast_fp16, y = sin_101_to_fp16)[name = string("op_12924_cast_fp16")]; + tensor k_311_cast_fp16 = add(x = var_12920_cast_fp16, y = var_12924_cast_fp16)[name = string("k_311_cast_fp16")]; + tensor var_12926 = const()[name = string("op_12926"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_103_cast_fp16 = reshape(shape = var_12926, x = k_311_cast_fp16)[name = string("cur_key_103_cast_fp16")]; + tensor var_12928_to_fp16 = const()[name = string("op_12928_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642240)))]; + tensor var_12929_cast_fp16 = mul(x = key_cache_103_cast_fp16, y = var_12928_to_fp16)[name = string("op_12929_cast_fp16")]; + tensor upd_103_to_fp16 = const()[name = string("upd_103_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642368)))]; + tensor var_12930_cast_fp16 = mul(x = cur_key_103_cast_fp16, y = upd_103_to_fp16)[name = string("op_12930_cast_fp16")]; + tensor key_103_cast_fp16 = add(x = var_12929_cast_fp16, y = var_12930_cast_fp16)[name = string("key_103_cast_fp16")]; + tensor var_12932_to_fp16 = const()[name = string("op_12932_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642240)))]; + tensor var_12933_cast_fp16 = mul(x = value_cache_103_cast_fp16, y = var_12932_to_fp16)[name = string("op_12933_cast_fp16")]; + tensor var_12934_cast_fp16 = mul(x = v_103_cast_fp16, y = upd_103_to_fp16)[name = string("op_12934_cast_fp16")]; + tensor value_103_cast_fp16 = add(x = var_12933_cast_fp16, y = var_12934_cast_fp16)[name = string("value_103_cast_fp16")]; + tensor var_12936 = const()[name = string("op_12936"), val = tensor([1, 8, 128, 16])]; + tensor kh_205_cast_fp16 = reshape(shape = var_12936, x = key_103_cast_fp16)[name = string("kh_205_cast_fp16")]; + tensor var_12938 = const()[name = string("op_12938"), val = tensor([1, 8, 128, 16])]; + tensor vh_205_cast_fp16 = reshape(shape = var_12938, x = value_103_cast_fp16)[name = string("vh_205_cast_fp16")]; + tensor transpose_204_perm_0 = const()[name = string("transpose_204_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_102_reps_0 = const()[name = string("tile_102_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_204_cast_fp16 = transpose(perm = transpose_204_perm_0, x = kh_205_cast_fp16)[name = string("transpose_173")]; + tensor tile_102_cast_fp16 = tile(reps = tile_102_reps_0, x = transpose_204_cast_fp16)[name = string("tile_102_cast_fp16")]; + tensor concat_257 = const()[name = string("concat_257"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_204_cast_fp16 = reshape(shape = concat_257, x = tile_102_cast_fp16)[name = string("reshape_204_cast_fp16")]; + tensor transpose_205_perm_0 = const()[name = string("transpose_205_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_258 = const()[name = string("concat_258"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_205_cast_fp16 = transpose(perm = transpose_205_perm_0, x = reshape_204_cast_fp16)[name = string("transpose_172")]; + tensor reshape_205_cast_fp16 = reshape(shape = concat_258, x = transpose_205_cast_fp16)[name = string("reshape_205_cast_fp16")]; + tensor transpose_206_perm_0 = const()[name = string("transpose_206_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_103_reps_0 = const()[name = string("tile_103_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_206_cast_fp16 = transpose(perm = transpose_206_perm_0, x = vh_205_cast_fp16)[name = string("transpose_171")]; + tensor tile_103_cast_fp16 = tile(reps = tile_103_reps_0, x = transpose_206_cast_fp16)[name = string("tile_103_cast_fp16")]; + tensor concat_259 = const()[name = string("concat_259"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_206_cast_fp16 = reshape(shape = concat_259, x = tile_103_cast_fp16)[name = string("reshape_206_cast_fp16")]; + tensor transpose_207_perm_0 = const()[name = string("transpose_207_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_260 = const()[name = string("concat_260"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_207_cast_fp16 = transpose(perm = transpose_207_perm_0, x = reshape_206_cast_fp16)[name = string("transpose_170")]; + tensor reshape_207_cast_fp16 = reshape(shape = concat_260, x = transpose_207_cast_fp16)[name = string("reshape_207_cast_fp16")]; + fp16 var_12942_to_fp16 = const()[name = string("op_12942_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_12943_cast_fp16 = mul(x = q_311_cast_fp16, y = var_12942_to_fp16)[name = string("op_12943_cast_fp16")]; + tensor transpose_521_perm_0 = const()[name = string("transpose_521_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_223_transpose_x_1 = const()[name = string("w_223_transpose_x_1"), val = bool(true)]; + bool w_223_transpose_y_1 = const()[name = string("w_223_transpose_y_1"), val = bool(false)]; + tensor transpose_521_cast_fp16 = transpose(perm = transpose_521_perm_0, x = reshape_205_cast_fp16)[name = string("transpose_169")]; + tensor w_223_cast_fp16 = matmul(transpose_x = w_223_transpose_x_1, transpose_y = w_223_transpose_y_1, x = var_12943_cast_fp16, y = transpose_521_cast_fp16)[name = string("w_223_cast_fp16")]; + tensor pad_103_to_fp16 = const()[name = string("pad_103_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642496)))]; + tensor var_12946_cast_fp16 = add(x = w_223_cast_fp16, y = pad_103_to_fp16)[name = string("op_12946_cast_fp16")]; + tensor w_225_cast_fp16 = softmax(axis = var_12821, x = var_12946_cast_fp16)[name = string("w_225_cast_fp16")]; + tensor transpose_522_perm_0 = const()[name = string("transpose_522_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_103_transpose_x_1 = const()[name = string("attn_103_transpose_x_1"), val = bool(false)]; + bool attn_103_transpose_y_1 = const()[name = string("attn_103_transpose_y_1"), val = bool(true)]; + tensor transpose_522_cast_fp16 = transpose(perm = transpose_522_perm_0, x = reshape_207_cast_fp16)[name = string("transpose_168")]; + tensor attn_103_cast_fp16 = matmul(transpose_x = attn_103_transpose_x_1, transpose_y = attn_103_transpose_y_1, x = transpose_522_cast_fp16, y = w_225_cast_fp16)[name = string("attn_103_cast_fp16")]; + tensor var_12950 = const()[name = string("op_12950"), val = tensor([1, 2048, 1, 1])]; + tensor input_549_cast_fp16 = reshape(shape = var_12950, x = attn_103_cast_fp16)[name = string("input_549_cast_fp16")]; + string attn_output_103_pad_type_0 = const()[name = string("attn_output_103_pad_type_0"), val = string("valid")]; + tensor attn_output_103_strides_0 = const()[name = string("attn_output_103_strides_0"), val = tensor([1, 1])]; + tensor attn_output_103_pad_0 = const()[name = string("attn_output_103_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_103_dilations_0 = const()[name = string("attn_output_103_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_103_groups_0 = const()[name = string("attn_output_103_groups_0"), val = int32(1)]; + tensor attn_output_103_cast_fp16 = conv(dilations = attn_output_103_dilations_0, groups = attn_output_103_groups_0, pad = attn_output_103_pad_0, pad_type = attn_output_103_pad_type_0, strides = attn_output_103_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_549_cast_fp16)[name = string("attn_output_103_cast_fp16")]; + tensor x_393_cast_fp16 = add(x = x_387_cast_fp16, y = attn_output_103_cast_fp16)[name = string("x_393_cast_fp16")]; + tensor var_12964_cast_fp16 = mul(x = x_393_cast_fp16, y = x_393_cast_fp16)[name = string("op_12964_cast_fp16")]; + tensor variance_433_axes_0 = const()[name = string("variance_433_axes_0"), val = tensor([1])]; + bool variance_433_keep_dims_0 = const()[name = string("variance_433_keep_dims_0"), val = bool(true)]; + tensor variance_433_cast_fp16 = reduce_mean(axes = variance_433_axes_0, keep_dims = variance_433_keep_dims_0, x = var_12964_cast_fp16)[name = string("variance_433_cast_fp16")]; + fp16 var_12967_to_fp16 = const()[name = string("op_12967_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12968_cast_fp16 = add(x = variance_433_cast_fp16, y = var_12967_to_fp16)[name = string("op_12968_cast_fp16")]; + fp32 var_12969_epsilon_0 = const()[name = string("op_12969_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12969_cast_fp16 = rsqrt(epsilon = var_12969_epsilon_0, x = var_12968_cast_fp16)[name = string("op_12969_cast_fp16")]; + tensor var_12970_cast_fp16 = mul(x = x_393_cast_fp16, y = var_12969_cast_fp16)[name = string("op_12970_cast_fp16")]; + tensor input_551_cast_fp16 = mul(x = var_12970_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_551_cast_fp16")]; + string input_553_pad_type_0 = const()[name = string("input_553_pad_type_0"), val = string("valid")]; + tensor input_553_strides_0 = const()[name = string("input_553_strides_0"), val = tensor([1, 1])]; + tensor input_553_pad_0 = const()[name = string("input_553_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_553_dilations_0 = const()[name = string("input_553_dilations_0"), val = tensor([1, 1])]; + int32 input_553_groups_0 = const()[name = string("input_553_groups_0"), val = int32(1)]; + tensor input_553_cast_fp16 = conv(dilations = input_553_dilations_0, groups = input_553_groups_0, pad = input_553_pad_0, pad_type = input_553_pad_type_0, strides = input_553_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_551_cast_fp16)[name = string("input_553_cast_fp16")]; + tensor var_12978_cast_fp16 = silu(x = input_553_cast_fp16)[name = string("op_12978_cast_fp16")]; + string var_12984_pad_type_0 = const()[name = string("op_12984_pad_type_0"), val = string("valid")]; + tensor var_12984_strides_0 = const()[name = string("op_12984_strides_0"), val = tensor([1, 1])]; + tensor var_12984_pad_0 = const()[name = string("op_12984_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12984_dilations_0 = const()[name = string("op_12984_dilations_0"), val = tensor([1, 1])]; + int32 var_12984_groups_0 = const()[name = string("op_12984_groups_0"), val = int32(1)]; + tensor var_12984_cast_fp16 = conv(dilations = var_12984_dilations_0, groups = var_12984_groups_0, pad = var_12984_pad_0, pad_type = var_12984_pad_type_0, strides = var_12984_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_551_cast_fp16)[name = string("op_12984_cast_fp16")]; + tensor input_555_cast_fp16 = mul(x = var_12978_cast_fp16, y = var_12984_cast_fp16)[name = string("input_555_cast_fp16")]; + string h_103_pad_type_0 = const()[name = string("h_103_pad_type_0"), val = string("valid")]; + tensor h_103_strides_0 = const()[name = string("h_103_strides_0"), val = tensor([1, 1])]; + tensor h_103_pad_0 = const()[name = string("h_103_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_103_dilations_0 = const()[name = string("h_103_dilations_0"), val = tensor([1, 1])]; + int32 h_103_groups_0 = const()[name = string("h_103_groups_0"), val = int32(1)]; + tensor h_103_cast_fp16 = conv(dilations = h_103_dilations_0, groups = h_103_groups_0, pad = h_103_pad_0, pad_type = h_103_pad_type_0, strides = h_103_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_555_cast_fp16)[name = string("h_103_cast_fp16")]; + tensor x_395_cast_fp16 = add(x = x_393_cast_fp16, y = h_103_cast_fp16)[name = string("x_395_cast_fp16")]; + tensor key_cache_105_begin_0 = const()[name = string("key_cache_105_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_105_end_0 = const()[name = string("key_cache_105_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_105_end_mask_0 = const()[name = string("key_cache_105_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_105_cast_fp16 = slice_by_index(begin = key_cache_105_begin_0, end = key_cache_105_end_0, end_mask = key_cache_105_end_mask_0, x = layer_key_caches_21_cast_fp16)[name = string("key_cache_105_cast_fp16")]; + tensor value_cache_105_begin_0 = const()[name = string("value_cache_105_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_105_end_0 = const()[name = string("value_cache_105_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_105_end_mask_0 = const()[name = string("value_cache_105_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_105_cast_fp16 = slice_by_index(begin = value_cache_105_begin_0, end = value_cache_105_end_0, end_mask = value_cache_105_end_mask_0, x = layer_value_caches_21_cast_fp16)[name = string("value_cache_105_cast_fp16")]; + int32 var_13037 = const()[name = string("op_13037"), val = int32(2)]; + int32 var_13041 = const()[name = string("op_13041"), val = int32(3)]; + tensor var_13056_cast_fp16 = mul(x = x_395_cast_fp16, y = x_395_cast_fp16)[name = string("op_13056_cast_fp16")]; + tensor variance_435_axes_0 = const()[name = string("variance_435_axes_0"), val = tensor([1])]; + bool variance_435_keep_dims_0 = const()[name = string("variance_435_keep_dims_0"), val = bool(true)]; + tensor variance_435_cast_fp16 = reduce_mean(axes = variance_435_axes_0, keep_dims = variance_435_keep_dims_0, x = var_13056_cast_fp16)[name = string("variance_435_cast_fp16")]; + fp16 var_13059_to_fp16 = const()[name = string("op_13059_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13060_cast_fp16 = add(x = variance_435_cast_fp16, y = var_13059_to_fp16)[name = string("op_13060_cast_fp16")]; + fp32 var_13061_epsilon_0 = const()[name = string("op_13061_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13061_cast_fp16 = rsqrt(epsilon = var_13061_epsilon_0, x = var_13060_cast_fp16)[name = string("op_13061_cast_fp16")]; + tensor var_13062_cast_fp16 = mul(x = x_395_cast_fp16, y = var_13061_cast_fp16)[name = string("op_13062_cast_fp16")]; + tensor input_557_cast_fp16 = mul(x = var_13062_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_557_cast_fp16")]; + string q_313_pad_type_0 = const()[name = string("q_313_pad_type_0"), val = string("valid")]; + tensor q_313_strides_0 = const()[name = string("q_313_strides_0"), val = tensor([1, 1])]; + tensor q_313_pad_0 = const()[name = string("q_313_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_313_dilations_0 = const()[name = string("q_313_dilations_0"), val = tensor([1, 1])]; + int32 q_313_groups_0 = const()[name = string("q_313_groups_0"), val = int32(1)]; + tensor q_313_cast_fp16 = conv(dilations = q_313_dilations_0, groups = q_313_groups_0, pad = q_313_pad_0, pad_type = q_313_pad_type_0, strides = q_313_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_557_cast_fp16)[name = string("q_313_cast_fp16")]; + string k_313_pad_type_0 = const()[name = string("k_313_pad_type_0"), val = string("valid")]; + tensor k_313_strides_0 = const()[name = string("k_313_strides_0"), val = tensor([1, 1])]; + tensor k_313_pad_0 = const()[name = string("k_313_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_313_dilations_0 = const()[name = string("k_313_dilations_0"), val = tensor([1, 1])]; + int32 k_313_groups_0 = const()[name = string("k_313_groups_0"), val = int32(1)]; + tensor k_313_cast_fp16 = conv(dilations = k_313_dilations_0, groups = k_313_groups_0, pad = k_313_pad_0, pad_type = k_313_pad_type_0, strides = k_313_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_557_cast_fp16)[name = string("k_313_cast_fp16")]; + string v_105_pad_type_0 = const()[name = string("v_105_pad_type_0"), val = string("valid")]; + tensor v_105_strides_0 = const()[name = string("v_105_strides_0"), val = tensor([1, 1])]; + tensor v_105_pad_0 = const()[name = string("v_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_105_dilations_0 = const()[name = string("v_105_dilations_0"), val = tensor([1, 1])]; + int32 v_105_groups_0 = const()[name = string("v_105_groups_0"), val = int32(1)]; + tensor v_105_cast_fp16 = conv(dilations = v_105_dilations_0, groups = v_105_groups_0, pad = v_105_pad_0, pad_type = v_105_pad_type_0, strides = v_105_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_557_cast_fp16)[name = string("v_105_cast_fp16")]; + tensor var_13096 = const()[name = string("op_13096"), val = tensor([16, 128, 1, 1])]; + tensor x_397_cast_fp16 = reshape(shape = var_13096, x = q_313_cast_fp16)[name = string("x_397_cast_fp16")]; + tensor var_13099_cast_fp16 = mul(x = x_397_cast_fp16, y = x_397_cast_fp16)[name = string("op_13099_cast_fp16")]; + tensor variance_437_axes_0 = const()[name = string("variance_437_axes_0"), val = tensor([1])]; + bool variance_437_keep_dims_0 = const()[name = string("variance_437_keep_dims_0"), val = bool(true)]; + tensor variance_437_cast_fp16 = reduce_mean(axes = variance_437_axes_0, keep_dims = variance_437_keep_dims_0, x = var_13099_cast_fp16)[name = string("variance_437_cast_fp16")]; + fp16 var_13102_to_fp16 = const()[name = string("op_13102_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13103_cast_fp16 = add(x = variance_437_cast_fp16, y = var_13102_to_fp16)[name = string("op_13103_cast_fp16")]; + fp32 var_13104_epsilon_0 = const()[name = string("op_13104_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13104_cast_fp16 = rsqrt(epsilon = var_13104_epsilon_0, x = var_13103_cast_fp16)[name = string("op_13104_cast_fp16")]; + tensor var_13105_cast_fp16 = mul(x = x_397_cast_fp16, y = var_13104_cast_fp16)[name = string("op_13105_cast_fp16")]; + tensor q_315_cast_fp16 = mul(x = var_13105_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_315_cast_fp16")]; + tensor var_13107 = const()[name = string("op_13107"), val = tensor([8, 128, 1, 1])]; + tensor x_399_cast_fp16 = reshape(shape = var_13107, x = k_313_cast_fp16)[name = string("x_399_cast_fp16")]; + tensor var_13110_cast_fp16 = mul(x = x_399_cast_fp16, y = x_399_cast_fp16)[name = string("op_13110_cast_fp16")]; + tensor variance_439_axes_0 = const()[name = string("variance_439_axes_0"), val = tensor([1])]; + bool variance_439_keep_dims_0 = const()[name = string("variance_439_keep_dims_0"), val = bool(true)]; + tensor variance_439_cast_fp16 = reduce_mean(axes = variance_439_axes_0, keep_dims = variance_439_keep_dims_0, x = var_13110_cast_fp16)[name = string("variance_439_cast_fp16")]; + fp16 var_13113_to_fp16 = const()[name = string("op_13113_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13114_cast_fp16 = add(x = variance_439_cast_fp16, y = var_13113_to_fp16)[name = string("op_13114_cast_fp16")]; + fp32 var_13115_epsilon_0 = const()[name = string("op_13115_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13115_cast_fp16 = rsqrt(epsilon = var_13115_epsilon_0, x = var_13114_cast_fp16)[name = string("op_13115_cast_fp16")]; + tensor var_13116_cast_fp16 = mul(x = x_399_cast_fp16, y = var_13115_cast_fp16)[name = string("op_13116_cast_fp16")]; + tensor k_315_cast_fp16 = mul(x = var_13116_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_315_cast_fp16")]; + tensor var_13118 = const()[name = string("op_13118"), val = tensor([1, 16, 128, 1])]; + tensor z_209_cast_fp16 = reshape(shape = var_13118, x = q_315_cast_fp16)[name = string("z_209_cast_fp16")]; + tensor var_13120 = const()[name = string("op_13120"), val = tensor([1, 8, 128, 1])]; + tensor z_211_cast_fp16 = reshape(shape = var_13120, x = k_315_cast_fp16)[name = string("z_211_cast_fp16")]; + tensor z1_209_begin_0 = const()[name = string("z1_209_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_209_end_0 = const()[name = string("z1_209_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_209_end_mask_0 = const()[name = string("z1_209_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_209_cast_fp16 = slice_by_index(begin = z1_209_begin_0, end = z1_209_end_0, end_mask = z1_209_end_mask_0, x = z_209_cast_fp16)[name = string("z1_209_cast_fp16")]; + tensor z2_209_begin_0 = const()[name = string("z2_209_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_209_end_0 = const()[name = string("z2_209_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_209_end_mask_0 = const()[name = string("z2_209_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_209_cast_fp16 = slice_by_index(begin = z2_209_begin_0, end = z2_209_end_0, end_mask = z2_209_end_mask_0, x = z_209_cast_fp16)[name = string("z2_209_cast_fp16")]; + tensor var_13128_cast_fp16 = mul(x = z_209_cast_fp16, y = cos_101_to_fp16)[name = string("op_13128_cast_fp16")]; + fp16 const_115_promoted_to_fp16 = const()[name = string("const_115_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13129_cast_fp16 = mul(x = z2_209_cast_fp16, y = const_115_promoted_to_fp16)[name = string("op_13129_cast_fp16")]; + bool var_13131_interleave_0 = const()[name = string("op_13131_interleave_0"), val = bool(false)]; + tensor var_13131_cast_fp16 = concat(axis = var_13037, interleave = var_13131_interleave_0, values = (var_13129_cast_fp16, z1_209_cast_fp16))[name = string("op_13131_cast_fp16")]; + tensor var_13132_cast_fp16 = mul(x = var_13131_cast_fp16, y = sin_101_to_fp16)[name = string("op_13132_cast_fp16")]; + tensor q_317_cast_fp16 = add(x = var_13128_cast_fp16, y = var_13132_cast_fp16)[name = string("q_317_cast_fp16")]; + tensor z1_211_begin_0 = const()[name = string("z1_211_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_211_end_0 = const()[name = string("z1_211_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_211_end_mask_0 = const()[name = string("z1_211_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_211_cast_fp16 = slice_by_index(begin = z1_211_begin_0, end = z1_211_end_0, end_mask = z1_211_end_mask_0, x = z_211_cast_fp16)[name = string("z1_211_cast_fp16")]; + tensor z2_211_begin_0 = const()[name = string("z2_211_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_211_end_0 = const()[name = string("z2_211_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_211_end_mask_0 = const()[name = string("z2_211_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_211_cast_fp16 = slice_by_index(begin = z2_211_begin_0, end = z2_211_end_0, end_mask = z2_211_end_mask_0, x = z_211_cast_fp16)[name = string("z2_211_cast_fp16")]; + tensor var_13140_cast_fp16 = mul(x = z_211_cast_fp16, y = cos_101_to_fp16)[name = string("op_13140_cast_fp16")]; + fp16 const_116_promoted_to_fp16 = const()[name = string("const_116_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13141_cast_fp16 = mul(x = z2_211_cast_fp16, y = const_116_promoted_to_fp16)[name = string("op_13141_cast_fp16")]; + bool var_13143_interleave_0 = const()[name = string("op_13143_interleave_0"), val = bool(false)]; + tensor var_13143_cast_fp16 = concat(axis = var_13037, interleave = var_13143_interleave_0, values = (var_13141_cast_fp16, z1_211_cast_fp16))[name = string("op_13143_cast_fp16")]; + tensor var_13144_cast_fp16 = mul(x = var_13143_cast_fp16, y = sin_101_to_fp16)[name = string("op_13144_cast_fp16")]; + tensor k_317_cast_fp16 = add(x = var_13140_cast_fp16, y = var_13144_cast_fp16)[name = string("k_317_cast_fp16")]; + tensor var_13146 = const()[name = string("op_13146"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_105_cast_fp16 = reshape(shape = var_13146, x = k_317_cast_fp16)[name = string("cur_key_105_cast_fp16")]; + tensor var_13148_to_fp16 = const()[name = string("op_13148_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642240)))]; + tensor var_13149_cast_fp16 = mul(x = key_cache_105_cast_fp16, y = var_13148_to_fp16)[name = string("op_13149_cast_fp16")]; + tensor upd_105_to_fp16 = const()[name = string("upd_105_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642368)))]; + tensor var_13150_cast_fp16 = mul(x = cur_key_105_cast_fp16, y = upd_105_to_fp16)[name = string("op_13150_cast_fp16")]; + tensor key_105_cast_fp16 = add(x = var_13149_cast_fp16, y = var_13150_cast_fp16)[name = string("key_105_cast_fp16")]; + tensor var_13152_to_fp16 = const()[name = string("op_13152_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642240)))]; + tensor var_13153_cast_fp16 = mul(x = value_cache_105_cast_fp16, y = var_13152_to_fp16)[name = string("op_13153_cast_fp16")]; + tensor var_13154_cast_fp16 = mul(x = v_105_cast_fp16, y = upd_105_to_fp16)[name = string("op_13154_cast_fp16")]; + tensor value_105_cast_fp16 = add(x = var_13153_cast_fp16, y = var_13154_cast_fp16)[name = string("value_105_cast_fp16")]; + tensor var_13156 = const()[name = string("op_13156"), val = tensor([1, 8, 128, 16])]; + tensor kh_209_cast_fp16 = reshape(shape = var_13156, x = key_105_cast_fp16)[name = string("kh_209_cast_fp16")]; + tensor var_13158 = const()[name = string("op_13158"), val = tensor([1, 8, 128, 16])]; + tensor vh_209_cast_fp16 = reshape(shape = var_13158, x = value_105_cast_fp16)[name = string("vh_209_cast_fp16")]; + tensor transpose_208_perm_0 = const()[name = string("transpose_208_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_104_reps_0 = const()[name = string("tile_104_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_208_cast_fp16 = transpose(perm = transpose_208_perm_0, x = kh_209_cast_fp16)[name = string("transpose_167")]; + tensor tile_104_cast_fp16 = tile(reps = tile_104_reps_0, x = transpose_208_cast_fp16)[name = string("tile_104_cast_fp16")]; + tensor concat_261 = const()[name = string("concat_261"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_208_cast_fp16 = reshape(shape = concat_261, x = tile_104_cast_fp16)[name = string("reshape_208_cast_fp16")]; + tensor transpose_209_perm_0 = const()[name = string("transpose_209_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_262 = const()[name = string("concat_262"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_209_cast_fp16 = transpose(perm = transpose_209_perm_0, x = reshape_208_cast_fp16)[name = string("transpose_166")]; + tensor reshape_209_cast_fp16 = reshape(shape = concat_262, x = transpose_209_cast_fp16)[name = string("reshape_209_cast_fp16")]; + tensor transpose_210_perm_0 = const()[name = string("transpose_210_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_105_reps_0 = const()[name = string("tile_105_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_210_cast_fp16 = transpose(perm = transpose_210_perm_0, x = vh_209_cast_fp16)[name = string("transpose_165")]; + tensor tile_105_cast_fp16 = tile(reps = tile_105_reps_0, x = transpose_210_cast_fp16)[name = string("tile_105_cast_fp16")]; + tensor concat_263 = const()[name = string("concat_263"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_210_cast_fp16 = reshape(shape = concat_263, x = tile_105_cast_fp16)[name = string("reshape_210_cast_fp16")]; + tensor transpose_211_perm_0 = const()[name = string("transpose_211_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_264 = const()[name = string("concat_264"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_211_cast_fp16 = transpose(perm = transpose_211_perm_0, x = reshape_210_cast_fp16)[name = string("transpose_164")]; + tensor reshape_211_cast_fp16 = reshape(shape = concat_264, x = transpose_211_cast_fp16)[name = string("reshape_211_cast_fp16")]; + fp16 var_13162_to_fp16 = const()[name = string("op_13162_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_13163_cast_fp16 = mul(x = q_317_cast_fp16, y = var_13162_to_fp16)[name = string("op_13163_cast_fp16")]; + tensor transpose_525_perm_0 = const()[name = string("transpose_525_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_227_transpose_x_1 = const()[name = string("w_227_transpose_x_1"), val = bool(true)]; + bool w_227_transpose_y_1 = const()[name = string("w_227_transpose_y_1"), val = bool(false)]; + tensor transpose_525_cast_fp16 = transpose(perm = transpose_525_perm_0, x = reshape_209_cast_fp16)[name = string("transpose_163")]; + tensor w_227_cast_fp16 = matmul(transpose_x = w_227_transpose_x_1, transpose_y = w_227_transpose_y_1, x = var_13163_cast_fp16, y = transpose_525_cast_fp16)[name = string("w_227_cast_fp16")]; + tensor pad_105_to_fp16 = const()[name = string("pad_105_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642496)))]; + tensor var_13166_cast_fp16 = add(x = w_227_cast_fp16, y = pad_105_to_fp16)[name = string("op_13166_cast_fp16")]; + tensor w_229_cast_fp16 = softmax(axis = var_13041, x = var_13166_cast_fp16)[name = string("w_229_cast_fp16")]; + tensor transpose_526_perm_0 = const()[name = string("transpose_526_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_105_transpose_x_1 = const()[name = string("attn_105_transpose_x_1"), val = bool(false)]; + bool attn_105_transpose_y_1 = const()[name = string("attn_105_transpose_y_1"), val = bool(true)]; + tensor transpose_526_cast_fp16 = transpose(perm = transpose_526_perm_0, x = reshape_211_cast_fp16)[name = string("transpose_162")]; + tensor attn_105_cast_fp16 = matmul(transpose_x = attn_105_transpose_x_1, transpose_y = attn_105_transpose_y_1, x = transpose_526_cast_fp16, y = w_229_cast_fp16)[name = string("attn_105_cast_fp16")]; + tensor var_13170 = const()[name = string("op_13170"), val = tensor([1, 2048, 1, 1])]; + tensor input_559_cast_fp16 = reshape(shape = var_13170, x = attn_105_cast_fp16)[name = string("input_559_cast_fp16")]; + string attn_output_105_pad_type_0 = const()[name = string("attn_output_105_pad_type_0"), val = string("valid")]; + tensor attn_output_105_strides_0 = const()[name = string("attn_output_105_strides_0"), val = tensor([1, 1])]; + tensor attn_output_105_pad_0 = const()[name = string("attn_output_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_105_dilations_0 = const()[name = string("attn_output_105_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_105_groups_0 = const()[name = string("attn_output_105_groups_0"), val = int32(1)]; + tensor attn_output_105_cast_fp16 = conv(dilations = attn_output_105_dilations_0, groups = attn_output_105_groups_0, pad = attn_output_105_pad_0, pad_type = attn_output_105_pad_type_0, strides = attn_output_105_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_559_cast_fp16)[name = string("attn_output_105_cast_fp16")]; + tensor x_401_cast_fp16 = add(x = x_395_cast_fp16, y = attn_output_105_cast_fp16)[name = string("x_401_cast_fp16")]; + tensor var_13184_cast_fp16 = mul(x = x_401_cast_fp16, y = x_401_cast_fp16)[name = string("op_13184_cast_fp16")]; + tensor variance_441_axes_0 = const()[name = string("variance_441_axes_0"), val = tensor([1])]; + bool variance_441_keep_dims_0 = const()[name = string("variance_441_keep_dims_0"), val = bool(true)]; + tensor variance_441_cast_fp16 = reduce_mean(axes = variance_441_axes_0, keep_dims = variance_441_keep_dims_0, x = var_13184_cast_fp16)[name = string("variance_441_cast_fp16")]; + fp16 var_13187_to_fp16 = const()[name = string("op_13187_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13188_cast_fp16 = add(x = variance_441_cast_fp16, y = var_13187_to_fp16)[name = string("op_13188_cast_fp16")]; + fp32 var_13189_epsilon_0 = const()[name = string("op_13189_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13189_cast_fp16 = rsqrt(epsilon = var_13189_epsilon_0, x = var_13188_cast_fp16)[name = string("op_13189_cast_fp16")]; + tensor var_13190_cast_fp16 = mul(x = x_401_cast_fp16, y = var_13189_cast_fp16)[name = string("op_13190_cast_fp16")]; + tensor input_561_cast_fp16 = mul(x = var_13190_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_561_cast_fp16")]; + string input_563_pad_type_0 = const()[name = string("input_563_pad_type_0"), val = string("valid")]; + tensor input_563_strides_0 = const()[name = string("input_563_strides_0"), val = tensor([1, 1])]; + tensor input_563_pad_0 = const()[name = string("input_563_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_563_dilations_0 = const()[name = string("input_563_dilations_0"), val = tensor([1, 1])]; + int32 input_563_groups_0 = const()[name = string("input_563_groups_0"), val = int32(1)]; + tensor input_563_cast_fp16 = conv(dilations = input_563_dilations_0, groups = input_563_groups_0, pad = input_563_pad_0, pad_type = input_563_pad_type_0, strides = input_563_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_561_cast_fp16)[name = string("input_563_cast_fp16")]; + tensor var_13198_cast_fp16 = silu(x = input_563_cast_fp16)[name = string("op_13198_cast_fp16")]; + string var_13204_pad_type_0 = const()[name = string("op_13204_pad_type_0"), val = string("valid")]; + tensor var_13204_strides_0 = const()[name = string("op_13204_strides_0"), val = tensor([1, 1])]; + tensor var_13204_pad_0 = const()[name = string("op_13204_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13204_dilations_0 = const()[name = string("op_13204_dilations_0"), val = tensor([1, 1])]; + int32 var_13204_groups_0 = const()[name = string("op_13204_groups_0"), val = int32(1)]; + tensor var_13204_cast_fp16 = conv(dilations = var_13204_dilations_0, groups = var_13204_groups_0, pad = var_13204_pad_0, pad_type = var_13204_pad_type_0, strides = var_13204_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_561_cast_fp16)[name = string("op_13204_cast_fp16")]; + tensor input_565_cast_fp16 = mul(x = var_13198_cast_fp16, y = var_13204_cast_fp16)[name = string("input_565_cast_fp16")]; + string h_105_pad_type_0 = const()[name = string("h_105_pad_type_0"), val = string("valid")]; + tensor h_105_strides_0 = const()[name = string("h_105_strides_0"), val = tensor([1, 1])]; + tensor h_105_pad_0 = const()[name = string("h_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_105_dilations_0 = const()[name = string("h_105_dilations_0"), val = tensor([1, 1])]; + int32 h_105_groups_0 = const()[name = string("h_105_groups_0"), val = int32(1)]; + tensor h_105_cast_fp16 = conv(dilations = h_105_dilations_0, groups = h_105_groups_0, pad = h_105_pad_0, pad_type = h_105_pad_type_0, strides = h_105_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_565_cast_fp16)[name = string("h_105_cast_fp16")]; + tensor x_403_cast_fp16 = add(x = x_401_cast_fp16, y = h_105_cast_fp16)[name = string("x_403_cast_fp16")]; + tensor key_cache_107_begin_0 = const()[name = string("key_cache_107_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_107_end_0 = const()[name = string("key_cache_107_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_107_end_mask_0 = const()[name = string("key_cache_107_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_107_cast_fp16 = slice_by_index(begin = key_cache_107_begin_0, end = key_cache_107_end_0, end_mask = key_cache_107_end_mask_0, x = layer_key_caches_21_cast_fp16)[name = string("key_cache_107_cast_fp16")]; + tensor value_cache_107_begin_0 = const()[name = string("value_cache_107_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_107_end_0 = const()[name = string("value_cache_107_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_107_end_mask_0 = const()[name = string("value_cache_107_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_107_cast_fp16 = slice_by_index(begin = value_cache_107_begin_0, end = value_cache_107_end_0, end_mask = value_cache_107_end_mask_0, x = layer_value_caches_21_cast_fp16)[name = string("value_cache_107_cast_fp16")]; + int32 var_13257 = const()[name = string("op_13257"), val = int32(2)]; + int32 var_13261 = const()[name = string("op_13261"), val = int32(3)]; + tensor var_13276_cast_fp16 = mul(x = x_403_cast_fp16, y = x_403_cast_fp16)[name = string("op_13276_cast_fp16")]; + tensor variance_443_axes_0 = const()[name = string("variance_443_axes_0"), val = tensor([1])]; + bool variance_443_keep_dims_0 = const()[name = string("variance_443_keep_dims_0"), val = bool(true)]; + tensor variance_443_cast_fp16 = reduce_mean(axes = variance_443_axes_0, keep_dims = variance_443_keep_dims_0, x = var_13276_cast_fp16)[name = string("variance_443_cast_fp16")]; + fp16 var_13279_to_fp16 = const()[name = string("op_13279_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13280_cast_fp16 = add(x = variance_443_cast_fp16, y = var_13279_to_fp16)[name = string("op_13280_cast_fp16")]; + fp32 var_13281_epsilon_0 = const()[name = string("op_13281_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13281_cast_fp16 = rsqrt(epsilon = var_13281_epsilon_0, x = var_13280_cast_fp16)[name = string("op_13281_cast_fp16")]; + tensor var_13282_cast_fp16 = mul(x = x_403_cast_fp16, y = var_13281_cast_fp16)[name = string("op_13282_cast_fp16")]; + tensor input_567_cast_fp16 = mul(x = var_13282_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_567_cast_fp16")]; + string q_319_pad_type_0 = const()[name = string("q_319_pad_type_0"), val = string("valid")]; + tensor q_319_strides_0 = const()[name = string("q_319_strides_0"), val = tensor([1, 1])]; + tensor q_319_pad_0 = const()[name = string("q_319_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_319_dilations_0 = const()[name = string("q_319_dilations_0"), val = tensor([1, 1])]; + int32 q_319_groups_0 = const()[name = string("q_319_groups_0"), val = int32(1)]; + tensor q_319_cast_fp16 = conv(dilations = q_319_dilations_0, groups = q_319_groups_0, pad = q_319_pad_0, pad_type = q_319_pad_type_0, strides = q_319_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_567_cast_fp16)[name = string("q_319_cast_fp16")]; + string k_319_pad_type_0 = const()[name = string("k_319_pad_type_0"), val = string("valid")]; + tensor k_319_strides_0 = const()[name = string("k_319_strides_0"), val = tensor([1, 1])]; + tensor k_319_pad_0 = const()[name = string("k_319_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_319_dilations_0 = const()[name = string("k_319_dilations_0"), val = tensor([1, 1])]; + int32 k_319_groups_0 = const()[name = string("k_319_groups_0"), val = int32(1)]; + tensor k_319_cast_fp16 = conv(dilations = k_319_dilations_0, groups = k_319_groups_0, pad = k_319_pad_0, pad_type = k_319_pad_type_0, strides = k_319_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_567_cast_fp16)[name = string("k_319_cast_fp16")]; + string v_107_pad_type_0 = const()[name = string("v_107_pad_type_0"), val = string("valid")]; + tensor v_107_strides_0 = const()[name = string("v_107_strides_0"), val = tensor([1, 1])]; + tensor v_107_pad_0 = const()[name = string("v_107_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_107_dilations_0 = const()[name = string("v_107_dilations_0"), val = tensor([1, 1])]; + int32 v_107_groups_0 = const()[name = string("v_107_groups_0"), val = int32(1)]; + tensor v_107_cast_fp16 = conv(dilations = v_107_dilations_0, groups = v_107_groups_0, pad = v_107_pad_0, pad_type = v_107_pad_type_0, strides = v_107_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_567_cast_fp16)[name = string("v_107_cast_fp16")]; + tensor var_13316 = const()[name = string("op_13316"), val = tensor([16, 128, 1, 1])]; + tensor x_405_cast_fp16 = reshape(shape = var_13316, x = q_319_cast_fp16)[name = string("x_405_cast_fp16")]; + tensor var_13319_cast_fp16 = mul(x = x_405_cast_fp16, y = x_405_cast_fp16)[name = string("op_13319_cast_fp16")]; + tensor variance_445_axes_0 = const()[name = string("variance_445_axes_0"), val = tensor([1])]; + bool variance_445_keep_dims_0 = const()[name = string("variance_445_keep_dims_0"), val = bool(true)]; + tensor variance_445_cast_fp16 = reduce_mean(axes = variance_445_axes_0, keep_dims = variance_445_keep_dims_0, x = var_13319_cast_fp16)[name = string("variance_445_cast_fp16")]; + fp16 var_13322_to_fp16 = const()[name = string("op_13322_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13323_cast_fp16 = add(x = variance_445_cast_fp16, y = var_13322_to_fp16)[name = string("op_13323_cast_fp16")]; + fp32 var_13324_epsilon_0 = const()[name = string("op_13324_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13324_cast_fp16 = rsqrt(epsilon = var_13324_epsilon_0, x = var_13323_cast_fp16)[name = string("op_13324_cast_fp16")]; + tensor var_13325_cast_fp16 = mul(x = x_405_cast_fp16, y = var_13324_cast_fp16)[name = string("op_13325_cast_fp16")]; + tensor q_321_cast_fp16 = mul(x = var_13325_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_321_cast_fp16")]; + tensor var_13327 = const()[name = string("op_13327"), val = tensor([8, 128, 1, 1])]; + tensor x_407_cast_fp16 = reshape(shape = var_13327, x = k_319_cast_fp16)[name = string("x_407_cast_fp16")]; + tensor var_13330_cast_fp16 = mul(x = x_407_cast_fp16, y = x_407_cast_fp16)[name = string("op_13330_cast_fp16")]; + tensor variance_447_axes_0 = const()[name = string("variance_447_axes_0"), val = tensor([1])]; + bool variance_447_keep_dims_0 = const()[name = string("variance_447_keep_dims_0"), val = bool(true)]; + tensor variance_447_cast_fp16 = reduce_mean(axes = variance_447_axes_0, keep_dims = variance_447_keep_dims_0, x = var_13330_cast_fp16)[name = string("variance_447_cast_fp16")]; + fp16 var_13333_to_fp16 = const()[name = string("op_13333_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13334_cast_fp16 = add(x = variance_447_cast_fp16, y = var_13333_to_fp16)[name = string("op_13334_cast_fp16")]; + fp32 var_13335_epsilon_0 = const()[name = string("op_13335_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13335_cast_fp16 = rsqrt(epsilon = var_13335_epsilon_0, x = var_13334_cast_fp16)[name = string("op_13335_cast_fp16")]; + tensor var_13336_cast_fp16 = mul(x = x_407_cast_fp16, y = var_13335_cast_fp16)[name = string("op_13336_cast_fp16")]; + tensor k_321_cast_fp16 = mul(x = var_13336_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_321_cast_fp16")]; + tensor var_13338 = const()[name = string("op_13338"), val = tensor([1, 16, 128, 1])]; + tensor z_213_cast_fp16 = reshape(shape = var_13338, x = q_321_cast_fp16)[name = string("z_213_cast_fp16")]; + tensor var_13340 = const()[name = string("op_13340"), val = tensor([1, 8, 128, 1])]; + tensor z_215_cast_fp16 = reshape(shape = var_13340, x = k_321_cast_fp16)[name = string("z_215_cast_fp16")]; + tensor z1_213_begin_0 = const()[name = string("z1_213_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_213_end_0 = const()[name = string("z1_213_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_213_end_mask_0 = const()[name = string("z1_213_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_213_cast_fp16 = slice_by_index(begin = z1_213_begin_0, end = z1_213_end_0, end_mask = z1_213_end_mask_0, x = z_213_cast_fp16)[name = string("z1_213_cast_fp16")]; + tensor z2_213_begin_0 = const()[name = string("z2_213_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_213_end_0 = const()[name = string("z2_213_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_213_end_mask_0 = const()[name = string("z2_213_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_213_cast_fp16 = slice_by_index(begin = z2_213_begin_0, end = z2_213_end_0, end_mask = z2_213_end_mask_0, x = z_213_cast_fp16)[name = string("z2_213_cast_fp16")]; + tensor var_13348_cast_fp16 = mul(x = z_213_cast_fp16, y = cos_101_to_fp16)[name = string("op_13348_cast_fp16")]; + fp16 const_117_promoted_to_fp16 = const()[name = string("const_117_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13349_cast_fp16 = mul(x = z2_213_cast_fp16, y = const_117_promoted_to_fp16)[name = string("op_13349_cast_fp16")]; + bool var_13351_interleave_0 = const()[name = string("op_13351_interleave_0"), val = bool(false)]; + tensor var_13351_cast_fp16 = concat(axis = var_13257, interleave = var_13351_interleave_0, values = (var_13349_cast_fp16, z1_213_cast_fp16))[name = string("op_13351_cast_fp16")]; + tensor var_13352_cast_fp16 = mul(x = var_13351_cast_fp16, y = sin_101_to_fp16)[name = string("op_13352_cast_fp16")]; + tensor q_323_cast_fp16 = add(x = var_13348_cast_fp16, y = var_13352_cast_fp16)[name = string("q_323_cast_fp16")]; + tensor z1_215_begin_0 = const()[name = string("z1_215_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_215_end_0 = const()[name = string("z1_215_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_215_end_mask_0 = const()[name = string("z1_215_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_215_cast_fp16 = slice_by_index(begin = z1_215_begin_0, end = z1_215_end_0, end_mask = z1_215_end_mask_0, x = z_215_cast_fp16)[name = string("z1_215_cast_fp16")]; + tensor z2_215_begin_0 = const()[name = string("z2_215_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_215_end_0 = const()[name = string("z2_215_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_215_end_mask_0 = const()[name = string("z2_215_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_215_cast_fp16 = slice_by_index(begin = z2_215_begin_0, end = z2_215_end_0, end_mask = z2_215_end_mask_0, x = z_215_cast_fp16)[name = string("z2_215_cast_fp16")]; + tensor var_13360_cast_fp16 = mul(x = z_215_cast_fp16, y = cos_101_to_fp16)[name = string("op_13360_cast_fp16")]; + fp16 const_118_promoted_to_fp16 = const()[name = string("const_118_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13361_cast_fp16 = mul(x = z2_215_cast_fp16, y = const_118_promoted_to_fp16)[name = string("op_13361_cast_fp16")]; + bool var_13363_interleave_0 = const()[name = string("op_13363_interleave_0"), val = bool(false)]; + tensor var_13363_cast_fp16 = concat(axis = var_13257, interleave = var_13363_interleave_0, values = (var_13361_cast_fp16, z1_215_cast_fp16))[name = string("op_13363_cast_fp16")]; + tensor var_13364_cast_fp16 = mul(x = var_13363_cast_fp16, y = sin_101_to_fp16)[name = string("op_13364_cast_fp16")]; + tensor k_323_cast_fp16 = add(x = var_13360_cast_fp16, y = var_13364_cast_fp16)[name = string("k_323_cast_fp16")]; + tensor var_13366 = const()[name = string("op_13366"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_107_cast_fp16 = reshape(shape = var_13366, x = k_323_cast_fp16)[name = string("cur_key_107_cast_fp16")]; + tensor var_13368_to_fp16 = const()[name = string("op_13368_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642240)))]; + tensor var_13369_cast_fp16 = mul(x = key_cache_107_cast_fp16, y = var_13368_to_fp16)[name = string("op_13369_cast_fp16")]; + tensor upd_107_to_fp16 = const()[name = string("upd_107_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642368)))]; + tensor var_13370_cast_fp16 = mul(x = cur_key_107_cast_fp16, y = upd_107_to_fp16)[name = string("op_13370_cast_fp16")]; + tensor key_107_cast_fp16 = add(x = var_13369_cast_fp16, y = var_13370_cast_fp16)[name = string("key_107_cast_fp16")]; + tensor var_13372_to_fp16 = const()[name = string("op_13372_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642240)))]; + tensor var_13373_cast_fp16 = mul(x = value_cache_107_cast_fp16, y = var_13372_to_fp16)[name = string("op_13373_cast_fp16")]; + tensor var_13374_cast_fp16 = mul(x = v_107_cast_fp16, y = upd_107_to_fp16)[name = string("op_13374_cast_fp16")]; + tensor value_107_cast_fp16 = add(x = var_13373_cast_fp16, y = var_13374_cast_fp16)[name = string("value_107_cast_fp16")]; + tensor var_13376 = const()[name = string("op_13376"), val = tensor([1, 8, 128, 16])]; + tensor kh_213_cast_fp16 = reshape(shape = var_13376, x = key_107_cast_fp16)[name = string("kh_213_cast_fp16")]; + tensor var_13378 = const()[name = string("op_13378"), val = tensor([1, 8, 128, 16])]; + tensor vh_213_cast_fp16 = reshape(shape = var_13378, x = value_107_cast_fp16)[name = string("vh_213_cast_fp16")]; + tensor transpose_212_perm_0 = const()[name = string("transpose_212_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_106_reps_0 = const()[name = string("tile_106_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_212_cast_fp16 = transpose(perm = transpose_212_perm_0, x = kh_213_cast_fp16)[name = string("transpose_161")]; + tensor tile_106_cast_fp16 = tile(reps = tile_106_reps_0, x = transpose_212_cast_fp16)[name = string("tile_106_cast_fp16")]; + tensor concat_265 = const()[name = string("concat_265"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_212_cast_fp16 = reshape(shape = concat_265, x = tile_106_cast_fp16)[name = string("reshape_212_cast_fp16")]; + tensor transpose_213_perm_0 = const()[name = string("transpose_213_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_266 = const()[name = string("concat_266"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_213_cast_fp16 = transpose(perm = transpose_213_perm_0, x = reshape_212_cast_fp16)[name = string("transpose_160")]; + tensor reshape_213_cast_fp16 = reshape(shape = concat_266, x = transpose_213_cast_fp16)[name = string("reshape_213_cast_fp16")]; + tensor transpose_214_perm_0 = const()[name = string("transpose_214_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_107_reps_0 = const()[name = string("tile_107_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_214_cast_fp16 = transpose(perm = transpose_214_perm_0, x = vh_213_cast_fp16)[name = string("transpose_159")]; + tensor tile_107_cast_fp16 = tile(reps = tile_107_reps_0, x = transpose_214_cast_fp16)[name = string("tile_107_cast_fp16")]; + tensor concat_267 = const()[name = string("concat_267"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_214_cast_fp16 = reshape(shape = concat_267, x = tile_107_cast_fp16)[name = string("reshape_214_cast_fp16")]; + tensor transpose_215_perm_0 = const()[name = string("transpose_215_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_268 = const()[name = string("concat_268"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_215_cast_fp16 = transpose(perm = transpose_215_perm_0, x = reshape_214_cast_fp16)[name = string("transpose_158")]; + tensor reshape_215_cast_fp16 = reshape(shape = concat_268, x = transpose_215_cast_fp16)[name = string("reshape_215_cast_fp16")]; + fp16 var_13382_to_fp16 = const()[name = string("op_13382_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_13383_cast_fp16 = mul(x = q_323_cast_fp16, y = var_13382_to_fp16)[name = string("op_13383_cast_fp16")]; + tensor transpose_529_perm_0 = const()[name = string("transpose_529_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_231_transpose_x_1 = const()[name = string("w_231_transpose_x_1"), val = bool(true)]; + bool w_231_transpose_y_1 = const()[name = string("w_231_transpose_y_1"), val = bool(false)]; + tensor transpose_529_cast_fp16 = transpose(perm = transpose_529_perm_0, x = reshape_213_cast_fp16)[name = string("transpose_157")]; + tensor w_231_cast_fp16 = matmul(transpose_x = w_231_transpose_x_1, transpose_y = w_231_transpose_y_1, x = var_13383_cast_fp16, y = transpose_529_cast_fp16)[name = string("w_231_cast_fp16")]; + tensor pad_107_to_fp16 = const()[name = string("pad_107_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642496)))]; + tensor var_13386_cast_fp16 = add(x = w_231_cast_fp16, y = pad_107_to_fp16)[name = string("op_13386_cast_fp16")]; + tensor w_233_cast_fp16 = softmax(axis = var_13261, x = var_13386_cast_fp16)[name = string("w_233_cast_fp16")]; + tensor transpose_530_perm_0 = const()[name = string("transpose_530_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_107_transpose_x_1 = const()[name = string("attn_107_transpose_x_1"), val = bool(false)]; + bool attn_107_transpose_y_1 = const()[name = string("attn_107_transpose_y_1"), val = bool(true)]; + tensor transpose_530_cast_fp16 = transpose(perm = transpose_530_perm_0, x = reshape_215_cast_fp16)[name = string("transpose_156")]; + tensor attn_107_cast_fp16 = matmul(transpose_x = attn_107_transpose_x_1, transpose_y = attn_107_transpose_y_1, x = transpose_530_cast_fp16, y = w_233_cast_fp16)[name = string("attn_107_cast_fp16")]; + tensor var_13390 = const()[name = string("op_13390"), val = tensor([1, 2048, 1, 1])]; + tensor input_569_cast_fp16 = reshape(shape = var_13390, x = attn_107_cast_fp16)[name = string("input_569_cast_fp16")]; + string attn_output_107_pad_type_0 = const()[name = string("attn_output_107_pad_type_0"), val = string("valid")]; + tensor attn_output_107_strides_0 = const()[name = string("attn_output_107_strides_0"), val = tensor([1, 1])]; + tensor attn_output_107_pad_0 = const()[name = string("attn_output_107_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_107_dilations_0 = const()[name = string("attn_output_107_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_107_groups_0 = const()[name = string("attn_output_107_groups_0"), val = int32(1)]; + tensor attn_output_107_cast_fp16 = conv(dilations = attn_output_107_dilations_0, groups = attn_output_107_groups_0, pad = attn_output_107_pad_0, pad_type = attn_output_107_pad_type_0, strides = attn_output_107_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_569_cast_fp16)[name = string("attn_output_107_cast_fp16")]; + tensor x_409_cast_fp16 = add(x = x_403_cast_fp16, y = attn_output_107_cast_fp16)[name = string("x_409_cast_fp16")]; + tensor var_13404_cast_fp16 = mul(x = x_409_cast_fp16, y = x_409_cast_fp16)[name = string("op_13404_cast_fp16")]; + tensor variance_449_axes_0 = const()[name = string("variance_449_axes_0"), val = tensor([1])]; + bool variance_449_keep_dims_0 = const()[name = string("variance_449_keep_dims_0"), val = bool(true)]; + tensor variance_449_cast_fp16 = reduce_mean(axes = variance_449_axes_0, keep_dims = variance_449_keep_dims_0, x = var_13404_cast_fp16)[name = string("variance_449_cast_fp16")]; + fp16 var_13407_to_fp16 = const()[name = string("op_13407_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13408_cast_fp16 = add(x = variance_449_cast_fp16, y = var_13407_to_fp16)[name = string("op_13408_cast_fp16")]; + fp32 var_13409_epsilon_0 = const()[name = string("op_13409_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13409_cast_fp16 = rsqrt(epsilon = var_13409_epsilon_0, x = var_13408_cast_fp16)[name = string("op_13409_cast_fp16")]; + tensor var_13410_cast_fp16 = mul(x = x_409_cast_fp16, y = var_13409_cast_fp16)[name = string("op_13410_cast_fp16")]; + tensor input_571_cast_fp16 = mul(x = var_13410_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_571_cast_fp16")]; + string input_573_pad_type_0 = const()[name = string("input_573_pad_type_0"), val = string("valid")]; + tensor input_573_strides_0 = const()[name = string("input_573_strides_0"), val = tensor([1, 1])]; + tensor input_573_pad_0 = const()[name = string("input_573_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_573_dilations_0 = const()[name = string("input_573_dilations_0"), val = tensor([1, 1])]; + int32 input_573_groups_0 = const()[name = string("input_573_groups_0"), val = int32(1)]; + tensor input_573_cast_fp16 = conv(dilations = input_573_dilations_0, groups = input_573_groups_0, pad = input_573_pad_0, pad_type = input_573_pad_type_0, strides = input_573_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_571_cast_fp16)[name = string("input_573_cast_fp16")]; + tensor var_13418_cast_fp16 = silu(x = input_573_cast_fp16)[name = string("op_13418_cast_fp16")]; + string var_13424_pad_type_0 = const()[name = string("op_13424_pad_type_0"), val = string("valid")]; + tensor var_13424_strides_0 = const()[name = string("op_13424_strides_0"), val = tensor([1, 1])]; + tensor var_13424_pad_0 = const()[name = string("op_13424_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13424_dilations_0 = const()[name = string("op_13424_dilations_0"), val = tensor([1, 1])]; + int32 var_13424_groups_0 = const()[name = string("op_13424_groups_0"), val = int32(1)]; + tensor var_13424_cast_fp16 = conv(dilations = var_13424_dilations_0, groups = var_13424_groups_0, pad = var_13424_pad_0, pad_type = var_13424_pad_type_0, strides = var_13424_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_571_cast_fp16)[name = string("op_13424_cast_fp16")]; + tensor input_575_cast_fp16 = mul(x = var_13418_cast_fp16, y = var_13424_cast_fp16)[name = string("input_575_cast_fp16")]; + string h_107_pad_type_0 = const()[name = string("h_107_pad_type_0"), val = string("valid")]; + tensor h_107_strides_0 = const()[name = string("h_107_strides_0"), val = tensor([1, 1])]; + tensor h_107_pad_0 = const()[name = string("h_107_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_107_dilations_0 = const()[name = string("h_107_dilations_0"), val = tensor([1, 1])]; + int32 h_107_groups_0 = const()[name = string("h_107_groups_0"), val = int32(1)]; + tensor h_107_cast_fp16 = conv(dilations = h_107_dilations_0, groups = h_107_groups_0, pad = h_107_pad_0, pad_type = h_107_pad_type_0, strides = h_107_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_575_cast_fp16)[name = string("h_107_cast_fp16")]; + tensor x_411_cast_fp16 = add(x = x_409_cast_fp16, y = h_107_cast_fp16)[name = string("x_411_cast_fp16")]; + tensor key_cache_109_begin_0 = const()[name = string("key_cache_109_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_109_end_0 = const()[name = string("key_cache_109_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_109_end_mask_0 = const()[name = string("key_cache_109_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_109_cast_fp16 = slice_by_index(begin = key_cache_109_begin_0, end = key_cache_109_end_0, end_mask = key_cache_109_end_mask_0, x = layer_key_caches_21_cast_fp16)[name = string("key_cache_109_cast_fp16")]; + tensor value_cache_109_begin_0 = const()[name = string("value_cache_109_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_109_end_0 = const()[name = string("value_cache_109_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_109_end_mask_0 = const()[name = string("value_cache_109_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_109_cast_fp16 = slice_by_index(begin = value_cache_109_begin_0, end = value_cache_109_end_0, end_mask = value_cache_109_end_mask_0, x = layer_value_caches_21_cast_fp16)[name = string("value_cache_109_cast_fp16")]; + int32 var_13477 = const()[name = string("op_13477"), val = int32(2)]; + int32 var_13481 = const()[name = string("op_13481"), val = int32(3)]; + tensor var_13496_cast_fp16 = mul(x = x_411_cast_fp16, y = x_411_cast_fp16)[name = string("op_13496_cast_fp16")]; + tensor variance_451_axes_0 = const()[name = string("variance_451_axes_0"), val = tensor([1])]; + bool variance_451_keep_dims_0 = const()[name = string("variance_451_keep_dims_0"), val = bool(true)]; + tensor variance_451_cast_fp16 = reduce_mean(axes = variance_451_axes_0, keep_dims = variance_451_keep_dims_0, x = var_13496_cast_fp16)[name = string("variance_451_cast_fp16")]; + fp16 var_13499_to_fp16 = const()[name = string("op_13499_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13500_cast_fp16 = add(x = variance_451_cast_fp16, y = var_13499_to_fp16)[name = string("op_13500_cast_fp16")]; + fp32 var_13501_epsilon_0 = const()[name = string("op_13501_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13501_cast_fp16 = rsqrt(epsilon = var_13501_epsilon_0, x = var_13500_cast_fp16)[name = string("op_13501_cast_fp16")]; + tensor var_13502_cast_fp16 = mul(x = x_411_cast_fp16, y = var_13501_cast_fp16)[name = string("op_13502_cast_fp16")]; + tensor input_577_cast_fp16 = mul(x = var_13502_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_577_cast_fp16")]; + string q_325_pad_type_0 = const()[name = string("q_325_pad_type_0"), val = string("valid")]; + tensor q_325_strides_0 = const()[name = string("q_325_strides_0"), val = tensor([1, 1])]; + tensor q_325_pad_0 = const()[name = string("q_325_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_325_dilations_0 = const()[name = string("q_325_dilations_0"), val = tensor([1, 1])]; + int32 q_325_groups_0 = const()[name = string("q_325_groups_0"), val = int32(1)]; + tensor q_325_cast_fp16 = conv(dilations = q_325_dilations_0, groups = q_325_groups_0, pad = q_325_pad_0, pad_type = q_325_pad_type_0, strides = q_325_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_577_cast_fp16)[name = string("q_325_cast_fp16")]; + string k_325_pad_type_0 = const()[name = string("k_325_pad_type_0"), val = string("valid")]; + tensor k_325_strides_0 = const()[name = string("k_325_strides_0"), val = tensor([1, 1])]; + tensor k_325_pad_0 = const()[name = string("k_325_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_325_dilations_0 = const()[name = string("k_325_dilations_0"), val = tensor([1, 1])]; + int32 k_325_groups_0 = const()[name = string("k_325_groups_0"), val = int32(1)]; + tensor k_325_cast_fp16 = conv(dilations = k_325_dilations_0, groups = k_325_groups_0, pad = k_325_pad_0, pad_type = k_325_pad_type_0, strides = k_325_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_577_cast_fp16)[name = string("k_325_cast_fp16")]; + string v_109_pad_type_0 = const()[name = string("v_109_pad_type_0"), val = string("valid")]; + tensor v_109_strides_0 = const()[name = string("v_109_strides_0"), val = tensor([1, 1])]; + tensor v_109_pad_0 = const()[name = string("v_109_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_109_dilations_0 = const()[name = string("v_109_dilations_0"), val = tensor([1, 1])]; + int32 v_109_groups_0 = const()[name = string("v_109_groups_0"), val = int32(1)]; + tensor v_109_cast_fp16 = conv(dilations = v_109_dilations_0, groups = v_109_groups_0, pad = v_109_pad_0, pad_type = v_109_pad_type_0, strides = v_109_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_577_cast_fp16)[name = string("v_109_cast_fp16")]; + tensor var_13536 = const()[name = string("op_13536"), val = tensor([16, 128, 1, 1])]; + tensor x_413_cast_fp16 = reshape(shape = var_13536, x = q_325_cast_fp16)[name = string("x_413_cast_fp16")]; + tensor var_13539_cast_fp16 = mul(x = x_413_cast_fp16, y = x_413_cast_fp16)[name = string("op_13539_cast_fp16")]; + tensor variance_453_axes_0 = const()[name = string("variance_453_axes_0"), val = tensor([1])]; + bool variance_453_keep_dims_0 = const()[name = string("variance_453_keep_dims_0"), val = bool(true)]; + tensor variance_453_cast_fp16 = reduce_mean(axes = variance_453_axes_0, keep_dims = variance_453_keep_dims_0, x = var_13539_cast_fp16)[name = string("variance_453_cast_fp16")]; + fp16 var_13542_to_fp16 = const()[name = string("op_13542_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13543_cast_fp16 = add(x = variance_453_cast_fp16, y = var_13542_to_fp16)[name = string("op_13543_cast_fp16")]; + fp32 var_13544_epsilon_0 = const()[name = string("op_13544_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13544_cast_fp16 = rsqrt(epsilon = var_13544_epsilon_0, x = var_13543_cast_fp16)[name = string("op_13544_cast_fp16")]; + tensor var_13545_cast_fp16 = mul(x = x_413_cast_fp16, y = var_13544_cast_fp16)[name = string("op_13545_cast_fp16")]; + tensor q_327_cast_fp16 = mul(x = var_13545_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_327_cast_fp16")]; + tensor var_13547 = const()[name = string("op_13547"), val = tensor([8, 128, 1, 1])]; + tensor x_415_cast_fp16 = reshape(shape = var_13547, x = k_325_cast_fp16)[name = string("x_415_cast_fp16")]; + tensor var_13550_cast_fp16 = mul(x = x_415_cast_fp16, y = x_415_cast_fp16)[name = string("op_13550_cast_fp16")]; + tensor variance_455_axes_0 = const()[name = string("variance_455_axes_0"), val = tensor([1])]; + bool variance_455_keep_dims_0 = const()[name = string("variance_455_keep_dims_0"), val = bool(true)]; + tensor variance_455_cast_fp16 = reduce_mean(axes = variance_455_axes_0, keep_dims = variance_455_keep_dims_0, x = var_13550_cast_fp16)[name = string("variance_455_cast_fp16")]; + fp16 var_13553_to_fp16 = const()[name = string("op_13553_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13554_cast_fp16 = add(x = variance_455_cast_fp16, y = var_13553_to_fp16)[name = string("op_13554_cast_fp16")]; + fp32 var_13555_epsilon_0 = const()[name = string("op_13555_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13555_cast_fp16 = rsqrt(epsilon = var_13555_epsilon_0, x = var_13554_cast_fp16)[name = string("op_13555_cast_fp16")]; + tensor var_13556_cast_fp16 = mul(x = x_415_cast_fp16, y = var_13555_cast_fp16)[name = string("op_13556_cast_fp16")]; + tensor k_327_cast_fp16 = mul(x = var_13556_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_327_cast_fp16")]; + tensor var_13558 = const()[name = string("op_13558"), val = tensor([1, 16, 128, 1])]; + tensor z_217_cast_fp16 = reshape(shape = var_13558, x = q_327_cast_fp16)[name = string("z_217_cast_fp16")]; + tensor var_13560 = const()[name = string("op_13560"), val = tensor([1, 8, 128, 1])]; + tensor z_219_cast_fp16 = reshape(shape = var_13560, x = k_327_cast_fp16)[name = string("z_219_cast_fp16")]; + tensor z1_217_begin_0 = const()[name = string("z1_217_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_217_end_0 = const()[name = string("z1_217_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_217_end_mask_0 = const()[name = string("z1_217_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_217_cast_fp16 = slice_by_index(begin = z1_217_begin_0, end = z1_217_end_0, end_mask = z1_217_end_mask_0, x = z_217_cast_fp16)[name = string("z1_217_cast_fp16")]; + tensor z2_217_begin_0 = const()[name = string("z2_217_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_217_end_0 = const()[name = string("z2_217_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_217_end_mask_0 = const()[name = string("z2_217_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_217_cast_fp16 = slice_by_index(begin = z2_217_begin_0, end = z2_217_end_0, end_mask = z2_217_end_mask_0, x = z_217_cast_fp16)[name = string("z2_217_cast_fp16")]; + tensor var_13568_cast_fp16 = mul(x = z_217_cast_fp16, y = cos_101_to_fp16)[name = string("op_13568_cast_fp16")]; + fp16 const_119_promoted_to_fp16 = const()[name = string("const_119_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13569_cast_fp16 = mul(x = z2_217_cast_fp16, y = const_119_promoted_to_fp16)[name = string("op_13569_cast_fp16")]; + bool var_13571_interleave_0 = const()[name = string("op_13571_interleave_0"), val = bool(false)]; + tensor var_13571_cast_fp16 = concat(axis = var_13477, interleave = var_13571_interleave_0, values = (var_13569_cast_fp16, z1_217_cast_fp16))[name = string("op_13571_cast_fp16")]; + tensor var_13572_cast_fp16 = mul(x = var_13571_cast_fp16, y = sin_101_to_fp16)[name = string("op_13572_cast_fp16")]; + tensor q_329_cast_fp16 = add(x = var_13568_cast_fp16, y = var_13572_cast_fp16)[name = string("q_329_cast_fp16")]; + tensor z1_219_begin_0 = const()[name = string("z1_219_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_219_end_0 = const()[name = string("z1_219_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_219_end_mask_0 = const()[name = string("z1_219_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_219_cast_fp16 = slice_by_index(begin = z1_219_begin_0, end = z1_219_end_0, end_mask = z1_219_end_mask_0, x = z_219_cast_fp16)[name = string("z1_219_cast_fp16")]; + tensor z2_219_begin_0 = const()[name = string("z2_219_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_219_end_0 = const()[name = string("z2_219_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_219_end_mask_0 = const()[name = string("z2_219_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_219_cast_fp16 = slice_by_index(begin = z2_219_begin_0, end = z2_219_end_0, end_mask = z2_219_end_mask_0, x = z_219_cast_fp16)[name = string("z2_219_cast_fp16")]; + tensor var_13580_cast_fp16 = mul(x = z_219_cast_fp16, y = cos_101_to_fp16)[name = string("op_13580_cast_fp16")]; + fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13581_cast_fp16 = mul(x = z2_219_cast_fp16, y = const_120_promoted_to_fp16)[name = string("op_13581_cast_fp16")]; + bool var_13583_interleave_0 = const()[name = string("op_13583_interleave_0"), val = bool(false)]; + tensor var_13583_cast_fp16 = concat(axis = var_13477, interleave = var_13583_interleave_0, values = (var_13581_cast_fp16, z1_219_cast_fp16))[name = string("op_13583_cast_fp16")]; + tensor var_13584_cast_fp16 = mul(x = var_13583_cast_fp16, y = sin_101_to_fp16)[name = string("op_13584_cast_fp16")]; + tensor k_329_cast_fp16 = add(x = var_13580_cast_fp16, y = var_13584_cast_fp16)[name = string("k_329_cast_fp16")]; + tensor var_13586 = const()[name = string("op_13586"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_109_cast_fp16 = reshape(shape = var_13586, x = k_329_cast_fp16)[name = string("cur_key_109_cast_fp16")]; + tensor var_13588_to_fp16 = const()[name = string("op_13588_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642240)))]; + tensor var_13589_cast_fp16 = mul(x = key_cache_109_cast_fp16, y = var_13588_to_fp16)[name = string("op_13589_cast_fp16")]; + tensor upd_109_to_fp16 = const()[name = string("upd_109_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642368)))]; + tensor var_13590_cast_fp16 = mul(x = cur_key_109_cast_fp16, y = upd_109_to_fp16)[name = string("op_13590_cast_fp16")]; + tensor key_109_cast_fp16 = add(x = var_13589_cast_fp16, y = var_13590_cast_fp16)[name = string("key_109_cast_fp16")]; + tensor var_13592_to_fp16 = const()[name = string("op_13592_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642240)))]; + tensor var_13593_cast_fp16 = mul(x = value_cache_109_cast_fp16, y = var_13592_to_fp16)[name = string("op_13593_cast_fp16")]; + tensor var_13594_cast_fp16 = mul(x = v_109_cast_fp16, y = upd_109_to_fp16)[name = string("op_13594_cast_fp16")]; + tensor value_109_cast_fp16 = add(x = var_13593_cast_fp16, y = var_13594_cast_fp16)[name = string("value_109_cast_fp16")]; + tensor var_13596 = const()[name = string("op_13596"), val = tensor([1, 8, 128, 16])]; + tensor kh_217_cast_fp16 = reshape(shape = var_13596, x = key_109_cast_fp16)[name = string("kh_217_cast_fp16")]; + tensor var_13598 = const()[name = string("op_13598"), val = tensor([1, 8, 128, 16])]; + tensor vh_217_cast_fp16 = reshape(shape = var_13598, x = value_109_cast_fp16)[name = string("vh_217_cast_fp16")]; + tensor transpose_216_perm_0 = const()[name = string("transpose_216_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_108_reps_0 = const()[name = string("tile_108_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_216_cast_fp16 = transpose(perm = transpose_216_perm_0, x = kh_217_cast_fp16)[name = string("transpose_155")]; + tensor tile_108_cast_fp16 = tile(reps = tile_108_reps_0, x = transpose_216_cast_fp16)[name = string("tile_108_cast_fp16")]; + tensor concat_269 = const()[name = string("concat_269"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_216_cast_fp16 = reshape(shape = concat_269, x = tile_108_cast_fp16)[name = string("reshape_216_cast_fp16")]; + tensor transpose_217_perm_0 = const()[name = string("transpose_217_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_270 = const()[name = string("concat_270"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_217_cast_fp16 = transpose(perm = transpose_217_perm_0, x = reshape_216_cast_fp16)[name = string("transpose_154")]; + tensor reshape_217_cast_fp16 = reshape(shape = concat_270, x = transpose_217_cast_fp16)[name = string("reshape_217_cast_fp16")]; + tensor transpose_218_perm_0 = const()[name = string("transpose_218_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_109_reps_0 = const()[name = string("tile_109_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_218_cast_fp16 = transpose(perm = transpose_218_perm_0, x = vh_217_cast_fp16)[name = string("transpose_153")]; + tensor tile_109_cast_fp16 = tile(reps = tile_109_reps_0, x = transpose_218_cast_fp16)[name = string("tile_109_cast_fp16")]; + tensor concat_271 = const()[name = string("concat_271"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_218_cast_fp16 = reshape(shape = concat_271, x = tile_109_cast_fp16)[name = string("reshape_218_cast_fp16")]; + tensor transpose_219_perm_0 = const()[name = string("transpose_219_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_272 = const()[name = string("concat_272"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_219_cast_fp16 = transpose(perm = transpose_219_perm_0, x = reshape_218_cast_fp16)[name = string("transpose_152")]; + tensor reshape_219_cast_fp16 = reshape(shape = concat_272, x = transpose_219_cast_fp16)[name = string("reshape_219_cast_fp16")]; + fp16 var_13602_to_fp16 = const()[name = string("op_13602_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_13603_cast_fp16 = mul(x = q_329_cast_fp16, y = var_13602_to_fp16)[name = string("op_13603_cast_fp16")]; + tensor transpose_533_perm_0 = const()[name = string("transpose_533_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_235_transpose_x_1 = const()[name = string("w_235_transpose_x_1"), val = bool(true)]; + bool w_235_transpose_y_1 = const()[name = string("w_235_transpose_y_1"), val = bool(false)]; + tensor transpose_533_cast_fp16 = transpose(perm = transpose_533_perm_0, x = reshape_217_cast_fp16)[name = string("transpose_151")]; + tensor w_235_cast_fp16 = matmul(transpose_x = w_235_transpose_x_1, transpose_y = w_235_transpose_y_1, x = var_13603_cast_fp16, y = transpose_533_cast_fp16)[name = string("w_235_cast_fp16")]; + tensor pad_109_to_fp16 = const()[name = string("pad_109_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642496)))]; + tensor var_13606_cast_fp16 = add(x = w_235_cast_fp16, y = pad_109_to_fp16)[name = string("op_13606_cast_fp16")]; + tensor w_237_cast_fp16 = softmax(axis = var_13481, x = var_13606_cast_fp16)[name = string("w_237_cast_fp16")]; + tensor transpose_534_perm_0 = const()[name = string("transpose_534_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_109_transpose_x_1 = const()[name = string("attn_109_transpose_x_1"), val = bool(false)]; + bool attn_109_transpose_y_1 = const()[name = string("attn_109_transpose_y_1"), val = bool(true)]; + tensor transpose_534_cast_fp16 = transpose(perm = transpose_534_perm_0, x = reshape_219_cast_fp16)[name = string("transpose_150")]; + tensor attn_109_cast_fp16 = matmul(transpose_x = attn_109_transpose_x_1, transpose_y = attn_109_transpose_y_1, x = transpose_534_cast_fp16, y = w_237_cast_fp16)[name = string("attn_109_cast_fp16")]; + tensor var_13610 = const()[name = string("op_13610"), val = tensor([1, 2048, 1, 1])]; + tensor input_579_cast_fp16 = reshape(shape = var_13610, x = attn_109_cast_fp16)[name = string("input_579_cast_fp16")]; + string attn_output_109_pad_type_0 = const()[name = string("attn_output_109_pad_type_0"), val = string("valid")]; + tensor attn_output_109_strides_0 = const()[name = string("attn_output_109_strides_0"), val = tensor([1, 1])]; + tensor attn_output_109_pad_0 = const()[name = string("attn_output_109_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_109_dilations_0 = const()[name = string("attn_output_109_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_109_groups_0 = const()[name = string("attn_output_109_groups_0"), val = int32(1)]; + tensor attn_output_109_cast_fp16 = conv(dilations = attn_output_109_dilations_0, groups = attn_output_109_groups_0, pad = attn_output_109_pad_0, pad_type = attn_output_109_pad_type_0, strides = attn_output_109_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_579_cast_fp16)[name = string("attn_output_109_cast_fp16")]; + tensor x_417_cast_fp16 = add(x = x_411_cast_fp16, y = attn_output_109_cast_fp16)[name = string("x_417_cast_fp16")]; + tensor var_13624_cast_fp16 = mul(x = x_417_cast_fp16, y = x_417_cast_fp16)[name = string("op_13624_cast_fp16")]; + tensor variance_457_axes_0 = const()[name = string("variance_457_axes_0"), val = tensor([1])]; + bool variance_457_keep_dims_0 = const()[name = string("variance_457_keep_dims_0"), val = bool(true)]; + tensor variance_457_cast_fp16 = reduce_mean(axes = variance_457_axes_0, keep_dims = variance_457_keep_dims_0, x = var_13624_cast_fp16)[name = string("variance_457_cast_fp16")]; + fp16 var_13627_to_fp16 = const()[name = string("op_13627_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13628_cast_fp16 = add(x = variance_457_cast_fp16, y = var_13627_to_fp16)[name = string("op_13628_cast_fp16")]; + fp32 var_13629_epsilon_0 = const()[name = string("op_13629_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13629_cast_fp16 = rsqrt(epsilon = var_13629_epsilon_0, x = var_13628_cast_fp16)[name = string("op_13629_cast_fp16")]; + tensor var_13630_cast_fp16 = mul(x = x_417_cast_fp16, y = var_13629_cast_fp16)[name = string("op_13630_cast_fp16")]; + tensor input_581_cast_fp16 = mul(x = var_13630_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_581_cast_fp16")]; + string input_583_pad_type_0 = const()[name = string("input_583_pad_type_0"), val = string("valid")]; + tensor input_583_strides_0 = const()[name = string("input_583_strides_0"), val = tensor([1, 1])]; + tensor input_583_pad_0 = const()[name = string("input_583_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_583_dilations_0 = const()[name = string("input_583_dilations_0"), val = tensor([1, 1])]; + int32 input_583_groups_0 = const()[name = string("input_583_groups_0"), val = int32(1)]; + tensor input_583_cast_fp16 = conv(dilations = input_583_dilations_0, groups = input_583_groups_0, pad = input_583_pad_0, pad_type = input_583_pad_type_0, strides = input_583_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_581_cast_fp16)[name = string("input_583_cast_fp16")]; + tensor var_13638_cast_fp16 = silu(x = input_583_cast_fp16)[name = string("op_13638_cast_fp16")]; + string var_13644_pad_type_0 = const()[name = string("op_13644_pad_type_0"), val = string("valid")]; + tensor var_13644_strides_0 = const()[name = string("op_13644_strides_0"), val = tensor([1, 1])]; + tensor var_13644_pad_0 = const()[name = string("op_13644_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13644_dilations_0 = const()[name = string("op_13644_dilations_0"), val = tensor([1, 1])]; + int32 var_13644_groups_0 = const()[name = string("op_13644_groups_0"), val = int32(1)]; + tensor var_13644_cast_fp16 = conv(dilations = var_13644_dilations_0, groups = var_13644_groups_0, pad = var_13644_pad_0, pad_type = var_13644_pad_type_0, strides = var_13644_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_581_cast_fp16)[name = string("op_13644_cast_fp16")]; + tensor input_585_cast_fp16 = mul(x = var_13638_cast_fp16, y = var_13644_cast_fp16)[name = string("input_585_cast_fp16")]; + string h_109_pad_type_0 = const()[name = string("h_109_pad_type_0"), val = string("valid")]; + tensor h_109_strides_0 = const()[name = string("h_109_strides_0"), val = tensor([1, 1])]; + tensor h_109_pad_0 = const()[name = string("h_109_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_109_dilations_0 = const()[name = string("h_109_dilations_0"), val = tensor([1, 1])]; + int32 h_109_groups_0 = const()[name = string("h_109_groups_0"), val = int32(1)]; + tensor h_109_cast_fp16 = conv(dilations = h_109_dilations_0, groups = h_109_groups_0, pad = h_109_pad_0, pad_type = h_109_pad_type_0, strides = h_109_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_585_cast_fp16)[name = string("h_109_cast_fp16")]; + tensor inputs_19_cast_fp16 = add(x = x_417_cast_fp16, y = h_109_cast_fp16)[name = string("inputs_19_cast_fp16")]; + int32 var_13672 = const()[name = string("op_13672"), val = int32(1)]; + bool layer_key_caches_23_interleave_0 = const()[name = string("layer_key_caches_23_interleave_0"), val = bool(false)]; + tensor layer_key_caches_23_cast_fp16 = concat(axis = var_13672, interleave = layer_key_caches_23_interleave_0, values = (key_101_cast_fp16, key_103_cast_fp16, key_105_cast_fp16, key_107_cast_fp16, key_109_cast_fp16))[name = string("layer_key_caches_23_cast_fp16")]; + int32 var_13675 = const()[name = string("op_13675"), val = int32(1)]; + bool layer_value_caches_23_interleave_0 = const()[name = string("layer_value_caches_23_interleave_0"), val = bool(false)]; + tensor layer_value_caches_23_cast_fp16 = concat(axis = var_13675, interleave = layer_value_caches_23_interleave_0, values = (value_101_cast_fp16, value_103_cast_fp16, value_105_cast_fp16, value_107_cast_fp16, value_109_cast_fp16))[name = string("layer_value_caches_23_cast_fp16")]; + tensor inputs_sq_19_cast_fp16 = mul(x = inputs_19_cast_fp16, y = inputs_19_cast_fp16)[name = string("inputs_sq_19_cast_fp16")]; + tensor variance_459_axes_0 = const()[name = string("variance_459_axes_0"), val = tensor([1])]; + bool variance_459_keep_dims_0 = const()[name = string("variance_459_keep_dims_0"), val = bool(true)]; + tensor variance_459_cast_fp16 = reduce_mean(axes = variance_459_axes_0, keep_dims = variance_459_keep_dims_0, x = inputs_sq_19_cast_fp16)[name = string("variance_459_cast_fp16")]; + fp16 var_13685_to_fp16 = const()[name = string("op_13685_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13686_cast_fp16 = add(x = variance_459_cast_fp16, y = var_13685_to_fp16)[name = string("op_13686_cast_fp16")]; + fp32 var_13687_epsilon_0 = const()[name = string("op_13687_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13687_cast_fp16 = rsqrt(epsilon = var_13687_epsilon_0, x = var_13686_cast_fp16)[name = string("op_13687_cast_fp16")]; + tensor hidden_states_19_cast_fp16 = mul(x = inputs_19_cast_fp16, y = var_13687_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; + tensor input_587_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_19_cast_fp16)[name = string("input_587_cast_fp16")]; + string logits_37_pad_type_0 = const()[name = string("logits_37_pad_type_0"), val = string("valid")]; + tensor logits_37_strides_0 = const()[name = string("logits_37_strides_0"), val = tensor([1, 1])]; + tensor logits_37_pad_0 = const()[name = string("logits_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_37_dilations_0 = const()[name = string("logits_37_dilations_0"), val = tensor([1, 1])]; + int32 logits_37_groups_0 = const()[name = string("logits_37_groups_0"), val = int32(1)]; + tensor lm_heads_9_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(97586816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(99684032))))[name = string("lm_heads_9_weight_to_fp16_palettized")]; + tensor logits_37_cast_fp16 = conv(dilations = logits_37_dilations_0, groups = logits_37_groups_0, pad = logits_37_pad_0, pad_type = logits_37_pad_type_0, strides = logits_37_strides_0, weight = lm_heads_9_weight_to_fp16_palettized, x = input_587_cast_fp16)[name = string("logits_37_cast_fp16")]; + tensor var_13705 = const()[name = string("op_13705"), val = tensor([1, 2048])]; + tensor logits_39_cast_fp16 = reshape(shape = var_13705, x = logits_37_cast_fp16)[name = string("logits_39_cast_fp16")]; + tensor scaled_logits_19_cast_fp16 = real_div(x = logits_39_cast_fp16, y = temperature)[name = string("scaled_logits_19_cast_fp16")]; + int32 var_13715 = const()[name = string("op_13715"), val = int32(100)]; + int32 top_values_19_axis_0 = const()[name = string("top_values_19_axis_0"), val = int32(1)]; + bool top_values_19_ascending_0 = const()[name = string("top_values_19_ascending_0"), val = bool(false)]; + bool top_values_19_sort_0 = const()[name = string("top_values_19_sort_0"), val = bool(true)]; + bool top_values_19_return_indices_0 = const()[name = string("top_values_19_return_indices_0"), val = bool(true)]; + string top_values_19_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_19_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_19_cast_fp16_cast_uint16_0, tensor top_values_19_cast_fp16_cast_uint16_1 = topk(ascending = top_values_19_ascending_0, axis = top_values_19_axis_0, k = var_13715, output_indices_dtype = top_values_19_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_19_return_indices_0, sort = top_values_19_sort_0, x = scaled_logits_19_cast_fp16)[name = string("top_values_19_cast_fp16_cast_uint16")]; + tensor var_13721_cast_fp16 = mul(x = top_values_19_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_13721_cast_fp16")]; + tensor var_13725_cast_fp16 = add(x = var_13721_cast_fp16, y = var_2438_cast_fp16)[name = string("op_13725_cast_fp16")]; + tensor reduce_min_9_axes_0 = const()[name = string("reduce_min_9_axes_0"), val = tensor([1])]; + bool reduce_min_9_keep_dims_0 = const()[name = string("reduce_min_9_keep_dims_0"), val = bool(true)]; + tensor reduce_min_9_cast_fp16 = reduce_min(axes = reduce_min_9_axes_0, keep_dims = reduce_min_9_keep_dims_0, x = var_13725_cast_fp16)[name = string("reduce_min_9_cast_fp16")]; + tensor var_13728_cast_fp16 = greater_equal(x = scaled_logits_19_cast_fp16, y = reduce_min_9_cast_fp16)[name = string("op_13728_cast_fp16")]; + fp16 var_13729_value_0_to_fp16 = const()[name = string("op_13729_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_13729_cast_fp16 = fill_like(ref_tensor = scaled_logits_19_cast_fp16, value = var_13729_value_0_to_fp16)[name = string("op_13729_cast_fp16")]; + tensor masked_logits_19_cast_fp16 = select(a = scaled_logits_19_cast_fp16, b = var_13729_cast_fp16, cond = var_13728_cast_fp16)[name = string("masked_logits_19_cast_fp16")]; + tensor var_13733_begin_0 = const()[name = string("op_13733_begin_0"), val = tensor([9, 0])]; + tensor var_13733_end_0 = const()[name = string("op_13733_end_0"), val = tensor([10, 2048])]; + tensor var_13733_end_mask_0 = const()[name = string("op_13733_end_mask_0"), val = tensor([false, true])]; + tensor var_13733_squeeze_mask_0 = const()[name = string("op_13733_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_13733_cast_fp16 = slice_by_index(begin = var_13733_begin_0, end = var_13733_end_0, end_mask = var_13733_end_mask_0, squeeze_mask = var_13733_squeeze_mask_0, x = gumbel)[name = string("op_13733_cast_fp16")]; + tensor var_13736 = const()[name = string("op_13736"), val = tensor([1, 2048])]; + tensor var_13737_cast_fp16 = reshape(shape = var_13736, x = var_13733_cast_fp16)[name = string("op_13737_cast_fp16")]; + tensor noisy_logits_19_cast_fp16 = add(x = masked_logits_19_cast_fp16, y = var_13737_cast_fp16)[name = string("noisy_logits_19_cast_fp16")]; + int32 code_19_axis_0 = const()[name = string("code_19_axis_0"), val = int32(1)]; + bool code_19_keep_dims_0 = const()[name = string("code_19_keep_dims_0"), val = bool(false)]; + string code_19_output_dtype_0 = const()[name = string("code_19_output_dtype_0"), val = string("int32")]; + tensor code_19_cast_fp16 = reduce_argmax(axis = code_19_axis_0, keep_dims = code_19_keep_dims_0, output_dtype = code_19_output_dtype_0, x = noisy_logits_19_cast_fp16)[name = string("code_19_cast_fp16")]; + int32 var_13748 = const()[name = string("op_13748"), val = int32(18432)]; + tensor input_589 = add(x = code_19_cast_fp16, y = var_13748)[name = string("input_589")]; + int32 code_embed_37_axis_0 = const()[name = string("code_embed_37_axis_0"), val = int32(0)]; + int32 code_embed_37_batch_dims_0 = const()[name = string("code_embed_37_batch_dims_0"), val = int32(0)]; + bool code_embed_37_validate_indices_0 = const()[name = string("code_embed_37_validate_indices_0"), val = bool(false)]; + string input_589_to_uint16_dtype_0 = const()[name = string("input_589_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_589_to_uint16 = cast(dtype = input_589_to_uint16_dtype_0, x = input_589)[name = string("cast_5")]; + tensor code_embed_37_cast_fp16_cast_uint16 = gather(axis = code_embed_37_axis_0, batch_dims = code_embed_37_batch_dims_0, indices = input_589_to_uint16, validate_indices = code_embed_37_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_37_cast_fp16_cast_uint16")]; + tensor var_13752 = const()[name = string("op_13752"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_39_cast_fp16 = reshape(shape = var_13752, x = code_embed_37_cast_fp16_cast_uint16)[name = string("code_embed_39_cast_fp16")]; + tensor embed_sum_21_cast_fp16 = add(x = embed_sum_19_cast_fp16, y = code_embed_39_cast_fp16)[name = string("embed_sum_21_cast_fp16")]; + tensor key_cache_111_begin_0 = const()[name = string("key_cache_111_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_111_end_0 = const()[name = string("key_cache_111_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_111_end_mask_0 = const()[name = string("key_cache_111_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_111_cast_fp16 = slice_by_index(begin = key_cache_111_begin_0, end = key_cache_111_end_0, end_mask = key_cache_111_end_mask_0, x = layer_key_caches_23_cast_fp16)[name = string("key_cache_111_cast_fp16")]; + tensor value_cache_111_begin_0 = const()[name = string("value_cache_111_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_111_end_0 = const()[name = string("value_cache_111_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_111_end_mask_0 = const()[name = string("value_cache_111_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_111_cast_fp16 = slice_by_index(begin = value_cache_111_begin_0, end = value_cache_111_end_0, end_mask = value_cache_111_end_mask_0, x = layer_value_caches_23_cast_fp16)[name = string("value_cache_111_cast_fp16")]; + int32 var_13851 = const()[name = string("op_13851"), val = int32(2)]; + int32 var_13855 = const()[name = string("op_13855"), val = int32(3)]; + tensor var_13870_cast_fp16 = mul(x = code_embed_39_cast_fp16, y = code_embed_39_cast_fp16)[name = string("op_13870_cast_fp16")]; + tensor variance_461_axes_0 = const()[name = string("variance_461_axes_0"), val = tensor([1])]; + bool variance_461_keep_dims_0 = const()[name = string("variance_461_keep_dims_0"), val = bool(true)]; + tensor variance_461_cast_fp16 = reduce_mean(axes = variance_461_axes_0, keep_dims = variance_461_keep_dims_0, x = var_13870_cast_fp16)[name = string("variance_461_cast_fp16")]; + fp16 var_13873_to_fp16 = const()[name = string("op_13873_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13874_cast_fp16 = add(x = variance_461_cast_fp16, y = var_13873_to_fp16)[name = string("op_13874_cast_fp16")]; + fp32 var_13875_epsilon_0 = const()[name = string("op_13875_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13875_cast_fp16 = rsqrt(epsilon = var_13875_epsilon_0, x = var_13874_cast_fp16)[name = string("op_13875_cast_fp16")]; + tensor var_13876_cast_fp16 = mul(x = code_embed_39_cast_fp16, y = var_13875_cast_fp16)[name = string("op_13876_cast_fp16")]; + tensor input_591_cast_fp16 = mul(x = var_13876_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_591_cast_fp16")]; + string q_331_pad_type_0 = const()[name = string("q_331_pad_type_0"), val = string("valid")]; + tensor q_331_strides_0 = const()[name = string("q_331_strides_0"), val = tensor([1, 1])]; + tensor q_331_pad_0 = const()[name = string("q_331_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_331_dilations_0 = const()[name = string("q_331_dilations_0"), val = tensor([1, 1])]; + int32 q_331_groups_0 = const()[name = string("q_331_groups_0"), val = int32(1)]; + tensor q_331_cast_fp16 = conv(dilations = q_331_dilations_0, groups = q_331_groups_0, pad = q_331_pad_0, pad_type = q_331_pad_type_0, strides = q_331_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_591_cast_fp16)[name = string("q_331_cast_fp16")]; + string k_331_pad_type_0 = const()[name = string("k_331_pad_type_0"), val = string("valid")]; + tensor k_331_strides_0 = const()[name = string("k_331_strides_0"), val = tensor([1, 1])]; + tensor k_331_pad_0 = const()[name = string("k_331_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_331_dilations_0 = const()[name = string("k_331_dilations_0"), val = tensor([1, 1])]; + int32 k_331_groups_0 = const()[name = string("k_331_groups_0"), val = int32(1)]; + tensor k_331_cast_fp16 = conv(dilations = k_331_dilations_0, groups = k_331_groups_0, pad = k_331_pad_0, pad_type = k_331_pad_type_0, strides = k_331_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_591_cast_fp16)[name = string("k_331_cast_fp16")]; + string v_111_pad_type_0 = const()[name = string("v_111_pad_type_0"), val = string("valid")]; + tensor v_111_strides_0 = const()[name = string("v_111_strides_0"), val = tensor([1, 1])]; + tensor v_111_pad_0 = const()[name = string("v_111_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_111_dilations_0 = const()[name = string("v_111_dilations_0"), val = tensor([1, 1])]; + int32 v_111_groups_0 = const()[name = string("v_111_groups_0"), val = int32(1)]; + tensor v_111_cast_fp16 = conv(dilations = v_111_dilations_0, groups = v_111_groups_0, pad = v_111_pad_0, pad_type = v_111_pad_type_0, strides = v_111_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_591_cast_fp16)[name = string("v_111_cast_fp16")]; + tensor var_13910 = const()[name = string("op_13910"), val = tensor([16, 128, 1, 1])]; + tensor x_419_cast_fp16 = reshape(shape = var_13910, x = q_331_cast_fp16)[name = string("x_419_cast_fp16")]; + tensor var_13913_cast_fp16 = mul(x = x_419_cast_fp16, y = x_419_cast_fp16)[name = string("op_13913_cast_fp16")]; + tensor variance_463_axes_0 = const()[name = string("variance_463_axes_0"), val = tensor([1])]; + bool variance_463_keep_dims_0 = const()[name = string("variance_463_keep_dims_0"), val = bool(true)]; + tensor variance_463_cast_fp16 = reduce_mean(axes = variance_463_axes_0, keep_dims = variance_463_keep_dims_0, x = var_13913_cast_fp16)[name = string("variance_463_cast_fp16")]; + fp16 var_13916_to_fp16 = const()[name = string("op_13916_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13917_cast_fp16 = add(x = variance_463_cast_fp16, y = var_13916_to_fp16)[name = string("op_13917_cast_fp16")]; + fp32 var_13918_epsilon_0 = const()[name = string("op_13918_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13918_cast_fp16 = rsqrt(epsilon = var_13918_epsilon_0, x = var_13917_cast_fp16)[name = string("op_13918_cast_fp16")]; + tensor var_13919_cast_fp16 = mul(x = x_419_cast_fp16, y = var_13918_cast_fp16)[name = string("op_13919_cast_fp16")]; + tensor q_333_cast_fp16 = mul(x = var_13919_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_333_cast_fp16")]; + tensor var_13921 = const()[name = string("op_13921"), val = tensor([8, 128, 1, 1])]; + tensor x_421_cast_fp16 = reshape(shape = var_13921, x = k_331_cast_fp16)[name = string("x_421_cast_fp16")]; + tensor var_13924_cast_fp16 = mul(x = x_421_cast_fp16, y = x_421_cast_fp16)[name = string("op_13924_cast_fp16")]; + tensor variance_465_axes_0 = const()[name = string("variance_465_axes_0"), val = tensor([1])]; + bool variance_465_keep_dims_0 = const()[name = string("variance_465_keep_dims_0"), val = bool(true)]; + tensor variance_465_cast_fp16 = reduce_mean(axes = variance_465_axes_0, keep_dims = variance_465_keep_dims_0, x = var_13924_cast_fp16)[name = string("variance_465_cast_fp16")]; + fp16 var_13927_to_fp16 = const()[name = string("op_13927_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13928_cast_fp16 = add(x = variance_465_cast_fp16, y = var_13927_to_fp16)[name = string("op_13928_cast_fp16")]; + fp32 var_13929_epsilon_0 = const()[name = string("op_13929_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13929_cast_fp16 = rsqrt(epsilon = var_13929_epsilon_0, x = var_13928_cast_fp16)[name = string("op_13929_cast_fp16")]; + tensor var_13930_cast_fp16 = mul(x = x_421_cast_fp16, y = var_13929_cast_fp16)[name = string("op_13930_cast_fp16")]; + tensor k_333_cast_fp16 = mul(x = var_13930_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_333_cast_fp16")]; + tensor var_13932 = const()[name = string("op_13932"), val = tensor([1, 16, 128, 1])]; + tensor z_221_cast_fp16 = reshape(shape = var_13932, x = q_333_cast_fp16)[name = string("z_221_cast_fp16")]; + tensor var_13934 = const()[name = string("op_13934"), val = tensor([1, 8, 128, 1])]; + tensor z_223_cast_fp16 = reshape(shape = var_13934, x = k_333_cast_fp16)[name = string("z_223_cast_fp16")]; + tensor z1_221_begin_0 = const()[name = string("z1_221_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_221_end_0 = const()[name = string("z1_221_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_221_end_mask_0 = const()[name = string("z1_221_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_221_cast_fp16 = slice_by_index(begin = z1_221_begin_0, end = z1_221_end_0, end_mask = z1_221_end_mask_0, x = z_221_cast_fp16)[name = string("z1_221_cast_fp16")]; + tensor z2_221_begin_0 = const()[name = string("z2_221_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_221_end_0 = const()[name = string("z2_221_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_221_end_mask_0 = const()[name = string("z2_221_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_221_cast_fp16 = slice_by_index(begin = z2_221_begin_0, end = z2_221_end_0, end_mask = z2_221_end_mask_0, x = z_221_cast_fp16)[name = string("z2_221_cast_fp16")]; + tensor cos_111_to_fp16 = const()[name = string("cos_111_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642624)))]; + tensor var_13942_cast_fp16 = mul(x = z_221_cast_fp16, y = cos_111_to_fp16)[name = string("op_13942_cast_fp16")]; + fp16 const_122_promoted_to_fp16 = const()[name = string("const_122_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13943_cast_fp16 = mul(x = z2_221_cast_fp16, y = const_122_promoted_to_fp16)[name = string("op_13943_cast_fp16")]; + bool var_13945_interleave_0 = const()[name = string("op_13945_interleave_0"), val = bool(false)]; + tensor var_13945_cast_fp16 = concat(axis = var_13851, interleave = var_13945_interleave_0, values = (var_13943_cast_fp16, z1_221_cast_fp16))[name = string("op_13945_cast_fp16")]; + tensor sin_111_to_fp16 = const()[name = string("sin_111_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141642944)))]; + tensor var_13946_cast_fp16 = mul(x = var_13945_cast_fp16, y = sin_111_to_fp16)[name = string("op_13946_cast_fp16")]; + tensor q_335_cast_fp16 = add(x = var_13942_cast_fp16, y = var_13946_cast_fp16)[name = string("q_335_cast_fp16")]; + tensor z1_223_begin_0 = const()[name = string("z1_223_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_223_end_0 = const()[name = string("z1_223_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_223_end_mask_0 = const()[name = string("z1_223_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_223_cast_fp16 = slice_by_index(begin = z1_223_begin_0, end = z1_223_end_0, end_mask = z1_223_end_mask_0, x = z_223_cast_fp16)[name = string("z1_223_cast_fp16")]; + tensor z2_223_begin_0 = const()[name = string("z2_223_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_223_end_0 = const()[name = string("z2_223_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_223_end_mask_0 = const()[name = string("z2_223_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_223_cast_fp16 = slice_by_index(begin = z2_223_begin_0, end = z2_223_end_0, end_mask = z2_223_end_mask_0, x = z_223_cast_fp16)[name = string("z2_223_cast_fp16")]; + tensor var_13954_cast_fp16 = mul(x = z_223_cast_fp16, y = cos_111_to_fp16)[name = string("op_13954_cast_fp16")]; + fp16 const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13955_cast_fp16 = mul(x = z2_223_cast_fp16, y = const_123_promoted_to_fp16)[name = string("op_13955_cast_fp16")]; + bool var_13957_interleave_0 = const()[name = string("op_13957_interleave_0"), val = bool(false)]; + tensor var_13957_cast_fp16 = concat(axis = var_13851, interleave = var_13957_interleave_0, values = (var_13955_cast_fp16, z1_223_cast_fp16))[name = string("op_13957_cast_fp16")]; + tensor var_13958_cast_fp16 = mul(x = var_13957_cast_fp16, y = sin_111_to_fp16)[name = string("op_13958_cast_fp16")]; + tensor k_335_cast_fp16 = add(x = var_13954_cast_fp16, y = var_13958_cast_fp16)[name = string("k_335_cast_fp16")]; + tensor var_13960 = const()[name = string("op_13960"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_111_cast_fp16 = reshape(shape = var_13960, x = k_335_cast_fp16)[name = string("cur_key_111_cast_fp16")]; + tensor var_13962_to_fp16 = const()[name = string("op_13962_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643264)))]; + tensor var_13963_cast_fp16 = mul(x = key_cache_111_cast_fp16, y = var_13962_to_fp16)[name = string("op_13963_cast_fp16")]; + tensor upd_111_to_fp16 = const()[name = string("upd_111_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643392)))]; + tensor var_13964_cast_fp16 = mul(x = cur_key_111_cast_fp16, y = upd_111_to_fp16)[name = string("op_13964_cast_fp16")]; + tensor key_111_cast_fp16 = add(x = var_13963_cast_fp16, y = var_13964_cast_fp16)[name = string("key_111_cast_fp16")]; + tensor var_13966_to_fp16 = const()[name = string("op_13966_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643264)))]; + tensor var_13967_cast_fp16 = mul(x = value_cache_111_cast_fp16, y = var_13966_to_fp16)[name = string("op_13967_cast_fp16")]; + tensor var_13968_cast_fp16 = mul(x = v_111_cast_fp16, y = upd_111_to_fp16)[name = string("op_13968_cast_fp16")]; + tensor value_111_cast_fp16 = add(x = var_13967_cast_fp16, y = var_13968_cast_fp16)[name = string("value_111_cast_fp16")]; + tensor var_13970 = const()[name = string("op_13970"), val = tensor([1, 8, 128, 16])]; + tensor kh_221_cast_fp16 = reshape(shape = var_13970, x = key_111_cast_fp16)[name = string("kh_221_cast_fp16")]; + tensor var_13972 = const()[name = string("op_13972"), val = tensor([1, 8, 128, 16])]; + tensor vh_221_cast_fp16 = reshape(shape = var_13972, x = value_111_cast_fp16)[name = string("vh_221_cast_fp16")]; + tensor transpose_220_perm_0 = const()[name = string("transpose_220_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_110_reps_0 = const()[name = string("tile_110_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_220_cast_fp16 = transpose(perm = transpose_220_perm_0, x = kh_221_cast_fp16)[name = string("transpose_149")]; + tensor tile_110_cast_fp16 = tile(reps = tile_110_reps_0, x = transpose_220_cast_fp16)[name = string("tile_110_cast_fp16")]; + tensor concat_278 = const()[name = string("concat_278"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_220_cast_fp16 = reshape(shape = concat_278, x = tile_110_cast_fp16)[name = string("reshape_220_cast_fp16")]; + tensor transpose_221_perm_0 = const()[name = string("transpose_221_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_279 = const()[name = string("concat_279"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_221_cast_fp16 = transpose(perm = transpose_221_perm_0, x = reshape_220_cast_fp16)[name = string("transpose_148")]; + tensor reshape_221_cast_fp16 = reshape(shape = concat_279, x = transpose_221_cast_fp16)[name = string("reshape_221_cast_fp16")]; + tensor transpose_222_perm_0 = const()[name = string("transpose_222_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_111_reps_0 = const()[name = string("tile_111_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_222_cast_fp16 = transpose(perm = transpose_222_perm_0, x = vh_221_cast_fp16)[name = string("transpose_147")]; + tensor tile_111_cast_fp16 = tile(reps = tile_111_reps_0, x = transpose_222_cast_fp16)[name = string("tile_111_cast_fp16")]; + tensor concat_280 = const()[name = string("concat_280"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_222_cast_fp16 = reshape(shape = concat_280, x = tile_111_cast_fp16)[name = string("reshape_222_cast_fp16")]; + tensor transpose_223_perm_0 = const()[name = string("transpose_223_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_281 = const()[name = string("concat_281"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_223_cast_fp16 = transpose(perm = transpose_223_perm_0, x = reshape_222_cast_fp16)[name = string("transpose_146")]; + tensor reshape_223_cast_fp16 = reshape(shape = concat_281, x = transpose_223_cast_fp16)[name = string("reshape_223_cast_fp16")]; + fp16 var_13976_to_fp16 = const()[name = string("op_13976_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_13977_cast_fp16 = mul(x = q_335_cast_fp16, y = var_13976_to_fp16)[name = string("op_13977_cast_fp16")]; + tensor transpose_537_perm_0 = const()[name = string("transpose_537_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_241_transpose_x_1 = const()[name = string("w_241_transpose_x_1"), val = bool(true)]; + bool w_241_transpose_y_1 = const()[name = string("w_241_transpose_y_1"), val = bool(false)]; + tensor transpose_537_cast_fp16 = transpose(perm = transpose_537_perm_0, x = reshape_221_cast_fp16)[name = string("transpose_145")]; + tensor w_241_cast_fp16 = matmul(transpose_x = w_241_transpose_x_1, transpose_y = w_241_transpose_y_1, x = var_13977_cast_fp16, y = transpose_537_cast_fp16)[name = string("w_241_cast_fp16")]; + tensor pad_111_to_fp16 = const()[name = string("pad_111_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643520)))]; + tensor var_13980_cast_fp16 = add(x = w_241_cast_fp16, y = pad_111_to_fp16)[name = string("op_13980_cast_fp16")]; + tensor w_243_cast_fp16 = softmax(axis = var_13855, x = var_13980_cast_fp16)[name = string("w_243_cast_fp16")]; + tensor transpose_538_perm_0 = const()[name = string("transpose_538_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_111_transpose_x_1 = const()[name = string("attn_111_transpose_x_1"), val = bool(false)]; + bool attn_111_transpose_y_1 = const()[name = string("attn_111_transpose_y_1"), val = bool(true)]; + tensor transpose_538_cast_fp16 = transpose(perm = transpose_538_perm_0, x = reshape_223_cast_fp16)[name = string("transpose_144")]; + tensor attn_111_cast_fp16 = matmul(transpose_x = attn_111_transpose_x_1, transpose_y = attn_111_transpose_y_1, x = transpose_538_cast_fp16, y = w_243_cast_fp16)[name = string("attn_111_cast_fp16")]; + tensor var_13984 = const()[name = string("op_13984"), val = tensor([1, 2048, 1, 1])]; + tensor input_593_cast_fp16 = reshape(shape = var_13984, x = attn_111_cast_fp16)[name = string("input_593_cast_fp16")]; + string attn_output_111_pad_type_0 = const()[name = string("attn_output_111_pad_type_0"), val = string("valid")]; + tensor attn_output_111_strides_0 = const()[name = string("attn_output_111_strides_0"), val = tensor([1, 1])]; + tensor attn_output_111_pad_0 = const()[name = string("attn_output_111_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_111_dilations_0 = const()[name = string("attn_output_111_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_111_groups_0 = const()[name = string("attn_output_111_groups_0"), val = int32(1)]; + tensor attn_output_111_cast_fp16 = conv(dilations = attn_output_111_dilations_0, groups = attn_output_111_groups_0, pad = attn_output_111_pad_0, pad_type = attn_output_111_pad_type_0, strides = attn_output_111_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_593_cast_fp16)[name = string("attn_output_111_cast_fp16")]; + tensor x_423_cast_fp16 = add(x = code_embed_39_cast_fp16, y = attn_output_111_cast_fp16)[name = string("x_423_cast_fp16")]; + tensor var_13998_cast_fp16 = mul(x = x_423_cast_fp16, y = x_423_cast_fp16)[name = string("op_13998_cast_fp16")]; + tensor variance_467_axes_0 = const()[name = string("variance_467_axes_0"), val = tensor([1])]; + bool variance_467_keep_dims_0 = const()[name = string("variance_467_keep_dims_0"), val = bool(true)]; + tensor variance_467_cast_fp16 = reduce_mean(axes = variance_467_axes_0, keep_dims = variance_467_keep_dims_0, x = var_13998_cast_fp16)[name = string("variance_467_cast_fp16")]; + fp16 var_14001_to_fp16 = const()[name = string("op_14001_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14002_cast_fp16 = add(x = variance_467_cast_fp16, y = var_14001_to_fp16)[name = string("op_14002_cast_fp16")]; + fp32 var_14003_epsilon_0 = const()[name = string("op_14003_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14003_cast_fp16 = rsqrt(epsilon = var_14003_epsilon_0, x = var_14002_cast_fp16)[name = string("op_14003_cast_fp16")]; + tensor var_14004_cast_fp16 = mul(x = x_423_cast_fp16, y = var_14003_cast_fp16)[name = string("op_14004_cast_fp16")]; + tensor input_595_cast_fp16 = mul(x = var_14004_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_595_cast_fp16")]; + string input_597_pad_type_0 = const()[name = string("input_597_pad_type_0"), val = string("valid")]; + tensor input_597_strides_0 = const()[name = string("input_597_strides_0"), val = tensor([1, 1])]; + tensor input_597_pad_0 = const()[name = string("input_597_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_597_dilations_0 = const()[name = string("input_597_dilations_0"), val = tensor([1, 1])]; + int32 input_597_groups_0 = const()[name = string("input_597_groups_0"), val = int32(1)]; + tensor input_597_cast_fp16 = conv(dilations = input_597_dilations_0, groups = input_597_groups_0, pad = input_597_pad_0, pad_type = input_597_pad_type_0, strides = input_597_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_595_cast_fp16)[name = string("input_597_cast_fp16")]; + tensor var_14012_cast_fp16 = silu(x = input_597_cast_fp16)[name = string("op_14012_cast_fp16")]; + string var_14018_pad_type_0 = const()[name = string("op_14018_pad_type_0"), val = string("valid")]; + tensor var_14018_strides_0 = const()[name = string("op_14018_strides_0"), val = tensor([1, 1])]; + tensor var_14018_pad_0 = const()[name = string("op_14018_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14018_dilations_0 = const()[name = string("op_14018_dilations_0"), val = tensor([1, 1])]; + int32 var_14018_groups_0 = const()[name = string("op_14018_groups_0"), val = int32(1)]; + tensor var_14018_cast_fp16 = conv(dilations = var_14018_dilations_0, groups = var_14018_groups_0, pad = var_14018_pad_0, pad_type = var_14018_pad_type_0, strides = var_14018_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_595_cast_fp16)[name = string("op_14018_cast_fp16")]; + tensor input_599_cast_fp16 = mul(x = var_14012_cast_fp16, y = var_14018_cast_fp16)[name = string("input_599_cast_fp16")]; + string h_111_pad_type_0 = const()[name = string("h_111_pad_type_0"), val = string("valid")]; + tensor h_111_strides_0 = const()[name = string("h_111_strides_0"), val = tensor([1, 1])]; + tensor h_111_pad_0 = const()[name = string("h_111_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_111_dilations_0 = const()[name = string("h_111_dilations_0"), val = tensor([1, 1])]; + int32 h_111_groups_0 = const()[name = string("h_111_groups_0"), val = int32(1)]; + tensor h_111_cast_fp16 = conv(dilations = h_111_dilations_0, groups = h_111_groups_0, pad = h_111_pad_0, pad_type = h_111_pad_type_0, strides = h_111_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_599_cast_fp16)[name = string("h_111_cast_fp16")]; + tensor x_425_cast_fp16 = add(x = x_423_cast_fp16, y = h_111_cast_fp16)[name = string("x_425_cast_fp16")]; + tensor key_cache_113_begin_0 = const()[name = string("key_cache_113_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_113_end_0 = const()[name = string("key_cache_113_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_113_end_mask_0 = const()[name = string("key_cache_113_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_113_cast_fp16 = slice_by_index(begin = key_cache_113_begin_0, end = key_cache_113_end_0, end_mask = key_cache_113_end_mask_0, x = layer_key_caches_23_cast_fp16)[name = string("key_cache_113_cast_fp16")]; + tensor value_cache_113_begin_0 = const()[name = string("value_cache_113_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_113_end_0 = const()[name = string("value_cache_113_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_113_end_mask_0 = const()[name = string("value_cache_113_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_113_cast_fp16 = slice_by_index(begin = value_cache_113_begin_0, end = value_cache_113_end_0, end_mask = value_cache_113_end_mask_0, x = layer_value_caches_23_cast_fp16)[name = string("value_cache_113_cast_fp16")]; + int32 var_14071 = const()[name = string("op_14071"), val = int32(2)]; + int32 var_14075 = const()[name = string("op_14075"), val = int32(3)]; + tensor var_14090_cast_fp16 = mul(x = x_425_cast_fp16, y = x_425_cast_fp16)[name = string("op_14090_cast_fp16")]; + tensor variance_469_axes_0 = const()[name = string("variance_469_axes_0"), val = tensor([1])]; + bool variance_469_keep_dims_0 = const()[name = string("variance_469_keep_dims_0"), val = bool(true)]; + tensor variance_469_cast_fp16 = reduce_mean(axes = variance_469_axes_0, keep_dims = variance_469_keep_dims_0, x = var_14090_cast_fp16)[name = string("variance_469_cast_fp16")]; + fp16 var_14093_to_fp16 = const()[name = string("op_14093_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14094_cast_fp16 = add(x = variance_469_cast_fp16, y = var_14093_to_fp16)[name = string("op_14094_cast_fp16")]; + fp32 var_14095_epsilon_0 = const()[name = string("op_14095_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14095_cast_fp16 = rsqrt(epsilon = var_14095_epsilon_0, x = var_14094_cast_fp16)[name = string("op_14095_cast_fp16")]; + tensor var_14096_cast_fp16 = mul(x = x_425_cast_fp16, y = var_14095_cast_fp16)[name = string("op_14096_cast_fp16")]; + tensor input_601_cast_fp16 = mul(x = var_14096_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_601_cast_fp16")]; + string q_337_pad_type_0 = const()[name = string("q_337_pad_type_0"), val = string("valid")]; + tensor q_337_strides_0 = const()[name = string("q_337_strides_0"), val = tensor([1, 1])]; + tensor q_337_pad_0 = const()[name = string("q_337_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_337_dilations_0 = const()[name = string("q_337_dilations_0"), val = tensor([1, 1])]; + int32 q_337_groups_0 = const()[name = string("q_337_groups_0"), val = int32(1)]; + tensor q_337_cast_fp16 = conv(dilations = q_337_dilations_0, groups = q_337_groups_0, pad = q_337_pad_0, pad_type = q_337_pad_type_0, strides = q_337_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_601_cast_fp16)[name = string("q_337_cast_fp16")]; + string k_337_pad_type_0 = const()[name = string("k_337_pad_type_0"), val = string("valid")]; + tensor k_337_strides_0 = const()[name = string("k_337_strides_0"), val = tensor([1, 1])]; + tensor k_337_pad_0 = const()[name = string("k_337_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_337_dilations_0 = const()[name = string("k_337_dilations_0"), val = tensor([1, 1])]; + int32 k_337_groups_0 = const()[name = string("k_337_groups_0"), val = int32(1)]; + tensor k_337_cast_fp16 = conv(dilations = k_337_dilations_0, groups = k_337_groups_0, pad = k_337_pad_0, pad_type = k_337_pad_type_0, strides = k_337_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_601_cast_fp16)[name = string("k_337_cast_fp16")]; + string v_113_pad_type_0 = const()[name = string("v_113_pad_type_0"), val = string("valid")]; + tensor v_113_strides_0 = const()[name = string("v_113_strides_0"), val = tensor([1, 1])]; + tensor v_113_pad_0 = const()[name = string("v_113_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_113_dilations_0 = const()[name = string("v_113_dilations_0"), val = tensor([1, 1])]; + int32 v_113_groups_0 = const()[name = string("v_113_groups_0"), val = int32(1)]; + tensor v_113_cast_fp16 = conv(dilations = v_113_dilations_0, groups = v_113_groups_0, pad = v_113_pad_0, pad_type = v_113_pad_type_0, strides = v_113_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_601_cast_fp16)[name = string("v_113_cast_fp16")]; + tensor var_14130 = const()[name = string("op_14130"), val = tensor([16, 128, 1, 1])]; + tensor x_427_cast_fp16 = reshape(shape = var_14130, x = q_337_cast_fp16)[name = string("x_427_cast_fp16")]; + tensor var_14133_cast_fp16 = mul(x = x_427_cast_fp16, y = x_427_cast_fp16)[name = string("op_14133_cast_fp16")]; + tensor variance_471_axes_0 = const()[name = string("variance_471_axes_0"), val = tensor([1])]; + bool variance_471_keep_dims_0 = const()[name = string("variance_471_keep_dims_0"), val = bool(true)]; + tensor variance_471_cast_fp16 = reduce_mean(axes = variance_471_axes_0, keep_dims = variance_471_keep_dims_0, x = var_14133_cast_fp16)[name = string("variance_471_cast_fp16")]; + fp16 var_14136_to_fp16 = const()[name = string("op_14136_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14137_cast_fp16 = add(x = variance_471_cast_fp16, y = var_14136_to_fp16)[name = string("op_14137_cast_fp16")]; + fp32 var_14138_epsilon_0 = const()[name = string("op_14138_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14138_cast_fp16 = rsqrt(epsilon = var_14138_epsilon_0, x = var_14137_cast_fp16)[name = string("op_14138_cast_fp16")]; + tensor var_14139_cast_fp16 = mul(x = x_427_cast_fp16, y = var_14138_cast_fp16)[name = string("op_14139_cast_fp16")]; + tensor q_339_cast_fp16 = mul(x = var_14139_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_339_cast_fp16")]; + tensor var_14141 = const()[name = string("op_14141"), val = tensor([8, 128, 1, 1])]; + tensor x_429_cast_fp16 = reshape(shape = var_14141, x = k_337_cast_fp16)[name = string("x_429_cast_fp16")]; + tensor var_14144_cast_fp16 = mul(x = x_429_cast_fp16, y = x_429_cast_fp16)[name = string("op_14144_cast_fp16")]; + tensor variance_473_axes_0 = const()[name = string("variance_473_axes_0"), val = tensor([1])]; + bool variance_473_keep_dims_0 = const()[name = string("variance_473_keep_dims_0"), val = bool(true)]; + tensor variance_473_cast_fp16 = reduce_mean(axes = variance_473_axes_0, keep_dims = variance_473_keep_dims_0, x = var_14144_cast_fp16)[name = string("variance_473_cast_fp16")]; + fp16 var_14147_to_fp16 = const()[name = string("op_14147_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14148_cast_fp16 = add(x = variance_473_cast_fp16, y = var_14147_to_fp16)[name = string("op_14148_cast_fp16")]; + fp32 var_14149_epsilon_0 = const()[name = string("op_14149_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14149_cast_fp16 = rsqrt(epsilon = var_14149_epsilon_0, x = var_14148_cast_fp16)[name = string("op_14149_cast_fp16")]; + tensor var_14150_cast_fp16 = mul(x = x_429_cast_fp16, y = var_14149_cast_fp16)[name = string("op_14150_cast_fp16")]; + tensor k_339_cast_fp16 = mul(x = var_14150_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_339_cast_fp16")]; + tensor var_14152 = const()[name = string("op_14152"), val = tensor([1, 16, 128, 1])]; + tensor z_225_cast_fp16 = reshape(shape = var_14152, x = q_339_cast_fp16)[name = string("z_225_cast_fp16")]; + tensor var_14154 = const()[name = string("op_14154"), val = tensor([1, 8, 128, 1])]; + tensor z_227_cast_fp16 = reshape(shape = var_14154, x = k_339_cast_fp16)[name = string("z_227_cast_fp16")]; + tensor z1_225_begin_0 = const()[name = string("z1_225_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_225_end_0 = const()[name = string("z1_225_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_225_end_mask_0 = const()[name = string("z1_225_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_225_cast_fp16 = slice_by_index(begin = z1_225_begin_0, end = z1_225_end_0, end_mask = z1_225_end_mask_0, x = z_225_cast_fp16)[name = string("z1_225_cast_fp16")]; + tensor z2_225_begin_0 = const()[name = string("z2_225_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_225_end_0 = const()[name = string("z2_225_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_225_end_mask_0 = const()[name = string("z2_225_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_225_cast_fp16 = slice_by_index(begin = z2_225_begin_0, end = z2_225_end_0, end_mask = z2_225_end_mask_0, x = z_225_cast_fp16)[name = string("z2_225_cast_fp16")]; + tensor var_14162_cast_fp16 = mul(x = z_225_cast_fp16, y = cos_111_to_fp16)[name = string("op_14162_cast_fp16")]; + fp16 const_124_promoted_to_fp16 = const()[name = string("const_124_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14163_cast_fp16 = mul(x = z2_225_cast_fp16, y = const_124_promoted_to_fp16)[name = string("op_14163_cast_fp16")]; + bool var_14165_interleave_0 = const()[name = string("op_14165_interleave_0"), val = bool(false)]; + tensor var_14165_cast_fp16 = concat(axis = var_14071, interleave = var_14165_interleave_0, values = (var_14163_cast_fp16, z1_225_cast_fp16))[name = string("op_14165_cast_fp16")]; + tensor var_14166_cast_fp16 = mul(x = var_14165_cast_fp16, y = sin_111_to_fp16)[name = string("op_14166_cast_fp16")]; + tensor q_341_cast_fp16 = add(x = var_14162_cast_fp16, y = var_14166_cast_fp16)[name = string("q_341_cast_fp16")]; + tensor z1_227_begin_0 = const()[name = string("z1_227_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_227_end_0 = const()[name = string("z1_227_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_227_end_mask_0 = const()[name = string("z1_227_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_227_cast_fp16 = slice_by_index(begin = z1_227_begin_0, end = z1_227_end_0, end_mask = z1_227_end_mask_0, x = z_227_cast_fp16)[name = string("z1_227_cast_fp16")]; + tensor z2_227_begin_0 = const()[name = string("z2_227_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_227_end_0 = const()[name = string("z2_227_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_227_end_mask_0 = const()[name = string("z2_227_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_227_cast_fp16 = slice_by_index(begin = z2_227_begin_0, end = z2_227_end_0, end_mask = z2_227_end_mask_0, x = z_227_cast_fp16)[name = string("z2_227_cast_fp16")]; + tensor var_14174_cast_fp16 = mul(x = z_227_cast_fp16, y = cos_111_to_fp16)[name = string("op_14174_cast_fp16")]; + fp16 const_125_promoted_to_fp16 = const()[name = string("const_125_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14175_cast_fp16 = mul(x = z2_227_cast_fp16, y = const_125_promoted_to_fp16)[name = string("op_14175_cast_fp16")]; + bool var_14177_interleave_0 = const()[name = string("op_14177_interleave_0"), val = bool(false)]; + tensor var_14177_cast_fp16 = concat(axis = var_14071, interleave = var_14177_interleave_0, values = (var_14175_cast_fp16, z1_227_cast_fp16))[name = string("op_14177_cast_fp16")]; + tensor var_14178_cast_fp16 = mul(x = var_14177_cast_fp16, y = sin_111_to_fp16)[name = string("op_14178_cast_fp16")]; + tensor k_341_cast_fp16 = add(x = var_14174_cast_fp16, y = var_14178_cast_fp16)[name = string("k_341_cast_fp16")]; + tensor var_14180 = const()[name = string("op_14180"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_113_cast_fp16 = reshape(shape = var_14180, x = k_341_cast_fp16)[name = string("cur_key_113_cast_fp16")]; + tensor var_14182_to_fp16 = const()[name = string("op_14182_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643264)))]; + tensor var_14183_cast_fp16 = mul(x = key_cache_113_cast_fp16, y = var_14182_to_fp16)[name = string("op_14183_cast_fp16")]; + tensor upd_113_to_fp16 = const()[name = string("upd_113_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643392)))]; + tensor var_14184_cast_fp16 = mul(x = cur_key_113_cast_fp16, y = upd_113_to_fp16)[name = string("op_14184_cast_fp16")]; + tensor key_113_cast_fp16 = add(x = var_14183_cast_fp16, y = var_14184_cast_fp16)[name = string("key_113_cast_fp16")]; + tensor var_14186_to_fp16 = const()[name = string("op_14186_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643264)))]; + tensor var_14187_cast_fp16 = mul(x = value_cache_113_cast_fp16, y = var_14186_to_fp16)[name = string("op_14187_cast_fp16")]; + tensor var_14188_cast_fp16 = mul(x = v_113_cast_fp16, y = upd_113_to_fp16)[name = string("op_14188_cast_fp16")]; + tensor value_113_cast_fp16 = add(x = var_14187_cast_fp16, y = var_14188_cast_fp16)[name = string("value_113_cast_fp16")]; + tensor var_14190 = const()[name = string("op_14190"), val = tensor([1, 8, 128, 16])]; + tensor kh_225_cast_fp16 = reshape(shape = var_14190, x = key_113_cast_fp16)[name = string("kh_225_cast_fp16")]; + tensor var_14192 = const()[name = string("op_14192"), val = tensor([1, 8, 128, 16])]; + tensor vh_225_cast_fp16 = reshape(shape = var_14192, x = value_113_cast_fp16)[name = string("vh_225_cast_fp16")]; + tensor transpose_224_perm_0 = const()[name = string("transpose_224_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_112_reps_0 = const()[name = string("tile_112_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_224_cast_fp16 = transpose(perm = transpose_224_perm_0, x = kh_225_cast_fp16)[name = string("transpose_143")]; + tensor tile_112_cast_fp16 = tile(reps = tile_112_reps_0, x = transpose_224_cast_fp16)[name = string("tile_112_cast_fp16")]; + tensor concat_282 = const()[name = string("concat_282"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_224_cast_fp16 = reshape(shape = concat_282, x = tile_112_cast_fp16)[name = string("reshape_224_cast_fp16")]; + tensor transpose_225_perm_0 = const()[name = string("transpose_225_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_283 = const()[name = string("concat_283"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_225_cast_fp16 = transpose(perm = transpose_225_perm_0, x = reshape_224_cast_fp16)[name = string("transpose_142")]; + tensor reshape_225_cast_fp16 = reshape(shape = concat_283, x = transpose_225_cast_fp16)[name = string("reshape_225_cast_fp16")]; + tensor transpose_226_perm_0 = const()[name = string("transpose_226_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_113_reps_0 = const()[name = string("tile_113_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_226_cast_fp16 = transpose(perm = transpose_226_perm_0, x = vh_225_cast_fp16)[name = string("transpose_141")]; + tensor tile_113_cast_fp16 = tile(reps = tile_113_reps_0, x = transpose_226_cast_fp16)[name = string("tile_113_cast_fp16")]; + tensor concat_284 = const()[name = string("concat_284"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_226_cast_fp16 = reshape(shape = concat_284, x = tile_113_cast_fp16)[name = string("reshape_226_cast_fp16")]; + tensor transpose_227_perm_0 = const()[name = string("transpose_227_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_285 = const()[name = string("concat_285"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_227_cast_fp16 = transpose(perm = transpose_227_perm_0, x = reshape_226_cast_fp16)[name = string("transpose_140")]; + tensor reshape_227_cast_fp16 = reshape(shape = concat_285, x = transpose_227_cast_fp16)[name = string("reshape_227_cast_fp16")]; + fp16 var_14196_to_fp16 = const()[name = string("op_14196_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_14197_cast_fp16 = mul(x = q_341_cast_fp16, y = var_14196_to_fp16)[name = string("op_14197_cast_fp16")]; + tensor transpose_541_perm_0 = const()[name = string("transpose_541_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_245_transpose_x_1 = const()[name = string("w_245_transpose_x_1"), val = bool(true)]; + bool w_245_transpose_y_1 = const()[name = string("w_245_transpose_y_1"), val = bool(false)]; + tensor transpose_541_cast_fp16 = transpose(perm = transpose_541_perm_0, x = reshape_225_cast_fp16)[name = string("transpose_139")]; + tensor w_245_cast_fp16 = matmul(transpose_x = w_245_transpose_x_1, transpose_y = w_245_transpose_y_1, x = var_14197_cast_fp16, y = transpose_541_cast_fp16)[name = string("w_245_cast_fp16")]; + tensor pad_113_to_fp16 = const()[name = string("pad_113_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643520)))]; + tensor var_14200_cast_fp16 = add(x = w_245_cast_fp16, y = pad_113_to_fp16)[name = string("op_14200_cast_fp16")]; + tensor w_247_cast_fp16 = softmax(axis = var_14075, x = var_14200_cast_fp16)[name = string("w_247_cast_fp16")]; + tensor transpose_542_perm_0 = const()[name = string("transpose_542_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_113_transpose_x_1 = const()[name = string("attn_113_transpose_x_1"), val = bool(false)]; + bool attn_113_transpose_y_1 = const()[name = string("attn_113_transpose_y_1"), val = bool(true)]; + tensor transpose_542_cast_fp16 = transpose(perm = transpose_542_perm_0, x = reshape_227_cast_fp16)[name = string("transpose_138")]; + tensor attn_113_cast_fp16 = matmul(transpose_x = attn_113_transpose_x_1, transpose_y = attn_113_transpose_y_1, x = transpose_542_cast_fp16, y = w_247_cast_fp16)[name = string("attn_113_cast_fp16")]; + tensor var_14204 = const()[name = string("op_14204"), val = tensor([1, 2048, 1, 1])]; + tensor input_603_cast_fp16 = reshape(shape = var_14204, x = attn_113_cast_fp16)[name = string("input_603_cast_fp16")]; + string attn_output_113_pad_type_0 = const()[name = string("attn_output_113_pad_type_0"), val = string("valid")]; + tensor attn_output_113_strides_0 = const()[name = string("attn_output_113_strides_0"), val = tensor([1, 1])]; + tensor attn_output_113_pad_0 = const()[name = string("attn_output_113_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_113_dilations_0 = const()[name = string("attn_output_113_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_113_groups_0 = const()[name = string("attn_output_113_groups_0"), val = int32(1)]; + tensor attn_output_113_cast_fp16 = conv(dilations = attn_output_113_dilations_0, groups = attn_output_113_groups_0, pad = attn_output_113_pad_0, pad_type = attn_output_113_pad_type_0, strides = attn_output_113_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_603_cast_fp16)[name = string("attn_output_113_cast_fp16")]; + tensor x_431_cast_fp16 = add(x = x_425_cast_fp16, y = attn_output_113_cast_fp16)[name = string("x_431_cast_fp16")]; + tensor var_14218_cast_fp16 = mul(x = x_431_cast_fp16, y = x_431_cast_fp16)[name = string("op_14218_cast_fp16")]; + tensor variance_475_axes_0 = const()[name = string("variance_475_axes_0"), val = tensor([1])]; + bool variance_475_keep_dims_0 = const()[name = string("variance_475_keep_dims_0"), val = bool(true)]; + tensor variance_475_cast_fp16 = reduce_mean(axes = variance_475_axes_0, keep_dims = variance_475_keep_dims_0, x = var_14218_cast_fp16)[name = string("variance_475_cast_fp16")]; + fp16 var_14221_to_fp16 = const()[name = string("op_14221_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14222_cast_fp16 = add(x = variance_475_cast_fp16, y = var_14221_to_fp16)[name = string("op_14222_cast_fp16")]; + fp32 var_14223_epsilon_0 = const()[name = string("op_14223_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14223_cast_fp16 = rsqrt(epsilon = var_14223_epsilon_0, x = var_14222_cast_fp16)[name = string("op_14223_cast_fp16")]; + tensor var_14224_cast_fp16 = mul(x = x_431_cast_fp16, y = var_14223_cast_fp16)[name = string("op_14224_cast_fp16")]; + tensor input_605_cast_fp16 = mul(x = var_14224_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_605_cast_fp16")]; + string input_607_pad_type_0 = const()[name = string("input_607_pad_type_0"), val = string("valid")]; + tensor input_607_strides_0 = const()[name = string("input_607_strides_0"), val = tensor([1, 1])]; + tensor input_607_pad_0 = const()[name = string("input_607_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_607_dilations_0 = const()[name = string("input_607_dilations_0"), val = tensor([1, 1])]; + int32 input_607_groups_0 = const()[name = string("input_607_groups_0"), val = int32(1)]; + tensor input_607_cast_fp16 = conv(dilations = input_607_dilations_0, groups = input_607_groups_0, pad = input_607_pad_0, pad_type = input_607_pad_type_0, strides = input_607_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_605_cast_fp16)[name = string("input_607_cast_fp16")]; + tensor var_14232_cast_fp16 = silu(x = input_607_cast_fp16)[name = string("op_14232_cast_fp16")]; + string var_14238_pad_type_0 = const()[name = string("op_14238_pad_type_0"), val = string("valid")]; + tensor var_14238_strides_0 = const()[name = string("op_14238_strides_0"), val = tensor([1, 1])]; + tensor var_14238_pad_0 = const()[name = string("op_14238_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14238_dilations_0 = const()[name = string("op_14238_dilations_0"), val = tensor([1, 1])]; + int32 var_14238_groups_0 = const()[name = string("op_14238_groups_0"), val = int32(1)]; + tensor var_14238_cast_fp16 = conv(dilations = var_14238_dilations_0, groups = var_14238_groups_0, pad = var_14238_pad_0, pad_type = var_14238_pad_type_0, strides = var_14238_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_605_cast_fp16)[name = string("op_14238_cast_fp16")]; + tensor input_609_cast_fp16 = mul(x = var_14232_cast_fp16, y = var_14238_cast_fp16)[name = string("input_609_cast_fp16")]; + string h_113_pad_type_0 = const()[name = string("h_113_pad_type_0"), val = string("valid")]; + tensor h_113_strides_0 = const()[name = string("h_113_strides_0"), val = tensor([1, 1])]; + tensor h_113_pad_0 = const()[name = string("h_113_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_113_dilations_0 = const()[name = string("h_113_dilations_0"), val = tensor([1, 1])]; + int32 h_113_groups_0 = const()[name = string("h_113_groups_0"), val = int32(1)]; + tensor h_113_cast_fp16 = conv(dilations = h_113_dilations_0, groups = h_113_groups_0, pad = h_113_pad_0, pad_type = h_113_pad_type_0, strides = h_113_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_609_cast_fp16)[name = string("h_113_cast_fp16")]; + tensor x_433_cast_fp16 = add(x = x_431_cast_fp16, y = h_113_cast_fp16)[name = string("x_433_cast_fp16")]; + tensor key_cache_115_begin_0 = const()[name = string("key_cache_115_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_115_end_0 = const()[name = string("key_cache_115_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_115_end_mask_0 = const()[name = string("key_cache_115_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_115_cast_fp16 = slice_by_index(begin = key_cache_115_begin_0, end = key_cache_115_end_0, end_mask = key_cache_115_end_mask_0, x = layer_key_caches_23_cast_fp16)[name = string("key_cache_115_cast_fp16")]; + tensor value_cache_115_begin_0 = const()[name = string("value_cache_115_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_115_end_0 = const()[name = string("value_cache_115_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_115_end_mask_0 = const()[name = string("value_cache_115_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_115_cast_fp16 = slice_by_index(begin = value_cache_115_begin_0, end = value_cache_115_end_0, end_mask = value_cache_115_end_mask_0, x = layer_value_caches_23_cast_fp16)[name = string("value_cache_115_cast_fp16")]; + int32 var_14291 = const()[name = string("op_14291"), val = int32(2)]; + int32 var_14295 = const()[name = string("op_14295"), val = int32(3)]; + tensor var_14310_cast_fp16 = mul(x = x_433_cast_fp16, y = x_433_cast_fp16)[name = string("op_14310_cast_fp16")]; + tensor variance_477_axes_0 = const()[name = string("variance_477_axes_0"), val = tensor([1])]; + bool variance_477_keep_dims_0 = const()[name = string("variance_477_keep_dims_0"), val = bool(true)]; + tensor variance_477_cast_fp16 = reduce_mean(axes = variance_477_axes_0, keep_dims = variance_477_keep_dims_0, x = var_14310_cast_fp16)[name = string("variance_477_cast_fp16")]; + fp16 var_14313_to_fp16 = const()[name = string("op_14313_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14314_cast_fp16 = add(x = variance_477_cast_fp16, y = var_14313_to_fp16)[name = string("op_14314_cast_fp16")]; + fp32 var_14315_epsilon_0 = const()[name = string("op_14315_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14315_cast_fp16 = rsqrt(epsilon = var_14315_epsilon_0, x = var_14314_cast_fp16)[name = string("op_14315_cast_fp16")]; + tensor var_14316_cast_fp16 = mul(x = x_433_cast_fp16, y = var_14315_cast_fp16)[name = string("op_14316_cast_fp16")]; + tensor input_611_cast_fp16 = mul(x = var_14316_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_611_cast_fp16")]; + string q_343_pad_type_0 = const()[name = string("q_343_pad_type_0"), val = string("valid")]; + tensor q_343_strides_0 = const()[name = string("q_343_strides_0"), val = tensor([1, 1])]; + tensor q_343_pad_0 = const()[name = string("q_343_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_343_dilations_0 = const()[name = string("q_343_dilations_0"), val = tensor([1, 1])]; + int32 q_343_groups_0 = const()[name = string("q_343_groups_0"), val = int32(1)]; + tensor q_343_cast_fp16 = conv(dilations = q_343_dilations_0, groups = q_343_groups_0, pad = q_343_pad_0, pad_type = q_343_pad_type_0, strides = q_343_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_611_cast_fp16)[name = string("q_343_cast_fp16")]; + string k_343_pad_type_0 = const()[name = string("k_343_pad_type_0"), val = string("valid")]; + tensor k_343_strides_0 = const()[name = string("k_343_strides_0"), val = tensor([1, 1])]; + tensor k_343_pad_0 = const()[name = string("k_343_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_343_dilations_0 = const()[name = string("k_343_dilations_0"), val = tensor([1, 1])]; + int32 k_343_groups_0 = const()[name = string("k_343_groups_0"), val = int32(1)]; + tensor k_343_cast_fp16 = conv(dilations = k_343_dilations_0, groups = k_343_groups_0, pad = k_343_pad_0, pad_type = k_343_pad_type_0, strides = k_343_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_611_cast_fp16)[name = string("k_343_cast_fp16")]; + string v_115_pad_type_0 = const()[name = string("v_115_pad_type_0"), val = string("valid")]; + tensor v_115_strides_0 = const()[name = string("v_115_strides_0"), val = tensor([1, 1])]; + tensor v_115_pad_0 = const()[name = string("v_115_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_115_dilations_0 = const()[name = string("v_115_dilations_0"), val = tensor([1, 1])]; + int32 v_115_groups_0 = const()[name = string("v_115_groups_0"), val = int32(1)]; + tensor v_115_cast_fp16 = conv(dilations = v_115_dilations_0, groups = v_115_groups_0, pad = v_115_pad_0, pad_type = v_115_pad_type_0, strides = v_115_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_611_cast_fp16)[name = string("v_115_cast_fp16")]; + tensor var_14350 = const()[name = string("op_14350"), val = tensor([16, 128, 1, 1])]; + tensor x_435_cast_fp16 = reshape(shape = var_14350, x = q_343_cast_fp16)[name = string("x_435_cast_fp16")]; + tensor var_14353_cast_fp16 = mul(x = x_435_cast_fp16, y = x_435_cast_fp16)[name = string("op_14353_cast_fp16")]; + tensor variance_479_axes_0 = const()[name = string("variance_479_axes_0"), val = tensor([1])]; + bool variance_479_keep_dims_0 = const()[name = string("variance_479_keep_dims_0"), val = bool(true)]; + tensor variance_479_cast_fp16 = reduce_mean(axes = variance_479_axes_0, keep_dims = variance_479_keep_dims_0, x = var_14353_cast_fp16)[name = string("variance_479_cast_fp16")]; + fp16 var_14356_to_fp16 = const()[name = string("op_14356_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14357_cast_fp16 = add(x = variance_479_cast_fp16, y = var_14356_to_fp16)[name = string("op_14357_cast_fp16")]; + fp32 var_14358_epsilon_0 = const()[name = string("op_14358_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14358_cast_fp16 = rsqrt(epsilon = var_14358_epsilon_0, x = var_14357_cast_fp16)[name = string("op_14358_cast_fp16")]; + tensor var_14359_cast_fp16 = mul(x = x_435_cast_fp16, y = var_14358_cast_fp16)[name = string("op_14359_cast_fp16")]; + tensor q_345_cast_fp16 = mul(x = var_14359_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_345_cast_fp16")]; + tensor var_14361 = const()[name = string("op_14361"), val = tensor([8, 128, 1, 1])]; + tensor x_437_cast_fp16 = reshape(shape = var_14361, x = k_343_cast_fp16)[name = string("x_437_cast_fp16")]; + tensor var_14364_cast_fp16 = mul(x = x_437_cast_fp16, y = x_437_cast_fp16)[name = string("op_14364_cast_fp16")]; + tensor variance_481_axes_0 = const()[name = string("variance_481_axes_0"), val = tensor([1])]; + bool variance_481_keep_dims_0 = const()[name = string("variance_481_keep_dims_0"), val = bool(true)]; + tensor variance_481_cast_fp16 = reduce_mean(axes = variance_481_axes_0, keep_dims = variance_481_keep_dims_0, x = var_14364_cast_fp16)[name = string("variance_481_cast_fp16")]; + fp16 var_14367_to_fp16 = const()[name = string("op_14367_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14368_cast_fp16 = add(x = variance_481_cast_fp16, y = var_14367_to_fp16)[name = string("op_14368_cast_fp16")]; + fp32 var_14369_epsilon_0 = const()[name = string("op_14369_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14369_cast_fp16 = rsqrt(epsilon = var_14369_epsilon_0, x = var_14368_cast_fp16)[name = string("op_14369_cast_fp16")]; + tensor var_14370_cast_fp16 = mul(x = x_437_cast_fp16, y = var_14369_cast_fp16)[name = string("op_14370_cast_fp16")]; + tensor k_345_cast_fp16 = mul(x = var_14370_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_345_cast_fp16")]; + tensor var_14372 = const()[name = string("op_14372"), val = tensor([1, 16, 128, 1])]; + tensor z_229_cast_fp16 = reshape(shape = var_14372, x = q_345_cast_fp16)[name = string("z_229_cast_fp16")]; + tensor var_14374 = const()[name = string("op_14374"), val = tensor([1, 8, 128, 1])]; + tensor z_231_cast_fp16 = reshape(shape = var_14374, x = k_345_cast_fp16)[name = string("z_231_cast_fp16")]; + tensor z1_229_begin_0 = const()[name = string("z1_229_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_229_end_0 = const()[name = string("z1_229_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_229_end_mask_0 = const()[name = string("z1_229_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_229_cast_fp16 = slice_by_index(begin = z1_229_begin_0, end = z1_229_end_0, end_mask = z1_229_end_mask_0, x = z_229_cast_fp16)[name = string("z1_229_cast_fp16")]; + tensor z2_229_begin_0 = const()[name = string("z2_229_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_229_end_0 = const()[name = string("z2_229_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_229_end_mask_0 = const()[name = string("z2_229_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_229_cast_fp16 = slice_by_index(begin = z2_229_begin_0, end = z2_229_end_0, end_mask = z2_229_end_mask_0, x = z_229_cast_fp16)[name = string("z2_229_cast_fp16")]; + tensor var_14382_cast_fp16 = mul(x = z_229_cast_fp16, y = cos_111_to_fp16)[name = string("op_14382_cast_fp16")]; + fp16 const_126_promoted_to_fp16 = const()[name = string("const_126_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14383_cast_fp16 = mul(x = z2_229_cast_fp16, y = const_126_promoted_to_fp16)[name = string("op_14383_cast_fp16")]; + bool var_14385_interleave_0 = const()[name = string("op_14385_interleave_0"), val = bool(false)]; + tensor var_14385_cast_fp16 = concat(axis = var_14291, interleave = var_14385_interleave_0, values = (var_14383_cast_fp16, z1_229_cast_fp16))[name = string("op_14385_cast_fp16")]; + tensor var_14386_cast_fp16 = mul(x = var_14385_cast_fp16, y = sin_111_to_fp16)[name = string("op_14386_cast_fp16")]; + tensor q_347_cast_fp16 = add(x = var_14382_cast_fp16, y = var_14386_cast_fp16)[name = string("q_347_cast_fp16")]; + tensor z1_231_begin_0 = const()[name = string("z1_231_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_231_end_0 = const()[name = string("z1_231_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_231_end_mask_0 = const()[name = string("z1_231_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_231_cast_fp16 = slice_by_index(begin = z1_231_begin_0, end = z1_231_end_0, end_mask = z1_231_end_mask_0, x = z_231_cast_fp16)[name = string("z1_231_cast_fp16")]; + tensor z2_231_begin_0 = const()[name = string("z2_231_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_231_end_0 = const()[name = string("z2_231_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_231_end_mask_0 = const()[name = string("z2_231_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_231_cast_fp16 = slice_by_index(begin = z2_231_begin_0, end = z2_231_end_0, end_mask = z2_231_end_mask_0, x = z_231_cast_fp16)[name = string("z2_231_cast_fp16")]; + tensor var_14394_cast_fp16 = mul(x = z_231_cast_fp16, y = cos_111_to_fp16)[name = string("op_14394_cast_fp16")]; + fp16 const_127_promoted_to_fp16 = const()[name = string("const_127_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14395_cast_fp16 = mul(x = z2_231_cast_fp16, y = const_127_promoted_to_fp16)[name = string("op_14395_cast_fp16")]; + bool var_14397_interleave_0 = const()[name = string("op_14397_interleave_0"), val = bool(false)]; + tensor var_14397_cast_fp16 = concat(axis = var_14291, interleave = var_14397_interleave_0, values = (var_14395_cast_fp16, z1_231_cast_fp16))[name = string("op_14397_cast_fp16")]; + tensor var_14398_cast_fp16 = mul(x = var_14397_cast_fp16, y = sin_111_to_fp16)[name = string("op_14398_cast_fp16")]; + tensor k_347_cast_fp16 = add(x = var_14394_cast_fp16, y = var_14398_cast_fp16)[name = string("k_347_cast_fp16")]; + tensor var_14400 = const()[name = string("op_14400"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_115_cast_fp16 = reshape(shape = var_14400, x = k_347_cast_fp16)[name = string("cur_key_115_cast_fp16")]; + tensor var_14402_to_fp16 = const()[name = string("op_14402_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643264)))]; + tensor var_14403_cast_fp16 = mul(x = key_cache_115_cast_fp16, y = var_14402_to_fp16)[name = string("op_14403_cast_fp16")]; + tensor upd_115_to_fp16 = const()[name = string("upd_115_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643392)))]; + tensor var_14404_cast_fp16 = mul(x = cur_key_115_cast_fp16, y = upd_115_to_fp16)[name = string("op_14404_cast_fp16")]; + tensor key_115_cast_fp16 = add(x = var_14403_cast_fp16, y = var_14404_cast_fp16)[name = string("key_115_cast_fp16")]; + tensor var_14406_to_fp16 = const()[name = string("op_14406_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643264)))]; + tensor var_14407_cast_fp16 = mul(x = value_cache_115_cast_fp16, y = var_14406_to_fp16)[name = string("op_14407_cast_fp16")]; + tensor var_14408_cast_fp16 = mul(x = v_115_cast_fp16, y = upd_115_to_fp16)[name = string("op_14408_cast_fp16")]; + tensor value_115_cast_fp16 = add(x = var_14407_cast_fp16, y = var_14408_cast_fp16)[name = string("value_115_cast_fp16")]; + tensor var_14410 = const()[name = string("op_14410"), val = tensor([1, 8, 128, 16])]; + tensor kh_229_cast_fp16 = reshape(shape = var_14410, x = key_115_cast_fp16)[name = string("kh_229_cast_fp16")]; + tensor var_14412 = const()[name = string("op_14412"), val = tensor([1, 8, 128, 16])]; + tensor vh_229_cast_fp16 = reshape(shape = var_14412, x = value_115_cast_fp16)[name = string("vh_229_cast_fp16")]; + tensor transpose_228_perm_0 = const()[name = string("transpose_228_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_114_reps_0 = const()[name = string("tile_114_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_228_cast_fp16 = transpose(perm = transpose_228_perm_0, x = kh_229_cast_fp16)[name = string("transpose_137")]; + tensor tile_114_cast_fp16 = tile(reps = tile_114_reps_0, x = transpose_228_cast_fp16)[name = string("tile_114_cast_fp16")]; + tensor concat_286 = const()[name = string("concat_286"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_228_cast_fp16 = reshape(shape = concat_286, x = tile_114_cast_fp16)[name = string("reshape_228_cast_fp16")]; + tensor transpose_229_perm_0 = const()[name = string("transpose_229_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_287 = const()[name = string("concat_287"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_229_cast_fp16 = transpose(perm = transpose_229_perm_0, x = reshape_228_cast_fp16)[name = string("transpose_136")]; + tensor reshape_229_cast_fp16 = reshape(shape = concat_287, x = transpose_229_cast_fp16)[name = string("reshape_229_cast_fp16")]; + tensor transpose_230_perm_0 = const()[name = string("transpose_230_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_115_reps_0 = const()[name = string("tile_115_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_230_cast_fp16 = transpose(perm = transpose_230_perm_0, x = vh_229_cast_fp16)[name = string("transpose_135")]; + tensor tile_115_cast_fp16 = tile(reps = tile_115_reps_0, x = transpose_230_cast_fp16)[name = string("tile_115_cast_fp16")]; + tensor concat_288 = const()[name = string("concat_288"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_230_cast_fp16 = reshape(shape = concat_288, x = tile_115_cast_fp16)[name = string("reshape_230_cast_fp16")]; + tensor transpose_231_perm_0 = const()[name = string("transpose_231_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_289 = const()[name = string("concat_289"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_231_cast_fp16 = transpose(perm = transpose_231_perm_0, x = reshape_230_cast_fp16)[name = string("transpose_134")]; + tensor reshape_231_cast_fp16 = reshape(shape = concat_289, x = transpose_231_cast_fp16)[name = string("reshape_231_cast_fp16")]; + fp16 var_14416_to_fp16 = const()[name = string("op_14416_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_14417_cast_fp16 = mul(x = q_347_cast_fp16, y = var_14416_to_fp16)[name = string("op_14417_cast_fp16")]; + tensor transpose_545_perm_0 = const()[name = string("transpose_545_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_249_transpose_x_1 = const()[name = string("w_249_transpose_x_1"), val = bool(true)]; + bool w_249_transpose_y_1 = const()[name = string("w_249_transpose_y_1"), val = bool(false)]; + tensor transpose_545_cast_fp16 = transpose(perm = transpose_545_perm_0, x = reshape_229_cast_fp16)[name = string("transpose_133")]; + tensor w_249_cast_fp16 = matmul(transpose_x = w_249_transpose_x_1, transpose_y = w_249_transpose_y_1, x = var_14417_cast_fp16, y = transpose_545_cast_fp16)[name = string("w_249_cast_fp16")]; + tensor pad_115_to_fp16 = const()[name = string("pad_115_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643520)))]; + tensor var_14420_cast_fp16 = add(x = w_249_cast_fp16, y = pad_115_to_fp16)[name = string("op_14420_cast_fp16")]; + tensor w_251_cast_fp16 = softmax(axis = var_14295, x = var_14420_cast_fp16)[name = string("w_251_cast_fp16")]; + tensor transpose_546_perm_0 = const()[name = string("transpose_546_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_115_transpose_x_1 = const()[name = string("attn_115_transpose_x_1"), val = bool(false)]; + bool attn_115_transpose_y_1 = const()[name = string("attn_115_transpose_y_1"), val = bool(true)]; + tensor transpose_546_cast_fp16 = transpose(perm = transpose_546_perm_0, x = reshape_231_cast_fp16)[name = string("transpose_132")]; + tensor attn_115_cast_fp16 = matmul(transpose_x = attn_115_transpose_x_1, transpose_y = attn_115_transpose_y_1, x = transpose_546_cast_fp16, y = w_251_cast_fp16)[name = string("attn_115_cast_fp16")]; + tensor var_14424 = const()[name = string("op_14424"), val = tensor([1, 2048, 1, 1])]; + tensor input_613_cast_fp16 = reshape(shape = var_14424, x = attn_115_cast_fp16)[name = string("input_613_cast_fp16")]; + string attn_output_115_pad_type_0 = const()[name = string("attn_output_115_pad_type_0"), val = string("valid")]; + tensor attn_output_115_strides_0 = const()[name = string("attn_output_115_strides_0"), val = tensor([1, 1])]; + tensor attn_output_115_pad_0 = const()[name = string("attn_output_115_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_115_dilations_0 = const()[name = string("attn_output_115_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_115_groups_0 = const()[name = string("attn_output_115_groups_0"), val = int32(1)]; + tensor attn_output_115_cast_fp16 = conv(dilations = attn_output_115_dilations_0, groups = attn_output_115_groups_0, pad = attn_output_115_pad_0, pad_type = attn_output_115_pad_type_0, strides = attn_output_115_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_613_cast_fp16)[name = string("attn_output_115_cast_fp16")]; + tensor x_439_cast_fp16 = add(x = x_433_cast_fp16, y = attn_output_115_cast_fp16)[name = string("x_439_cast_fp16")]; + tensor var_14438_cast_fp16 = mul(x = x_439_cast_fp16, y = x_439_cast_fp16)[name = string("op_14438_cast_fp16")]; + tensor variance_483_axes_0 = const()[name = string("variance_483_axes_0"), val = tensor([1])]; + bool variance_483_keep_dims_0 = const()[name = string("variance_483_keep_dims_0"), val = bool(true)]; + tensor variance_483_cast_fp16 = reduce_mean(axes = variance_483_axes_0, keep_dims = variance_483_keep_dims_0, x = var_14438_cast_fp16)[name = string("variance_483_cast_fp16")]; + fp16 var_14441_to_fp16 = const()[name = string("op_14441_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14442_cast_fp16 = add(x = variance_483_cast_fp16, y = var_14441_to_fp16)[name = string("op_14442_cast_fp16")]; + fp32 var_14443_epsilon_0 = const()[name = string("op_14443_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14443_cast_fp16 = rsqrt(epsilon = var_14443_epsilon_0, x = var_14442_cast_fp16)[name = string("op_14443_cast_fp16")]; + tensor var_14444_cast_fp16 = mul(x = x_439_cast_fp16, y = var_14443_cast_fp16)[name = string("op_14444_cast_fp16")]; + tensor input_615_cast_fp16 = mul(x = var_14444_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_615_cast_fp16")]; + string input_617_pad_type_0 = const()[name = string("input_617_pad_type_0"), val = string("valid")]; + tensor input_617_strides_0 = const()[name = string("input_617_strides_0"), val = tensor([1, 1])]; + tensor input_617_pad_0 = const()[name = string("input_617_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_617_dilations_0 = const()[name = string("input_617_dilations_0"), val = tensor([1, 1])]; + int32 input_617_groups_0 = const()[name = string("input_617_groups_0"), val = int32(1)]; + tensor input_617_cast_fp16 = conv(dilations = input_617_dilations_0, groups = input_617_groups_0, pad = input_617_pad_0, pad_type = input_617_pad_type_0, strides = input_617_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_615_cast_fp16)[name = string("input_617_cast_fp16")]; + tensor var_14452_cast_fp16 = silu(x = input_617_cast_fp16)[name = string("op_14452_cast_fp16")]; + string var_14458_pad_type_0 = const()[name = string("op_14458_pad_type_0"), val = string("valid")]; + tensor var_14458_strides_0 = const()[name = string("op_14458_strides_0"), val = tensor([1, 1])]; + tensor var_14458_pad_0 = const()[name = string("op_14458_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14458_dilations_0 = const()[name = string("op_14458_dilations_0"), val = tensor([1, 1])]; + int32 var_14458_groups_0 = const()[name = string("op_14458_groups_0"), val = int32(1)]; + tensor var_14458_cast_fp16 = conv(dilations = var_14458_dilations_0, groups = var_14458_groups_0, pad = var_14458_pad_0, pad_type = var_14458_pad_type_0, strides = var_14458_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_615_cast_fp16)[name = string("op_14458_cast_fp16")]; + tensor input_619_cast_fp16 = mul(x = var_14452_cast_fp16, y = var_14458_cast_fp16)[name = string("input_619_cast_fp16")]; + string h_115_pad_type_0 = const()[name = string("h_115_pad_type_0"), val = string("valid")]; + tensor h_115_strides_0 = const()[name = string("h_115_strides_0"), val = tensor([1, 1])]; + tensor h_115_pad_0 = const()[name = string("h_115_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_115_dilations_0 = const()[name = string("h_115_dilations_0"), val = tensor([1, 1])]; + int32 h_115_groups_0 = const()[name = string("h_115_groups_0"), val = int32(1)]; + tensor h_115_cast_fp16 = conv(dilations = h_115_dilations_0, groups = h_115_groups_0, pad = h_115_pad_0, pad_type = h_115_pad_type_0, strides = h_115_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_619_cast_fp16)[name = string("h_115_cast_fp16")]; + tensor x_441_cast_fp16 = add(x = x_439_cast_fp16, y = h_115_cast_fp16)[name = string("x_441_cast_fp16")]; + tensor key_cache_117_begin_0 = const()[name = string("key_cache_117_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_117_end_0 = const()[name = string("key_cache_117_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_117_end_mask_0 = const()[name = string("key_cache_117_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_117_cast_fp16 = slice_by_index(begin = key_cache_117_begin_0, end = key_cache_117_end_0, end_mask = key_cache_117_end_mask_0, x = layer_key_caches_23_cast_fp16)[name = string("key_cache_117_cast_fp16")]; + tensor value_cache_117_begin_0 = const()[name = string("value_cache_117_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_117_end_0 = const()[name = string("value_cache_117_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_117_end_mask_0 = const()[name = string("value_cache_117_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_117_cast_fp16 = slice_by_index(begin = value_cache_117_begin_0, end = value_cache_117_end_0, end_mask = value_cache_117_end_mask_0, x = layer_value_caches_23_cast_fp16)[name = string("value_cache_117_cast_fp16")]; + int32 var_14511 = const()[name = string("op_14511"), val = int32(2)]; + int32 var_14515 = const()[name = string("op_14515"), val = int32(3)]; + tensor var_14530_cast_fp16 = mul(x = x_441_cast_fp16, y = x_441_cast_fp16)[name = string("op_14530_cast_fp16")]; + tensor variance_485_axes_0 = const()[name = string("variance_485_axes_0"), val = tensor([1])]; + bool variance_485_keep_dims_0 = const()[name = string("variance_485_keep_dims_0"), val = bool(true)]; + tensor variance_485_cast_fp16 = reduce_mean(axes = variance_485_axes_0, keep_dims = variance_485_keep_dims_0, x = var_14530_cast_fp16)[name = string("variance_485_cast_fp16")]; + fp16 var_14533_to_fp16 = const()[name = string("op_14533_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14534_cast_fp16 = add(x = variance_485_cast_fp16, y = var_14533_to_fp16)[name = string("op_14534_cast_fp16")]; + fp32 var_14535_epsilon_0 = const()[name = string("op_14535_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14535_cast_fp16 = rsqrt(epsilon = var_14535_epsilon_0, x = var_14534_cast_fp16)[name = string("op_14535_cast_fp16")]; + tensor var_14536_cast_fp16 = mul(x = x_441_cast_fp16, y = var_14535_cast_fp16)[name = string("op_14536_cast_fp16")]; + tensor input_621_cast_fp16 = mul(x = var_14536_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_621_cast_fp16")]; + string q_349_pad_type_0 = const()[name = string("q_349_pad_type_0"), val = string("valid")]; + tensor q_349_strides_0 = const()[name = string("q_349_strides_0"), val = tensor([1, 1])]; + tensor q_349_pad_0 = const()[name = string("q_349_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_349_dilations_0 = const()[name = string("q_349_dilations_0"), val = tensor([1, 1])]; + int32 q_349_groups_0 = const()[name = string("q_349_groups_0"), val = int32(1)]; + tensor q_349_cast_fp16 = conv(dilations = q_349_dilations_0, groups = q_349_groups_0, pad = q_349_pad_0, pad_type = q_349_pad_type_0, strides = q_349_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_621_cast_fp16)[name = string("q_349_cast_fp16")]; + string k_349_pad_type_0 = const()[name = string("k_349_pad_type_0"), val = string("valid")]; + tensor k_349_strides_0 = const()[name = string("k_349_strides_0"), val = tensor([1, 1])]; + tensor k_349_pad_0 = const()[name = string("k_349_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_349_dilations_0 = const()[name = string("k_349_dilations_0"), val = tensor([1, 1])]; + int32 k_349_groups_0 = const()[name = string("k_349_groups_0"), val = int32(1)]; + tensor k_349_cast_fp16 = conv(dilations = k_349_dilations_0, groups = k_349_groups_0, pad = k_349_pad_0, pad_type = k_349_pad_type_0, strides = k_349_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_621_cast_fp16)[name = string("k_349_cast_fp16")]; + string v_117_pad_type_0 = const()[name = string("v_117_pad_type_0"), val = string("valid")]; + tensor v_117_strides_0 = const()[name = string("v_117_strides_0"), val = tensor([1, 1])]; + tensor v_117_pad_0 = const()[name = string("v_117_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_117_dilations_0 = const()[name = string("v_117_dilations_0"), val = tensor([1, 1])]; + int32 v_117_groups_0 = const()[name = string("v_117_groups_0"), val = int32(1)]; + tensor v_117_cast_fp16 = conv(dilations = v_117_dilations_0, groups = v_117_groups_0, pad = v_117_pad_0, pad_type = v_117_pad_type_0, strides = v_117_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_621_cast_fp16)[name = string("v_117_cast_fp16")]; + tensor var_14570 = const()[name = string("op_14570"), val = tensor([16, 128, 1, 1])]; + tensor x_443_cast_fp16 = reshape(shape = var_14570, x = q_349_cast_fp16)[name = string("x_443_cast_fp16")]; + tensor var_14573_cast_fp16 = mul(x = x_443_cast_fp16, y = x_443_cast_fp16)[name = string("op_14573_cast_fp16")]; + tensor variance_487_axes_0 = const()[name = string("variance_487_axes_0"), val = tensor([1])]; + bool variance_487_keep_dims_0 = const()[name = string("variance_487_keep_dims_0"), val = bool(true)]; + tensor variance_487_cast_fp16 = reduce_mean(axes = variance_487_axes_0, keep_dims = variance_487_keep_dims_0, x = var_14573_cast_fp16)[name = string("variance_487_cast_fp16")]; + fp16 var_14576_to_fp16 = const()[name = string("op_14576_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14577_cast_fp16 = add(x = variance_487_cast_fp16, y = var_14576_to_fp16)[name = string("op_14577_cast_fp16")]; + fp32 var_14578_epsilon_0 = const()[name = string("op_14578_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14578_cast_fp16 = rsqrt(epsilon = var_14578_epsilon_0, x = var_14577_cast_fp16)[name = string("op_14578_cast_fp16")]; + tensor var_14579_cast_fp16 = mul(x = x_443_cast_fp16, y = var_14578_cast_fp16)[name = string("op_14579_cast_fp16")]; + tensor q_351_cast_fp16 = mul(x = var_14579_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_351_cast_fp16")]; + tensor var_14581 = const()[name = string("op_14581"), val = tensor([8, 128, 1, 1])]; + tensor x_445_cast_fp16 = reshape(shape = var_14581, x = k_349_cast_fp16)[name = string("x_445_cast_fp16")]; + tensor var_14584_cast_fp16 = mul(x = x_445_cast_fp16, y = x_445_cast_fp16)[name = string("op_14584_cast_fp16")]; + tensor variance_489_axes_0 = const()[name = string("variance_489_axes_0"), val = tensor([1])]; + bool variance_489_keep_dims_0 = const()[name = string("variance_489_keep_dims_0"), val = bool(true)]; + tensor variance_489_cast_fp16 = reduce_mean(axes = variance_489_axes_0, keep_dims = variance_489_keep_dims_0, x = var_14584_cast_fp16)[name = string("variance_489_cast_fp16")]; + fp16 var_14587_to_fp16 = const()[name = string("op_14587_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14588_cast_fp16 = add(x = variance_489_cast_fp16, y = var_14587_to_fp16)[name = string("op_14588_cast_fp16")]; + fp32 var_14589_epsilon_0 = const()[name = string("op_14589_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14589_cast_fp16 = rsqrt(epsilon = var_14589_epsilon_0, x = var_14588_cast_fp16)[name = string("op_14589_cast_fp16")]; + tensor var_14590_cast_fp16 = mul(x = x_445_cast_fp16, y = var_14589_cast_fp16)[name = string("op_14590_cast_fp16")]; + tensor k_351_cast_fp16 = mul(x = var_14590_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_351_cast_fp16")]; + tensor var_14592 = const()[name = string("op_14592"), val = tensor([1, 16, 128, 1])]; + tensor z_233_cast_fp16 = reshape(shape = var_14592, x = q_351_cast_fp16)[name = string("z_233_cast_fp16")]; + tensor var_14594 = const()[name = string("op_14594"), val = tensor([1, 8, 128, 1])]; + tensor z_235_cast_fp16 = reshape(shape = var_14594, x = k_351_cast_fp16)[name = string("z_235_cast_fp16")]; + tensor z1_233_begin_0 = const()[name = string("z1_233_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_233_end_0 = const()[name = string("z1_233_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_233_end_mask_0 = const()[name = string("z1_233_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_233_cast_fp16 = slice_by_index(begin = z1_233_begin_0, end = z1_233_end_0, end_mask = z1_233_end_mask_0, x = z_233_cast_fp16)[name = string("z1_233_cast_fp16")]; + tensor z2_233_begin_0 = const()[name = string("z2_233_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_233_end_0 = const()[name = string("z2_233_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_233_end_mask_0 = const()[name = string("z2_233_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_233_cast_fp16 = slice_by_index(begin = z2_233_begin_0, end = z2_233_end_0, end_mask = z2_233_end_mask_0, x = z_233_cast_fp16)[name = string("z2_233_cast_fp16")]; + tensor var_14602_cast_fp16 = mul(x = z_233_cast_fp16, y = cos_111_to_fp16)[name = string("op_14602_cast_fp16")]; + fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14603_cast_fp16 = mul(x = z2_233_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_14603_cast_fp16")]; + bool var_14605_interleave_0 = const()[name = string("op_14605_interleave_0"), val = bool(false)]; + tensor var_14605_cast_fp16 = concat(axis = var_14511, interleave = var_14605_interleave_0, values = (var_14603_cast_fp16, z1_233_cast_fp16))[name = string("op_14605_cast_fp16")]; + tensor var_14606_cast_fp16 = mul(x = var_14605_cast_fp16, y = sin_111_to_fp16)[name = string("op_14606_cast_fp16")]; + tensor q_353_cast_fp16 = add(x = var_14602_cast_fp16, y = var_14606_cast_fp16)[name = string("q_353_cast_fp16")]; + tensor z1_235_begin_0 = const()[name = string("z1_235_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_235_end_0 = const()[name = string("z1_235_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_235_end_mask_0 = const()[name = string("z1_235_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_235_cast_fp16 = slice_by_index(begin = z1_235_begin_0, end = z1_235_end_0, end_mask = z1_235_end_mask_0, x = z_235_cast_fp16)[name = string("z1_235_cast_fp16")]; + tensor z2_235_begin_0 = const()[name = string("z2_235_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_235_end_0 = const()[name = string("z2_235_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_235_end_mask_0 = const()[name = string("z2_235_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_235_cast_fp16 = slice_by_index(begin = z2_235_begin_0, end = z2_235_end_0, end_mask = z2_235_end_mask_0, x = z_235_cast_fp16)[name = string("z2_235_cast_fp16")]; + tensor var_14614_cast_fp16 = mul(x = z_235_cast_fp16, y = cos_111_to_fp16)[name = string("op_14614_cast_fp16")]; + fp16 const_129_promoted_to_fp16 = const()[name = string("const_129_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14615_cast_fp16 = mul(x = z2_235_cast_fp16, y = const_129_promoted_to_fp16)[name = string("op_14615_cast_fp16")]; + bool var_14617_interleave_0 = const()[name = string("op_14617_interleave_0"), val = bool(false)]; + tensor var_14617_cast_fp16 = concat(axis = var_14511, interleave = var_14617_interleave_0, values = (var_14615_cast_fp16, z1_235_cast_fp16))[name = string("op_14617_cast_fp16")]; + tensor var_14618_cast_fp16 = mul(x = var_14617_cast_fp16, y = sin_111_to_fp16)[name = string("op_14618_cast_fp16")]; + tensor k_353_cast_fp16 = add(x = var_14614_cast_fp16, y = var_14618_cast_fp16)[name = string("k_353_cast_fp16")]; + tensor var_14620 = const()[name = string("op_14620"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_117_cast_fp16 = reshape(shape = var_14620, x = k_353_cast_fp16)[name = string("cur_key_117_cast_fp16")]; + tensor var_14622_to_fp16 = const()[name = string("op_14622_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643264)))]; + tensor var_14623_cast_fp16 = mul(x = key_cache_117_cast_fp16, y = var_14622_to_fp16)[name = string("op_14623_cast_fp16")]; + tensor upd_117_to_fp16 = const()[name = string("upd_117_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643392)))]; + tensor var_14624_cast_fp16 = mul(x = cur_key_117_cast_fp16, y = upd_117_to_fp16)[name = string("op_14624_cast_fp16")]; + tensor key_117_cast_fp16 = add(x = var_14623_cast_fp16, y = var_14624_cast_fp16)[name = string("key_117_cast_fp16")]; + tensor var_14626_to_fp16 = const()[name = string("op_14626_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643264)))]; + tensor var_14627_cast_fp16 = mul(x = value_cache_117_cast_fp16, y = var_14626_to_fp16)[name = string("op_14627_cast_fp16")]; + tensor var_14628_cast_fp16 = mul(x = v_117_cast_fp16, y = upd_117_to_fp16)[name = string("op_14628_cast_fp16")]; + tensor value_117_cast_fp16 = add(x = var_14627_cast_fp16, y = var_14628_cast_fp16)[name = string("value_117_cast_fp16")]; + tensor var_14630 = const()[name = string("op_14630"), val = tensor([1, 8, 128, 16])]; + tensor kh_233_cast_fp16 = reshape(shape = var_14630, x = key_117_cast_fp16)[name = string("kh_233_cast_fp16")]; + tensor var_14632 = const()[name = string("op_14632"), val = tensor([1, 8, 128, 16])]; + tensor vh_233_cast_fp16 = reshape(shape = var_14632, x = value_117_cast_fp16)[name = string("vh_233_cast_fp16")]; + tensor transpose_232_perm_0 = const()[name = string("transpose_232_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_116_reps_0 = const()[name = string("tile_116_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_232_cast_fp16 = transpose(perm = transpose_232_perm_0, x = kh_233_cast_fp16)[name = string("transpose_131")]; + tensor tile_116_cast_fp16 = tile(reps = tile_116_reps_0, x = transpose_232_cast_fp16)[name = string("tile_116_cast_fp16")]; + tensor concat_290 = const()[name = string("concat_290"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_232_cast_fp16 = reshape(shape = concat_290, x = tile_116_cast_fp16)[name = string("reshape_232_cast_fp16")]; + tensor transpose_233_perm_0 = const()[name = string("transpose_233_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_291 = const()[name = string("concat_291"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_233_cast_fp16 = transpose(perm = transpose_233_perm_0, x = reshape_232_cast_fp16)[name = string("transpose_130")]; + tensor reshape_233_cast_fp16 = reshape(shape = concat_291, x = transpose_233_cast_fp16)[name = string("reshape_233_cast_fp16")]; + tensor transpose_234_perm_0 = const()[name = string("transpose_234_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_117_reps_0 = const()[name = string("tile_117_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_234_cast_fp16 = transpose(perm = transpose_234_perm_0, x = vh_233_cast_fp16)[name = string("transpose_129")]; + tensor tile_117_cast_fp16 = tile(reps = tile_117_reps_0, x = transpose_234_cast_fp16)[name = string("tile_117_cast_fp16")]; + tensor concat_292 = const()[name = string("concat_292"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_234_cast_fp16 = reshape(shape = concat_292, x = tile_117_cast_fp16)[name = string("reshape_234_cast_fp16")]; + tensor transpose_235_perm_0 = const()[name = string("transpose_235_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_293 = const()[name = string("concat_293"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_235_cast_fp16 = transpose(perm = transpose_235_perm_0, x = reshape_234_cast_fp16)[name = string("transpose_128")]; + tensor reshape_235_cast_fp16 = reshape(shape = concat_293, x = transpose_235_cast_fp16)[name = string("reshape_235_cast_fp16")]; + fp16 var_14636_to_fp16 = const()[name = string("op_14636_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_14637_cast_fp16 = mul(x = q_353_cast_fp16, y = var_14636_to_fp16)[name = string("op_14637_cast_fp16")]; + tensor transpose_549_perm_0 = const()[name = string("transpose_549_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_253_transpose_x_1 = const()[name = string("w_253_transpose_x_1"), val = bool(true)]; + bool w_253_transpose_y_1 = const()[name = string("w_253_transpose_y_1"), val = bool(false)]; + tensor transpose_549_cast_fp16 = transpose(perm = transpose_549_perm_0, x = reshape_233_cast_fp16)[name = string("transpose_127")]; + tensor w_253_cast_fp16 = matmul(transpose_x = w_253_transpose_x_1, transpose_y = w_253_transpose_y_1, x = var_14637_cast_fp16, y = transpose_549_cast_fp16)[name = string("w_253_cast_fp16")]; + tensor pad_117_to_fp16 = const()[name = string("pad_117_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643520)))]; + tensor var_14640_cast_fp16 = add(x = w_253_cast_fp16, y = pad_117_to_fp16)[name = string("op_14640_cast_fp16")]; + tensor w_255_cast_fp16 = softmax(axis = var_14515, x = var_14640_cast_fp16)[name = string("w_255_cast_fp16")]; + tensor transpose_550_perm_0 = const()[name = string("transpose_550_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_117_transpose_x_1 = const()[name = string("attn_117_transpose_x_1"), val = bool(false)]; + bool attn_117_transpose_y_1 = const()[name = string("attn_117_transpose_y_1"), val = bool(true)]; + tensor transpose_550_cast_fp16 = transpose(perm = transpose_550_perm_0, x = reshape_235_cast_fp16)[name = string("transpose_126")]; + tensor attn_117_cast_fp16 = matmul(transpose_x = attn_117_transpose_x_1, transpose_y = attn_117_transpose_y_1, x = transpose_550_cast_fp16, y = w_255_cast_fp16)[name = string("attn_117_cast_fp16")]; + tensor var_14644 = const()[name = string("op_14644"), val = tensor([1, 2048, 1, 1])]; + tensor input_623_cast_fp16 = reshape(shape = var_14644, x = attn_117_cast_fp16)[name = string("input_623_cast_fp16")]; + string attn_output_117_pad_type_0 = const()[name = string("attn_output_117_pad_type_0"), val = string("valid")]; + tensor attn_output_117_strides_0 = const()[name = string("attn_output_117_strides_0"), val = tensor([1, 1])]; + tensor attn_output_117_pad_0 = const()[name = string("attn_output_117_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_117_dilations_0 = const()[name = string("attn_output_117_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_117_groups_0 = const()[name = string("attn_output_117_groups_0"), val = int32(1)]; + tensor attn_output_117_cast_fp16 = conv(dilations = attn_output_117_dilations_0, groups = attn_output_117_groups_0, pad = attn_output_117_pad_0, pad_type = attn_output_117_pad_type_0, strides = attn_output_117_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_623_cast_fp16)[name = string("attn_output_117_cast_fp16")]; + tensor x_447_cast_fp16 = add(x = x_441_cast_fp16, y = attn_output_117_cast_fp16)[name = string("x_447_cast_fp16")]; + tensor var_14658_cast_fp16 = mul(x = x_447_cast_fp16, y = x_447_cast_fp16)[name = string("op_14658_cast_fp16")]; + tensor variance_491_axes_0 = const()[name = string("variance_491_axes_0"), val = tensor([1])]; + bool variance_491_keep_dims_0 = const()[name = string("variance_491_keep_dims_0"), val = bool(true)]; + tensor variance_491_cast_fp16 = reduce_mean(axes = variance_491_axes_0, keep_dims = variance_491_keep_dims_0, x = var_14658_cast_fp16)[name = string("variance_491_cast_fp16")]; + fp16 var_14661_to_fp16 = const()[name = string("op_14661_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14662_cast_fp16 = add(x = variance_491_cast_fp16, y = var_14661_to_fp16)[name = string("op_14662_cast_fp16")]; + fp32 var_14663_epsilon_0 = const()[name = string("op_14663_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14663_cast_fp16 = rsqrt(epsilon = var_14663_epsilon_0, x = var_14662_cast_fp16)[name = string("op_14663_cast_fp16")]; + tensor var_14664_cast_fp16 = mul(x = x_447_cast_fp16, y = var_14663_cast_fp16)[name = string("op_14664_cast_fp16")]; + tensor input_625_cast_fp16 = mul(x = var_14664_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_625_cast_fp16")]; + string input_627_pad_type_0 = const()[name = string("input_627_pad_type_0"), val = string("valid")]; + tensor input_627_strides_0 = const()[name = string("input_627_strides_0"), val = tensor([1, 1])]; + tensor input_627_pad_0 = const()[name = string("input_627_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_627_dilations_0 = const()[name = string("input_627_dilations_0"), val = tensor([1, 1])]; + int32 input_627_groups_0 = const()[name = string("input_627_groups_0"), val = int32(1)]; + tensor input_627_cast_fp16 = conv(dilations = input_627_dilations_0, groups = input_627_groups_0, pad = input_627_pad_0, pad_type = input_627_pad_type_0, strides = input_627_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_625_cast_fp16)[name = string("input_627_cast_fp16")]; + tensor var_14672_cast_fp16 = silu(x = input_627_cast_fp16)[name = string("op_14672_cast_fp16")]; + string var_14678_pad_type_0 = const()[name = string("op_14678_pad_type_0"), val = string("valid")]; + tensor var_14678_strides_0 = const()[name = string("op_14678_strides_0"), val = tensor([1, 1])]; + tensor var_14678_pad_0 = const()[name = string("op_14678_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14678_dilations_0 = const()[name = string("op_14678_dilations_0"), val = tensor([1, 1])]; + int32 var_14678_groups_0 = const()[name = string("op_14678_groups_0"), val = int32(1)]; + tensor var_14678_cast_fp16 = conv(dilations = var_14678_dilations_0, groups = var_14678_groups_0, pad = var_14678_pad_0, pad_type = var_14678_pad_type_0, strides = var_14678_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_625_cast_fp16)[name = string("op_14678_cast_fp16")]; + tensor input_629_cast_fp16 = mul(x = var_14672_cast_fp16, y = var_14678_cast_fp16)[name = string("input_629_cast_fp16")]; + string h_117_pad_type_0 = const()[name = string("h_117_pad_type_0"), val = string("valid")]; + tensor h_117_strides_0 = const()[name = string("h_117_strides_0"), val = tensor([1, 1])]; + tensor h_117_pad_0 = const()[name = string("h_117_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_117_dilations_0 = const()[name = string("h_117_dilations_0"), val = tensor([1, 1])]; + int32 h_117_groups_0 = const()[name = string("h_117_groups_0"), val = int32(1)]; + tensor h_117_cast_fp16 = conv(dilations = h_117_dilations_0, groups = h_117_groups_0, pad = h_117_pad_0, pad_type = h_117_pad_type_0, strides = h_117_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_629_cast_fp16)[name = string("h_117_cast_fp16")]; + tensor x_449_cast_fp16 = add(x = x_447_cast_fp16, y = h_117_cast_fp16)[name = string("x_449_cast_fp16")]; + tensor key_cache_119_begin_0 = const()[name = string("key_cache_119_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_119_end_0 = const()[name = string("key_cache_119_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_119_end_mask_0 = const()[name = string("key_cache_119_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_119_cast_fp16 = slice_by_index(begin = key_cache_119_begin_0, end = key_cache_119_end_0, end_mask = key_cache_119_end_mask_0, x = layer_key_caches_23_cast_fp16)[name = string("key_cache_119_cast_fp16")]; + tensor value_cache_119_begin_0 = const()[name = string("value_cache_119_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_119_end_0 = const()[name = string("value_cache_119_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_119_end_mask_0 = const()[name = string("value_cache_119_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_119_cast_fp16 = slice_by_index(begin = value_cache_119_begin_0, end = value_cache_119_end_0, end_mask = value_cache_119_end_mask_0, x = layer_value_caches_23_cast_fp16)[name = string("value_cache_119_cast_fp16")]; + int32 var_14731 = const()[name = string("op_14731"), val = int32(2)]; + int32 var_14735 = const()[name = string("op_14735"), val = int32(3)]; + tensor var_14750_cast_fp16 = mul(x = x_449_cast_fp16, y = x_449_cast_fp16)[name = string("op_14750_cast_fp16")]; + tensor variance_493_axes_0 = const()[name = string("variance_493_axes_0"), val = tensor([1])]; + bool variance_493_keep_dims_0 = const()[name = string("variance_493_keep_dims_0"), val = bool(true)]; + tensor variance_493_cast_fp16 = reduce_mean(axes = variance_493_axes_0, keep_dims = variance_493_keep_dims_0, x = var_14750_cast_fp16)[name = string("variance_493_cast_fp16")]; + fp16 var_14753_to_fp16 = const()[name = string("op_14753_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14754_cast_fp16 = add(x = variance_493_cast_fp16, y = var_14753_to_fp16)[name = string("op_14754_cast_fp16")]; + fp32 var_14755_epsilon_0 = const()[name = string("op_14755_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14755_cast_fp16 = rsqrt(epsilon = var_14755_epsilon_0, x = var_14754_cast_fp16)[name = string("op_14755_cast_fp16")]; + tensor var_14756_cast_fp16 = mul(x = x_449_cast_fp16, y = var_14755_cast_fp16)[name = string("op_14756_cast_fp16")]; + tensor input_631_cast_fp16 = mul(x = var_14756_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_631_cast_fp16")]; + string q_355_pad_type_0 = const()[name = string("q_355_pad_type_0"), val = string("valid")]; + tensor q_355_strides_0 = const()[name = string("q_355_strides_0"), val = tensor([1, 1])]; + tensor q_355_pad_0 = const()[name = string("q_355_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_355_dilations_0 = const()[name = string("q_355_dilations_0"), val = tensor([1, 1])]; + int32 q_355_groups_0 = const()[name = string("q_355_groups_0"), val = int32(1)]; + tensor q_355_cast_fp16 = conv(dilations = q_355_dilations_0, groups = q_355_groups_0, pad = q_355_pad_0, pad_type = q_355_pad_type_0, strides = q_355_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_631_cast_fp16)[name = string("q_355_cast_fp16")]; + string k_355_pad_type_0 = const()[name = string("k_355_pad_type_0"), val = string("valid")]; + tensor k_355_strides_0 = const()[name = string("k_355_strides_0"), val = tensor([1, 1])]; + tensor k_355_pad_0 = const()[name = string("k_355_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_355_dilations_0 = const()[name = string("k_355_dilations_0"), val = tensor([1, 1])]; + int32 k_355_groups_0 = const()[name = string("k_355_groups_0"), val = int32(1)]; + tensor k_355_cast_fp16 = conv(dilations = k_355_dilations_0, groups = k_355_groups_0, pad = k_355_pad_0, pad_type = k_355_pad_type_0, strides = k_355_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_631_cast_fp16)[name = string("k_355_cast_fp16")]; + string v_119_pad_type_0 = const()[name = string("v_119_pad_type_0"), val = string("valid")]; + tensor v_119_strides_0 = const()[name = string("v_119_strides_0"), val = tensor([1, 1])]; + tensor v_119_pad_0 = const()[name = string("v_119_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_119_dilations_0 = const()[name = string("v_119_dilations_0"), val = tensor([1, 1])]; + int32 v_119_groups_0 = const()[name = string("v_119_groups_0"), val = int32(1)]; + tensor v_119_cast_fp16 = conv(dilations = v_119_dilations_0, groups = v_119_groups_0, pad = v_119_pad_0, pad_type = v_119_pad_type_0, strides = v_119_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_631_cast_fp16)[name = string("v_119_cast_fp16")]; + tensor var_14790 = const()[name = string("op_14790"), val = tensor([16, 128, 1, 1])]; + tensor x_451_cast_fp16 = reshape(shape = var_14790, x = q_355_cast_fp16)[name = string("x_451_cast_fp16")]; + tensor var_14793_cast_fp16 = mul(x = x_451_cast_fp16, y = x_451_cast_fp16)[name = string("op_14793_cast_fp16")]; + tensor variance_495_axes_0 = const()[name = string("variance_495_axes_0"), val = tensor([1])]; + bool variance_495_keep_dims_0 = const()[name = string("variance_495_keep_dims_0"), val = bool(true)]; + tensor variance_495_cast_fp16 = reduce_mean(axes = variance_495_axes_0, keep_dims = variance_495_keep_dims_0, x = var_14793_cast_fp16)[name = string("variance_495_cast_fp16")]; + fp16 var_14796_to_fp16 = const()[name = string("op_14796_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14797_cast_fp16 = add(x = variance_495_cast_fp16, y = var_14796_to_fp16)[name = string("op_14797_cast_fp16")]; + fp32 var_14798_epsilon_0 = const()[name = string("op_14798_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14798_cast_fp16 = rsqrt(epsilon = var_14798_epsilon_0, x = var_14797_cast_fp16)[name = string("op_14798_cast_fp16")]; + tensor var_14799_cast_fp16 = mul(x = x_451_cast_fp16, y = var_14798_cast_fp16)[name = string("op_14799_cast_fp16")]; + tensor q_357_cast_fp16 = mul(x = var_14799_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_357_cast_fp16")]; + tensor var_14801 = const()[name = string("op_14801"), val = tensor([8, 128, 1, 1])]; + tensor x_453_cast_fp16 = reshape(shape = var_14801, x = k_355_cast_fp16)[name = string("x_453_cast_fp16")]; + tensor var_14804_cast_fp16 = mul(x = x_453_cast_fp16, y = x_453_cast_fp16)[name = string("op_14804_cast_fp16")]; + tensor variance_497_axes_0 = const()[name = string("variance_497_axes_0"), val = tensor([1])]; + bool variance_497_keep_dims_0 = const()[name = string("variance_497_keep_dims_0"), val = bool(true)]; + tensor variance_497_cast_fp16 = reduce_mean(axes = variance_497_axes_0, keep_dims = variance_497_keep_dims_0, x = var_14804_cast_fp16)[name = string("variance_497_cast_fp16")]; + fp16 var_14807_to_fp16 = const()[name = string("op_14807_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14808_cast_fp16 = add(x = variance_497_cast_fp16, y = var_14807_to_fp16)[name = string("op_14808_cast_fp16")]; + fp32 var_14809_epsilon_0 = const()[name = string("op_14809_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14809_cast_fp16 = rsqrt(epsilon = var_14809_epsilon_0, x = var_14808_cast_fp16)[name = string("op_14809_cast_fp16")]; + tensor var_14810_cast_fp16 = mul(x = x_453_cast_fp16, y = var_14809_cast_fp16)[name = string("op_14810_cast_fp16")]; + tensor k_357_cast_fp16 = mul(x = var_14810_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_357_cast_fp16")]; + tensor var_14812 = const()[name = string("op_14812"), val = tensor([1, 16, 128, 1])]; + tensor z_237_cast_fp16 = reshape(shape = var_14812, x = q_357_cast_fp16)[name = string("z_237_cast_fp16")]; + tensor var_14814 = const()[name = string("op_14814"), val = tensor([1, 8, 128, 1])]; + tensor z_239_cast_fp16 = reshape(shape = var_14814, x = k_357_cast_fp16)[name = string("z_239_cast_fp16")]; + tensor z1_237_begin_0 = const()[name = string("z1_237_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_237_end_0 = const()[name = string("z1_237_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_237_end_mask_0 = const()[name = string("z1_237_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_237_cast_fp16 = slice_by_index(begin = z1_237_begin_0, end = z1_237_end_0, end_mask = z1_237_end_mask_0, x = z_237_cast_fp16)[name = string("z1_237_cast_fp16")]; + tensor z2_237_begin_0 = const()[name = string("z2_237_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_237_end_0 = const()[name = string("z2_237_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_237_end_mask_0 = const()[name = string("z2_237_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_237_cast_fp16 = slice_by_index(begin = z2_237_begin_0, end = z2_237_end_0, end_mask = z2_237_end_mask_0, x = z_237_cast_fp16)[name = string("z2_237_cast_fp16")]; + tensor var_14822_cast_fp16 = mul(x = z_237_cast_fp16, y = cos_111_to_fp16)[name = string("op_14822_cast_fp16")]; + fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14823_cast_fp16 = mul(x = z2_237_cast_fp16, y = const_130_promoted_to_fp16)[name = string("op_14823_cast_fp16")]; + bool var_14825_interleave_0 = const()[name = string("op_14825_interleave_0"), val = bool(false)]; + tensor var_14825_cast_fp16 = concat(axis = var_14731, interleave = var_14825_interleave_0, values = (var_14823_cast_fp16, z1_237_cast_fp16))[name = string("op_14825_cast_fp16")]; + tensor var_14826_cast_fp16 = mul(x = var_14825_cast_fp16, y = sin_111_to_fp16)[name = string("op_14826_cast_fp16")]; + tensor q_359_cast_fp16 = add(x = var_14822_cast_fp16, y = var_14826_cast_fp16)[name = string("q_359_cast_fp16")]; + tensor z1_239_begin_0 = const()[name = string("z1_239_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_239_end_0 = const()[name = string("z1_239_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_239_end_mask_0 = const()[name = string("z1_239_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_239_cast_fp16 = slice_by_index(begin = z1_239_begin_0, end = z1_239_end_0, end_mask = z1_239_end_mask_0, x = z_239_cast_fp16)[name = string("z1_239_cast_fp16")]; + tensor z2_239_begin_0 = const()[name = string("z2_239_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_239_end_0 = const()[name = string("z2_239_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_239_end_mask_0 = const()[name = string("z2_239_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_239_cast_fp16 = slice_by_index(begin = z2_239_begin_0, end = z2_239_end_0, end_mask = z2_239_end_mask_0, x = z_239_cast_fp16)[name = string("z2_239_cast_fp16")]; + tensor var_14834_cast_fp16 = mul(x = z_239_cast_fp16, y = cos_111_to_fp16)[name = string("op_14834_cast_fp16")]; + fp16 const_131_promoted_to_fp16 = const()[name = string("const_131_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14835_cast_fp16 = mul(x = z2_239_cast_fp16, y = const_131_promoted_to_fp16)[name = string("op_14835_cast_fp16")]; + bool var_14837_interleave_0 = const()[name = string("op_14837_interleave_0"), val = bool(false)]; + tensor var_14837_cast_fp16 = concat(axis = var_14731, interleave = var_14837_interleave_0, values = (var_14835_cast_fp16, z1_239_cast_fp16))[name = string("op_14837_cast_fp16")]; + tensor var_14838_cast_fp16 = mul(x = var_14837_cast_fp16, y = sin_111_to_fp16)[name = string("op_14838_cast_fp16")]; + tensor k_359_cast_fp16 = add(x = var_14834_cast_fp16, y = var_14838_cast_fp16)[name = string("k_359_cast_fp16")]; + tensor var_14840 = const()[name = string("op_14840"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_119_cast_fp16 = reshape(shape = var_14840, x = k_359_cast_fp16)[name = string("cur_key_119_cast_fp16")]; + tensor var_14842_to_fp16 = const()[name = string("op_14842_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643264)))]; + tensor var_14843_cast_fp16 = mul(x = key_cache_119_cast_fp16, y = var_14842_to_fp16)[name = string("op_14843_cast_fp16")]; + tensor upd_119_to_fp16 = const()[name = string("upd_119_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643392)))]; + tensor var_14844_cast_fp16 = mul(x = cur_key_119_cast_fp16, y = upd_119_to_fp16)[name = string("op_14844_cast_fp16")]; + tensor key_119_cast_fp16 = add(x = var_14843_cast_fp16, y = var_14844_cast_fp16)[name = string("key_119_cast_fp16")]; + tensor var_14846_to_fp16 = const()[name = string("op_14846_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643264)))]; + tensor var_14847_cast_fp16 = mul(x = value_cache_119_cast_fp16, y = var_14846_to_fp16)[name = string("op_14847_cast_fp16")]; + tensor var_14848_cast_fp16 = mul(x = v_119_cast_fp16, y = upd_119_to_fp16)[name = string("op_14848_cast_fp16")]; + tensor value_119_cast_fp16 = add(x = var_14847_cast_fp16, y = var_14848_cast_fp16)[name = string("value_119_cast_fp16")]; + tensor var_14850 = const()[name = string("op_14850"), val = tensor([1, 8, 128, 16])]; + tensor kh_237_cast_fp16 = reshape(shape = var_14850, x = key_119_cast_fp16)[name = string("kh_237_cast_fp16")]; + tensor var_14852 = const()[name = string("op_14852"), val = tensor([1, 8, 128, 16])]; + tensor vh_237_cast_fp16 = reshape(shape = var_14852, x = value_119_cast_fp16)[name = string("vh_237_cast_fp16")]; + tensor transpose_236_perm_0 = const()[name = string("transpose_236_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_118_reps_0 = const()[name = string("tile_118_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_236_cast_fp16 = transpose(perm = transpose_236_perm_0, x = kh_237_cast_fp16)[name = string("transpose_125")]; + tensor tile_118_cast_fp16 = tile(reps = tile_118_reps_0, x = transpose_236_cast_fp16)[name = string("tile_118_cast_fp16")]; + tensor concat_294 = const()[name = string("concat_294"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_236_cast_fp16 = reshape(shape = concat_294, x = tile_118_cast_fp16)[name = string("reshape_236_cast_fp16")]; + tensor transpose_237_perm_0 = const()[name = string("transpose_237_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_295 = const()[name = string("concat_295"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_237_cast_fp16 = transpose(perm = transpose_237_perm_0, x = reshape_236_cast_fp16)[name = string("transpose_124")]; + tensor reshape_237_cast_fp16 = reshape(shape = concat_295, x = transpose_237_cast_fp16)[name = string("reshape_237_cast_fp16")]; + tensor transpose_238_perm_0 = const()[name = string("transpose_238_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_119_reps_0 = const()[name = string("tile_119_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_238_cast_fp16 = transpose(perm = transpose_238_perm_0, x = vh_237_cast_fp16)[name = string("transpose_123")]; + tensor tile_119_cast_fp16 = tile(reps = tile_119_reps_0, x = transpose_238_cast_fp16)[name = string("tile_119_cast_fp16")]; + tensor concat_296 = const()[name = string("concat_296"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_238_cast_fp16 = reshape(shape = concat_296, x = tile_119_cast_fp16)[name = string("reshape_238_cast_fp16")]; + tensor transpose_239_perm_0 = const()[name = string("transpose_239_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_297 = const()[name = string("concat_297"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_239_cast_fp16 = transpose(perm = transpose_239_perm_0, x = reshape_238_cast_fp16)[name = string("transpose_122")]; + tensor reshape_239_cast_fp16 = reshape(shape = concat_297, x = transpose_239_cast_fp16)[name = string("reshape_239_cast_fp16")]; + fp16 var_14856_to_fp16 = const()[name = string("op_14856_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_14857_cast_fp16 = mul(x = q_359_cast_fp16, y = var_14856_to_fp16)[name = string("op_14857_cast_fp16")]; + tensor transpose_553_perm_0 = const()[name = string("transpose_553_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_257_transpose_x_1 = const()[name = string("w_257_transpose_x_1"), val = bool(true)]; + bool w_257_transpose_y_1 = const()[name = string("w_257_transpose_y_1"), val = bool(false)]; + tensor transpose_553_cast_fp16 = transpose(perm = transpose_553_perm_0, x = reshape_237_cast_fp16)[name = string("transpose_121")]; + tensor w_257_cast_fp16 = matmul(transpose_x = w_257_transpose_x_1, transpose_y = w_257_transpose_y_1, x = var_14857_cast_fp16, y = transpose_553_cast_fp16)[name = string("w_257_cast_fp16")]; + tensor pad_119_to_fp16 = const()[name = string("pad_119_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643520)))]; + tensor var_14860_cast_fp16 = add(x = w_257_cast_fp16, y = pad_119_to_fp16)[name = string("op_14860_cast_fp16")]; + tensor w_259_cast_fp16 = softmax(axis = var_14735, x = var_14860_cast_fp16)[name = string("w_259_cast_fp16")]; + tensor transpose_554_perm_0 = const()[name = string("transpose_554_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_119_transpose_x_1 = const()[name = string("attn_119_transpose_x_1"), val = bool(false)]; + bool attn_119_transpose_y_1 = const()[name = string("attn_119_transpose_y_1"), val = bool(true)]; + tensor transpose_554_cast_fp16 = transpose(perm = transpose_554_perm_0, x = reshape_239_cast_fp16)[name = string("transpose_120")]; + tensor attn_119_cast_fp16 = matmul(transpose_x = attn_119_transpose_x_1, transpose_y = attn_119_transpose_y_1, x = transpose_554_cast_fp16, y = w_259_cast_fp16)[name = string("attn_119_cast_fp16")]; + tensor var_14864 = const()[name = string("op_14864"), val = tensor([1, 2048, 1, 1])]; + tensor input_633_cast_fp16 = reshape(shape = var_14864, x = attn_119_cast_fp16)[name = string("input_633_cast_fp16")]; + string attn_output_119_pad_type_0 = const()[name = string("attn_output_119_pad_type_0"), val = string("valid")]; + tensor attn_output_119_strides_0 = const()[name = string("attn_output_119_strides_0"), val = tensor([1, 1])]; + tensor attn_output_119_pad_0 = const()[name = string("attn_output_119_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_119_dilations_0 = const()[name = string("attn_output_119_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_119_groups_0 = const()[name = string("attn_output_119_groups_0"), val = int32(1)]; + tensor attn_output_119_cast_fp16 = conv(dilations = attn_output_119_dilations_0, groups = attn_output_119_groups_0, pad = attn_output_119_pad_0, pad_type = attn_output_119_pad_type_0, strides = attn_output_119_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_633_cast_fp16)[name = string("attn_output_119_cast_fp16")]; + tensor x_455_cast_fp16 = add(x = x_449_cast_fp16, y = attn_output_119_cast_fp16)[name = string("x_455_cast_fp16")]; + tensor var_14878_cast_fp16 = mul(x = x_455_cast_fp16, y = x_455_cast_fp16)[name = string("op_14878_cast_fp16")]; + tensor variance_499_axes_0 = const()[name = string("variance_499_axes_0"), val = tensor([1])]; + bool variance_499_keep_dims_0 = const()[name = string("variance_499_keep_dims_0"), val = bool(true)]; + tensor variance_499_cast_fp16 = reduce_mean(axes = variance_499_axes_0, keep_dims = variance_499_keep_dims_0, x = var_14878_cast_fp16)[name = string("variance_499_cast_fp16")]; + fp16 var_14881_to_fp16 = const()[name = string("op_14881_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14882_cast_fp16 = add(x = variance_499_cast_fp16, y = var_14881_to_fp16)[name = string("op_14882_cast_fp16")]; + fp32 var_14883_epsilon_0 = const()[name = string("op_14883_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14883_cast_fp16 = rsqrt(epsilon = var_14883_epsilon_0, x = var_14882_cast_fp16)[name = string("op_14883_cast_fp16")]; + tensor var_14884_cast_fp16 = mul(x = x_455_cast_fp16, y = var_14883_cast_fp16)[name = string("op_14884_cast_fp16")]; + tensor input_635_cast_fp16 = mul(x = var_14884_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_635_cast_fp16")]; + string input_637_pad_type_0 = const()[name = string("input_637_pad_type_0"), val = string("valid")]; + tensor input_637_strides_0 = const()[name = string("input_637_strides_0"), val = tensor([1, 1])]; + tensor input_637_pad_0 = const()[name = string("input_637_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_637_dilations_0 = const()[name = string("input_637_dilations_0"), val = tensor([1, 1])]; + int32 input_637_groups_0 = const()[name = string("input_637_groups_0"), val = int32(1)]; + tensor input_637_cast_fp16 = conv(dilations = input_637_dilations_0, groups = input_637_groups_0, pad = input_637_pad_0, pad_type = input_637_pad_type_0, strides = input_637_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_635_cast_fp16)[name = string("input_637_cast_fp16")]; + tensor var_14892_cast_fp16 = silu(x = input_637_cast_fp16)[name = string("op_14892_cast_fp16")]; + string var_14898_pad_type_0 = const()[name = string("op_14898_pad_type_0"), val = string("valid")]; + tensor var_14898_strides_0 = const()[name = string("op_14898_strides_0"), val = tensor([1, 1])]; + tensor var_14898_pad_0 = const()[name = string("op_14898_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14898_dilations_0 = const()[name = string("op_14898_dilations_0"), val = tensor([1, 1])]; + int32 var_14898_groups_0 = const()[name = string("op_14898_groups_0"), val = int32(1)]; + tensor var_14898_cast_fp16 = conv(dilations = var_14898_dilations_0, groups = var_14898_groups_0, pad = var_14898_pad_0, pad_type = var_14898_pad_type_0, strides = var_14898_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_635_cast_fp16)[name = string("op_14898_cast_fp16")]; + tensor input_639_cast_fp16 = mul(x = var_14892_cast_fp16, y = var_14898_cast_fp16)[name = string("input_639_cast_fp16")]; + string h_119_pad_type_0 = const()[name = string("h_119_pad_type_0"), val = string("valid")]; + tensor h_119_strides_0 = const()[name = string("h_119_strides_0"), val = tensor([1, 1])]; + tensor h_119_pad_0 = const()[name = string("h_119_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_119_dilations_0 = const()[name = string("h_119_dilations_0"), val = tensor([1, 1])]; + int32 h_119_groups_0 = const()[name = string("h_119_groups_0"), val = int32(1)]; + tensor h_119_cast_fp16 = conv(dilations = h_119_dilations_0, groups = h_119_groups_0, pad = h_119_pad_0, pad_type = h_119_pad_type_0, strides = h_119_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_639_cast_fp16)[name = string("h_119_cast_fp16")]; + tensor inputs_21_cast_fp16 = add(x = x_455_cast_fp16, y = h_119_cast_fp16)[name = string("inputs_21_cast_fp16")]; + int32 var_14926 = const()[name = string("op_14926"), val = int32(1)]; + bool layer_key_caches_25_interleave_0 = const()[name = string("layer_key_caches_25_interleave_0"), val = bool(false)]; + tensor layer_key_caches_25_cast_fp16 = concat(axis = var_14926, interleave = layer_key_caches_25_interleave_0, values = (key_111_cast_fp16, key_113_cast_fp16, key_115_cast_fp16, key_117_cast_fp16, key_119_cast_fp16))[name = string("layer_key_caches_25_cast_fp16")]; + int32 var_14929 = const()[name = string("op_14929"), val = int32(1)]; + bool layer_value_caches_25_interleave_0 = const()[name = string("layer_value_caches_25_interleave_0"), val = bool(false)]; + tensor layer_value_caches_25_cast_fp16 = concat(axis = var_14929, interleave = layer_value_caches_25_interleave_0, values = (value_111_cast_fp16, value_113_cast_fp16, value_115_cast_fp16, value_117_cast_fp16, value_119_cast_fp16))[name = string("layer_value_caches_25_cast_fp16")]; + tensor inputs_sq_21_cast_fp16 = mul(x = inputs_21_cast_fp16, y = inputs_21_cast_fp16)[name = string("inputs_sq_21_cast_fp16")]; + tensor variance_501_axes_0 = const()[name = string("variance_501_axes_0"), val = tensor([1])]; + bool variance_501_keep_dims_0 = const()[name = string("variance_501_keep_dims_0"), val = bool(true)]; + tensor variance_501_cast_fp16 = reduce_mean(axes = variance_501_axes_0, keep_dims = variance_501_keep_dims_0, x = inputs_sq_21_cast_fp16)[name = string("variance_501_cast_fp16")]; + fp16 var_14939_to_fp16 = const()[name = string("op_14939_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14940_cast_fp16 = add(x = variance_501_cast_fp16, y = var_14939_to_fp16)[name = string("op_14940_cast_fp16")]; + fp32 var_14941_epsilon_0 = const()[name = string("op_14941_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14941_cast_fp16 = rsqrt(epsilon = var_14941_epsilon_0, x = var_14940_cast_fp16)[name = string("op_14941_cast_fp16")]; + tensor hidden_states_21_cast_fp16 = mul(x = inputs_21_cast_fp16, y = var_14941_cast_fp16)[name = string("hidden_states_21_cast_fp16")]; + tensor input_641_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_21_cast_fp16)[name = string("input_641_cast_fp16")]; + string logits_41_pad_type_0 = const()[name = string("logits_41_pad_type_0"), val = string("valid")]; + tensor logits_41_strides_0 = const()[name = string("logits_41_strides_0"), val = tensor([1, 1])]; + tensor logits_41_pad_0 = const()[name = string("logits_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_41_dilations_0 = const()[name = string("logits_41_dilations_0"), val = tensor([1, 1])]; + int32 logits_41_groups_0 = const()[name = string("logits_41_groups_0"), val = int32(1)]; + tensor lm_heads_10_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(99684608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101781824))))[name = string("lm_heads_10_weight_to_fp16_palettized")]; + tensor logits_41_cast_fp16 = conv(dilations = logits_41_dilations_0, groups = logits_41_groups_0, pad = logits_41_pad_0, pad_type = logits_41_pad_type_0, strides = logits_41_strides_0, weight = lm_heads_10_weight_to_fp16_palettized, x = input_641_cast_fp16)[name = string("logits_41_cast_fp16")]; + tensor var_14959 = const()[name = string("op_14959"), val = tensor([1, 2048])]; + tensor logits_43_cast_fp16 = reshape(shape = var_14959, x = logits_41_cast_fp16)[name = string("logits_43_cast_fp16")]; + tensor scaled_logits_21_cast_fp16 = real_div(x = logits_43_cast_fp16, y = temperature)[name = string("scaled_logits_21_cast_fp16")]; + int32 var_14969 = const()[name = string("op_14969"), val = int32(100)]; + int32 top_values_21_axis_0 = const()[name = string("top_values_21_axis_0"), val = int32(1)]; + bool top_values_21_ascending_0 = const()[name = string("top_values_21_ascending_0"), val = bool(false)]; + bool top_values_21_sort_0 = const()[name = string("top_values_21_sort_0"), val = bool(true)]; + bool top_values_21_return_indices_0 = const()[name = string("top_values_21_return_indices_0"), val = bool(true)]; + string top_values_21_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_21_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_21_cast_fp16_cast_uint16_0, tensor top_values_21_cast_fp16_cast_uint16_1 = topk(ascending = top_values_21_ascending_0, axis = top_values_21_axis_0, k = var_14969, output_indices_dtype = top_values_21_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_21_return_indices_0, sort = top_values_21_sort_0, x = scaled_logits_21_cast_fp16)[name = string("top_values_21_cast_fp16_cast_uint16")]; + tensor var_14975_cast_fp16 = mul(x = top_values_21_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_14975_cast_fp16")]; + tensor var_14979_cast_fp16 = add(x = var_14975_cast_fp16, y = var_2438_cast_fp16)[name = string("op_14979_cast_fp16")]; + tensor reduce_min_10_axes_0 = const()[name = string("reduce_min_10_axes_0"), val = tensor([1])]; + bool reduce_min_10_keep_dims_0 = const()[name = string("reduce_min_10_keep_dims_0"), val = bool(true)]; + tensor reduce_min_10_cast_fp16 = reduce_min(axes = reduce_min_10_axes_0, keep_dims = reduce_min_10_keep_dims_0, x = var_14979_cast_fp16)[name = string("reduce_min_10_cast_fp16")]; + tensor var_14982_cast_fp16 = greater_equal(x = scaled_logits_21_cast_fp16, y = reduce_min_10_cast_fp16)[name = string("op_14982_cast_fp16")]; + fp16 var_14983_value_0_to_fp16 = const()[name = string("op_14983_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_14983_cast_fp16 = fill_like(ref_tensor = scaled_logits_21_cast_fp16, value = var_14983_value_0_to_fp16)[name = string("op_14983_cast_fp16")]; + tensor masked_logits_21_cast_fp16 = select(a = scaled_logits_21_cast_fp16, b = var_14983_cast_fp16, cond = var_14982_cast_fp16)[name = string("masked_logits_21_cast_fp16")]; + tensor var_14987_begin_0 = const()[name = string("op_14987_begin_0"), val = tensor([10, 0])]; + tensor var_14987_end_0 = const()[name = string("op_14987_end_0"), val = tensor([11, 2048])]; + tensor var_14987_end_mask_0 = const()[name = string("op_14987_end_mask_0"), val = tensor([false, true])]; + tensor var_14987_squeeze_mask_0 = const()[name = string("op_14987_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_14987_cast_fp16 = slice_by_index(begin = var_14987_begin_0, end = var_14987_end_0, end_mask = var_14987_end_mask_0, squeeze_mask = var_14987_squeeze_mask_0, x = gumbel)[name = string("op_14987_cast_fp16")]; + tensor var_14990 = const()[name = string("op_14990"), val = tensor([1, 2048])]; + tensor var_14991_cast_fp16 = reshape(shape = var_14990, x = var_14987_cast_fp16)[name = string("op_14991_cast_fp16")]; + tensor noisy_logits_21_cast_fp16 = add(x = masked_logits_21_cast_fp16, y = var_14991_cast_fp16)[name = string("noisy_logits_21_cast_fp16")]; + int32 code_21_axis_0 = const()[name = string("code_21_axis_0"), val = int32(1)]; + bool code_21_keep_dims_0 = const()[name = string("code_21_keep_dims_0"), val = bool(false)]; + string code_21_output_dtype_0 = const()[name = string("code_21_output_dtype_0"), val = string("int32")]; + tensor code_21_cast_fp16 = reduce_argmax(axis = code_21_axis_0, keep_dims = code_21_keep_dims_0, output_dtype = code_21_output_dtype_0, x = noisy_logits_21_cast_fp16)[name = string("code_21_cast_fp16")]; + int32 var_15002 = const()[name = string("op_15002"), val = int32(20480)]; + tensor input_643 = add(x = code_21_cast_fp16, y = var_15002)[name = string("input_643")]; + int32 code_embed_41_axis_0 = const()[name = string("code_embed_41_axis_0"), val = int32(0)]; + int32 code_embed_41_batch_dims_0 = const()[name = string("code_embed_41_batch_dims_0"), val = int32(0)]; + bool code_embed_41_validate_indices_0 = const()[name = string("code_embed_41_validate_indices_0"), val = bool(false)]; + string input_643_to_uint16_dtype_0 = const()[name = string("input_643_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_643_to_uint16 = cast(dtype = input_643_to_uint16_dtype_0, x = input_643)[name = string("cast_4")]; + tensor code_embed_41_cast_fp16_cast_uint16 = gather(axis = code_embed_41_axis_0, batch_dims = code_embed_41_batch_dims_0, indices = input_643_to_uint16, validate_indices = code_embed_41_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_41_cast_fp16_cast_uint16")]; + tensor var_15006 = const()[name = string("op_15006"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_43_cast_fp16 = reshape(shape = var_15006, x = code_embed_41_cast_fp16_cast_uint16)[name = string("code_embed_43_cast_fp16")]; + tensor embed_sum_23_cast_fp16 = add(x = embed_sum_21_cast_fp16, y = code_embed_43_cast_fp16)[name = string("embed_sum_23_cast_fp16")]; + tensor key_cache_121_begin_0 = const()[name = string("key_cache_121_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_121_end_0 = const()[name = string("key_cache_121_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_121_end_mask_0 = const()[name = string("key_cache_121_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_121_cast_fp16 = slice_by_index(begin = key_cache_121_begin_0, end = key_cache_121_end_0, end_mask = key_cache_121_end_mask_0, x = layer_key_caches_25_cast_fp16)[name = string("key_cache_121_cast_fp16")]; + tensor value_cache_121_begin_0 = const()[name = string("value_cache_121_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_121_end_0 = const()[name = string("value_cache_121_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_121_end_mask_0 = const()[name = string("value_cache_121_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_121_cast_fp16 = slice_by_index(begin = value_cache_121_begin_0, end = value_cache_121_end_0, end_mask = value_cache_121_end_mask_0, x = layer_value_caches_25_cast_fp16)[name = string("value_cache_121_cast_fp16")]; + int32 var_15105 = const()[name = string("op_15105"), val = int32(2)]; + int32 var_15109 = const()[name = string("op_15109"), val = int32(3)]; + tensor var_15124_cast_fp16 = mul(x = code_embed_43_cast_fp16, y = code_embed_43_cast_fp16)[name = string("op_15124_cast_fp16")]; + tensor variance_503_axes_0 = const()[name = string("variance_503_axes_0"), val = tensor([1])]; + bool variance_503_keep_dims_0 = const()[name = string("variance_503_keep_dims_0"), val = bool(true)]; + tensor variance_503_cast_fp16 = reduce_mean(axes = variance_503_axes_0, keep_dims = variance_503_keep_dims_0, x = var_15124_cast_fp16)[name = string("variance_503_cast_fp16")]; + fp16 var_15127_to_fp16 = const()[name = string("op_15127_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15128_cast_fp16 = add(x = variance_503_cast_fp16, y = var_15127_to_fp16)[name = string("op_15128_cast_fp16")]; + fp32 var_15129_epsilon_0 = const()[name = string("op_15129_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15129_cast_fp16 = rsqrt(epsilon = var_15129_epsilon_0, x = var_15128_cast_fp16)[name = string("op_15129_cast_fp16")]; + tensor var_15130_cast_fp16 = mul(x = code_embed_43_cast_fp16, y = var_15129_cast_fp16)[name = string("op_15130_cast_fp16")]; + tensor input_645_cast_fp16 = mul(x = var_15130_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_645_cast_fp16")]; + string q_361_pad_type_0 = const()[name = string("q_361_pad_type_0"), val = string("valid")]; + tensor q_361_strides_0 = const()[name = string("q_361_strides_0"), val = tensor([1, 1])]; + tensor q_361_pad_0 = const()[name = string("q_361_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_361_dilations_0 = const()[name = string("q_361_dilations_0"), val = tensor([1, 1])]; + int32 q_361_groups_0 = const()[name = string("q_361_groups_0"), val = int32(1)]; + tensor q_361_cast_fp16 = conv(dilations = q_361_dilations_0, groups = q_361_groups_0, pad = q_361_pad_0, pad_type = q_361_pad_type_0, strides = q_361_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_645_cast_fp16)[name = string("q_361_cast_fp16")]; + string k_361_pad_type_0 = const()[name = string("k_361_pad_type_0"), val = string("valid")]; + tensor k_361_strides_0 = const()[name = string("k_361_strides_0"), val = tensor([1, 1])]; + tensor k_361_pad_0 = const()[name = string("k_361_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_361_dilations_0 = const()[name = string("k_361_dilations_0"), val = tensor([1, 1])]; + int32 k_361_groups_0 = const()[name = string("k_361_groups_0"), val = int32(1)]; + tensor k_361_cast_fp16 = conv(dilations = k_361_dilations_0, groups = k_361_groups_0, pad = k_361_pad_0, pad_type = k_361_pad_type_0, strides = k_361_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_645_cast_fp16)[name = string("k_361_cast_fp16")]; + string v_121_pad_type_0 = const()[name = string("v_121_pad_type_0"), val = string("valid")]; + tensor v_121_strides_0 = const()[name = string("v_121_strides_0"), val = tensor([1, 1])]; + tensor v_121_pad_0 = const()[name = string("v_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_121_dilations_0 = const()[name = string("v_121_dilations_0"), val = tensor([1, 1])]; + int32 v_121_groups_0 = const()[name = string("v_121_groups_0"), val = int32(1)]; + tensor v_121_cast_fp16 = conv(dilations = v_121_dilations_0, groups = v_121_groups_0, pad = v_121_pad_0, pad_type = v_121_pad_type_0, strides = v_121_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_645_cast_fp16)[name = string("v_121_cast_fp16")]; + tensor var_15164 = const()[name = string("op_15164"), val = tensor([16, 128, 1, 1])]; + tensor x_457_cast_fp16 = reshape(shape = var_15164, x = q_361_cast_fp16)[name = string("x_457_cast_fp16")]; + tensor var_15167_cast_fp16 = mul(x = x_457_cast_fp16, y = x_457_cast_fp16)[name = string("op_15167_cast_fp16")]; + tensor variance_505_axes_0 = const()[name = string("variance_505_axes_0"), val = tensor([1])]; + bool variance_505_keep_dims_0 = const()[name = string("variance_505_keep_dims_0"), val = bool(true)]; + tensor variance_505_cast_fp16 = reduce_mean(axes = variance_505_axes_0, keep_dims = variance_505_keep_dims_0, x = var_15167_cast_fp16)[name = string("variance_505_cast_fp16")]; + fp16 var_15170_to_fp16 = const()[name = string("op_15170_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15171_cast_fp16 = add(x = variance_505_cast_fp16, y = var_15170_to_fp16)[name = string("op_15171_cast_fp16")]; + fp32 var_15172_epsilon_0 = const()[name = string("op_15172_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15172_cast_fp16 = rsqrt(epsilon = var_15172_epsilon_0, x = var_15171_cast_fp16)[name = string("op_15172_cast_fp16")]; + tensor var_15173_cast_fp16 = mul(x = x_457_cast_fp16, y = var_15172_cast_fp16)[name = string("op_15173_cast_fp16")]; + tensor q_363_cast_fp16 = mul(x = var_15173_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_363_cast_fp16")]; + tensor var_15175 = const()[name = string("op_15175"), val = tensor([8, 128, 1, 1])]; + tensor x_459_cast_fp16 = reshape(shape = var_15175, x = k_361_cast_fp16)[name = string("x_459_cast_fp16")]; + tensor var_15178_cast_fp16 = mul(x = x_459_cast_fp16, y = x_459_cast_fp16)[name = string("op_15178_cast_fp16")]; + tensor variance_507_axes_0 = const()[name = string("variance_507_axes_0"), val = tensor([1])]; + bool variance_507_keep_dims_0 = const()[name = string("variance_507_keep_dims_0"), val = bool(true)]; + tensor variance_507_cast_fp16 = reduce_mean(axes = variance_507_axes_0, keep_dims = variance_507_keep_dims_0, x = var_15178_cast_fp16)[name = string("variance_507_cast_fp16")]; + fp16 var_15181_to_fp16 = const()[name = string("op_15181_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15182_cast_fp16 = add(x = variance_507_cast_fp16, y = var_15181_to_fp16)[name = string("op_15182_cast_fp16")]; + fp32 var_15183_epsilon_0 = const()[name = string("op_15183_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15183_cast_fp16 = rsqrt(epsilon = var_15183_epsilon_0, x = var_15182_cast_fp16)[name = string("op_15183_cast_fp16")]; + tensor var_15184_cast_fp16 = mul(x = x_459_cast_fp16, y = var_15183_cast_fp16)[name = string("op_15184_cast_fp16")]; + tensor k_363_cast_fp16 = mul(x = var_15184_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_363_cast_fp16")]; + tensor var_15186 = const()[name = string("op_15186"), val = tensor([1, 16, 128, 1])]; + tensor z_241_cast_fp16 = reshape(shape = var_15186, x = q_363_cast_fp16)[name = string("z_241_cast_fp16")]; + tensor var_15188 = const()[name = string("op_15188"), val = tensor([1, 8, 128, 1])]; + tensor z_243_cast_fp16 = reshape(shape = var_15188, x = k_363_cast_fp16)[name = string("z_243_cast_fp16")]; + tensor z1_241_begin_0 = const()[name = string("z1_241_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_241_end_0 = const()[name = string("z1_241_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_241_end_mask_0 = const()[name = string("z1_241_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_241_cast_fp16 = slice_by_index(begin = z1_241_begin_0, end = z1_241_end_0, end_mask = z1_241_end_mask_0, x = z_241_cast_fp16)[name = string("z1_241_cast_fp16")]; + tensor z2_241_begin_0 = const()[name = string("z2_241_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_241_end_0 = const()[name = string("z2_241_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_241_end_mask_0 = const()[name = string("z2_241_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_241_cast_fp16 = slice_by_index(begin = z2_241_begin_0, end = z2_241_end_0, end_mask = z2_241_end_mask_0, x = z_241_cast_fp16)[name = string("z2_241_cast_fp16")]; + tensor cos_121_to_fp16 = const()[name = string("cos_121_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643648)))]; + tensor var_15196_cast_fp16 = mul(x = z_241_cast_fp16, y = cos_121_to_fp16)[name = string("op_15196_cast_fp16")]; + fp16 const_133_promoted_to_fp16 = const()[name = string("const_133_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15197_cast_fp16 = mul(x = z2_241_cast_fp16, y = const_133_promoted_to_fp16)[name = string("op_15197_cast_fp16")]; + bool var_15199_interleave_0 = const()[name = string("op_15199_interleave_0"), val = bool(false)]; + tensor var_15199_cast_fp16 = concat(axis = var_15105, interleave = var_15199_interleave_0, values = (var_15197_cast_fp16, z1_241_cast_fp16))[name = string("op_15199_cast_fp16")]; + tensor sin_121_to_fp16 = const()[name = string("sin_121_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141643968)))]; + tensor var_15200_cast_fp16 = mul(x = var_15199_cast_fp16, y = sin_121_to_fp16)[name = string("op_15200_cast_fp16")]; + tensor q_365_cast_fp16 = add(x = var_15196_cast_fp16, y = var_15200_cast_fp16)[name = string("q_365_cast_fp16")]; + tensor z1_243_begin_0 = const()[name = string("z1_243_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_243_end_0 = const()[name = string("z1_243_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_243_end_mask_0 = const()[name = string("z1_243_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_243_cast_fp16 = slice_by_index(begin = z1_243_begin_0, end = z1_243_end_0, end_mask = z1_243_end_mask_0, x = z_243_cast_fp16)[name = string("z1_243_cast_fp16")]; + tensor z2_243_begin_0 = const()[name = string("z2_243_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_243_end_0 = const()[name = string("z2_243_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_243_end_mask_0 = const()[name = string("z2_243_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_243_cast_fp16 = slice_by_index(begin = z2_243_begin_0, end = z2_243_end_0, end_mask = z2_243_end_mask_0, x = z_243_cast_fp16)[name = string("z2_243_cast_fp16")]; + tensor var_15208_cast_fp16 = mul(x = z_243_cast_fp16, y = cos_121_to_fp16)[name = string("op_15208_cast_fp16")]; + fp16 const_134_promoted_to_fp16 = const()[name = string("const_134_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15209_cast_fp16 = mul(x = z2_243_cast_fp16, y = const_134_promoted_to_fp16)[name = string("op_15209_cast_fp16")]; + bool var_15211_interleave_0 = const()[name = string("op_15211_interleave_0"), val = bool(false)]; + tensor var_15211_cast_fp16 = concat(axis = var_15105, interleave = var_15211_interleave_0, values = (var_15209_cast_fp16, z1_243_cast_fp16))[name = string("op_15211_cast_fp16")]; + tensor var_15212_cast_fp16 = mul(x = var_15211_cast_fp16, y = sin_121_to_fp16)[name = string("op_15212_cast_fp16")]; + tensor k_365_cast_fp16 = add(x = var_15208_cast_fp16, y = var_15212_cast_fp16)[name = string("k_365_cast_fp16")]; + tensor var_15214 = const()[name = string("op_15214"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_121_cast_fp16 = reshape(shape = var_15214, x = k_365_cast_fp16)[name = string("cur_key_121_cast_fp16")]; + tensor var_15216_to_fp16 = const()[name = string("op_15216_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644288)))]; + tensor var_15217_cast_fp16 = mul(x = key_cache_121_cast_fp16, y = var_15216_to_fp16)[name = string("op_15217_cast_fp16")]; + tensor upd_121_to_fp16 = const()[name = string("upd_121_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644416)))]; + tensor var_15218_cast_fp16 = mul(x = cur_key_121_cast_fp16, y = upd_121_to_fp16)[name = string("op_15218_cast_fp16")]; + tensor key_121_cast_fp16 = add(x = var_15217_cast_fp16, y = var_15218_cast_fp16)[name = string("key_121_cast_fp16")]; + tensor var_15220_to_fp16 = const()[name = string("op_15220_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644288)))]; + tensor var_15221_cast_fp16 = mul(x = value_cache_121_cast_fp16, y = var_15220_to_fp16)[name = string("op_15221_cast_fp16")]; + tensor var_15222_cast_fp16 = mul(x = v_121_cast_fp16, y = upd_121_to_fp16)[name = string("op_15222_cast_fp16")]; + tensor value_121_cast_fp16 = add(x = var_15221_cast_fp16, y = var_15222_cast_fp16)[name = string("value_121_cast_fp16")]; + tensor var_15224 = const()[name = string("op_15224"), val = tensor([1, 8, 128, 16])]; + tensor kh_241_cast_fp16 = reshape(shape = var_15224, x = key_121_cast_fp16)[name = string("kh_241_cast_fp16")]; + tensor var_15226 = const()[name = string("op_15226"), val = tensor([1, 8, 128, 16])]; + tensor vh_241_cast_fp16 = reshape(shape = var_15226, x = value_121_cast_fp16)[name = string("vh_241_cast_fp16")]; + tensor transpose_240_perm_0 = const()[name = string("transpose_240_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_120_reps_0 = const()[name = string("tile_120_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_240_cast_fp16 = transpose(perm = transpose_240_perm_0, x = kh_241_cast_fp16)[name = string("transpose_119")]; + tensor tile_120_cast_fp16 = tile(reps = tile_120_reps_0, x = transpose_240_cast_fp16)[name = string("tile_120_cast_fp16")]; + tensor concat_303 = const()[name = string("concat_303"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_240_cast_fp16 = reshape(shape = concat_303, x = tile_120_cast_fp16)[name = string("reshape_240_cast_fp16")]; + tensor transpose_241_perm_0 = const()[name = string("transpose_241_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_304 = const()[name = string("concat_304"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_241_cast_fp16 = transpose(perm = transpose_241_perm_0, x = reshape_240_cast_fp16)[name = string("transpose_118")]; + tensor reshape_241_cast_fp16 = reshape(shape = concat_304, x = transpose_241_cast_fp16)[name = string("reshape_241_cast_fp16")]; + tensor transpose_242_perm_0 = const()[name = string("transpose_242_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_121_reps_0 = const()[name = string("tile_121_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_242_cast_fp16 = transpose(perm = transpose_242_perm_0, x = vh_241_cast_fp16)[name = string("transpose_117")]; + tensor tile_121_cast_fp16 = tile(reps = tile_121_reps_0, x = transpose_242_cast_fp16)[name = string("tile_121_cast_fp16")]; + tensor concat_305 = const()[name = string("concat_305"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_242_cast_fp16 = reshape(shape = concat_305, x = tile_121_cast_fp16)[name = string("reshape_242_cast_fp16")]; + tensor transpose_243_perm_0 = const()[name = string("transpose_243_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_306 = const()[name = string("concat_306"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_243_cast_fp16 = transpose(perm = transpose_243_perm_0, x = reshape_242_cast_fp16)[name = string("transpose_116")]; + tensor reshape_243_cast_fp16 = reshape(shape = concat_306, x = transpose_243_cast_fp16)[name = string("reshape_243_cast_fp16")]; + fp16 var_15230_to_fp16 = const()[name = string("op_15230_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_15231_cast_fp16 = mul(x = q_365_cast_fp16, y = var_15230_to_fp16)[name = string("op_15231_cast_fp16")]; + tensor transpose_557_perm_0 = const()[name = string("transpose_557_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_263_transpose_x_1 = const()[name = string("w_263_transpose_x_1"), val = bool(true)]; + bool w_263_transpose_y_1 = const()[name = string("w_263_transpose_y_1"), val = bool(false)]; + tensor transpose_557_cast_fp16 = transpose(perm = transpose_557_perm_0, x = reshape_241_cast_fp16)[name = string("transpose_115")]; + tensor w_263_cast_fp16 = matmul(transpose_x = w_263_transpose_x_1, transpose_y = w_263_transpose_y_1, x = var_15231_cast_fp16, y = transpose_557_cast_fp16)[name = string("w_263_cast_fp16")]; + tensor pad_121_to_fp16 = const()[name = string("pad_121_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644544)))]; + tensor var_15234_cast_fp16 = add(x = w_263_cast_fp16, y = pad_121_to_fp16)[name = string("op_15234_cast_fp16")]; + tensor w_265_cast_fp16 = softmax(axis = var_15109, x = var_15234_cast_fp16)[name = string("w_265_cast_fp16")]; + tensor transpose_558_perm_0 = const()[name = string("transpose_558_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_121_transpose_x_1 = const()[name = string("attn_121_transpose_x_1"), val = bool(false)]; + bool attn_121_transpose_y_1 = const()[name = string("attn_121_transpose_y_1"), val = bool(true)]; + tensor transpose_558_cast_fp16 = transpose(perm = transpose_558_perm_0, x = reshape_243_cast_fp16)[name = string("transpose_114")]; + tensor attn_121_cast_fp16 = matmul(transpose_x = attn_121_transpose_x_1, transpose_y = attn_121_transpose_y_1, x = transpose_558_cast_fp16, y = w_265_cast_fp16)[name = string("attn_121_cast_fp16")]; + tensor var_15238 = const()[name = string("op_15238"), val = tensor([1, 2048, 1, 1])]; + tensor input_647_cast_fp16 = reshape(shape = var_15238, x = attn_121_cast_fp16)[name = string("input_647_cast_fp16")]; + string attn_output_121_pad_type_0 = const()[name = string("attn_output_121_pad_type_0"), val = string("valid")]; + tensor attn_output_121_strides_0 = const()[name = string("attn_output_121_strides_0"), val = tensor([1, 1])]; + tensor attn_output_121_pad_0 = const()[name = string("attn_output_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_121_dilations_0 = const()[name = string("attn_output_121_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_121_groups_0 = const()[name = string("attn_output_121_groups_0"), val = int32(1)]; + tensor attn_output_121_cast_fp16 = conv(dilations = attn_output_121_dilations_0, groups = attn_output_121_groups_0, pad = attn_output_121_pad_0, pad_type = attn_output_121_pad_type_0, strides = attn_output_121_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_647_cast_fp16)[name = string("attn_output_121_cast_fp16")]; + tensor x_461_cast_fp16 = add(x = code_embed_43_cast_fp16, y = attn_output_121_cast_fp16)[name = string("x_461_cast_fp16")]; + tensor var_15252_cast_fp16 = mul(x = x_461_cast_fp16, y = x_461_cast_fp16)[name = string("op_15252_cast_fp16")]; + tensor variance_509_axes_0 = const()[name = string("variance_509_axes_0"), val = tensor([1])]; + bool variance_509_keep_dims_0 = const()[name = string("variance_509_keep_dims_0"), val = bool(true)]; + tensor variance_509_cast_fp16 = reduce_mean(axes = variance_509_axes_0, keep_dims = variance_509_keep_dims_0, x = var_15252_cast_fp16)[name = string("variance_509_cast_fp16")]; + fp16 var_15255_to_fp16 = const()[name = string("op_15255_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15256_cast_fp16 = add(x = variance_509_cast_fp16, y = var_15255_to_fp16)[name = string("op_15256_cast_fp16")]; + fp32 var_15257_epsilon_0 = const()[name = string("op_15257_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15257_cast_fp16 = rsqrt(epsilon = var_15257_epsilon_0, x = var_15256_cast_fp16)[name = string("op_15257_cast_fp16")]; + tensor var_15258_cast_fp16 = mul(x = x_461_cast_fp16, y = var_15257_cast_fp16)[name = string("op_15258_cast_fp16")]; + tensor input_649_cast_fp16 = mul(x = var_15258_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_649_cast_fp16")]; + string input_651_pad_type_0 = const()[name = string("input_651_pad_type_0"), val = string("valid")]; + tensor input_651_strides_0 = const()[name = string("input_651_strides_0"), val = tensor([1, 1])]; + tensor input_651_pad_0 = const()[name = string("input_651_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_651_dilations_0 = const()[name = string("input_651_dilations_0"), val = tensor([1, 1])]; + int32 input_651_groups_0 = const()[name = string("input_651_groups_0"), val = int32(1)]; + tensor input_651_cast_fp16 = conv(dilations = input_651_dilations_0, groups = input_651_groups_0, pad = input_651_pad_0, pad_type = input_651_pad_type_0, strides = input_651_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_649_cast_fp16)[name = string("input_651_cast_fp16")]; + tensor var_15266_cast_fp16 = silu(x = input_651_cast_fp16)[name = string("op_15266_cast_fp16")]; + string var_15272_pad_type_0 = const()[name = string("op_15272_pad_type_0"), val = string("valid")]; + tensor var_15272_strides_0 = const()[name = string("op_15272_strides_0"), val = tensor([1, 1])]; + tensor var_15272_pad_0 = const()[name = string("op_15272_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15272_dilations_0 = const()[name = string("op_15272_dilations_0"), val = tensor([1, 1])]; + int32 var_15272_groups_0 = const()[name = string("op_15272_groups_0"), val = int32(1)]; + tensor var_15272_cast_fp16 = conv(dilations = var_15272_dilations_0, groups = var_15272_groups_0, pad = var_15272_pad_0, pad_type = var_15272_pad_type_0, strides = var_15272_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_649_cast_fp16)[name = string("op_15272_cast_fp16")]; + tensor input_653_cast_fp16 = mul(x = var_15266_cast_fp16, y = var_15272_cast_fp16)[name = string("input_653_cast_fp16")]; + string h_121_pad_type_0 = const()[name = string("h_121_pad_type_0"), val = string("valid")]; + tensor h_121_strides_0 = const()[name = string("h_121_strides_0"), val = tensor([1, 1])]; + tensor h_121_pad_0 = const()[name = string("h_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_121_dilations_0 = const()[name = string("h_121_dilations_0"), val = tensor([1, 1])]; + int32 h_121_groups_0 = const()[name = string("h_121_groups_0"), val = int32(1)]; + tensor h_121_cast_fp16 = conv(dilations = h_121_dilations_0, groups = h_121_groups_0, pad = h_121_pad_0, pad_type = h_121_pad_type_0, strides = h_121_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_653_cast_fp16)[name = string("h_121_cast_fp16")]; + tensor x_463_cast_fp16 = add(x = x_461_cast_fp16, y = h_121_cast_fp16)[name = string("x_463_cast_fp16")]; + tensor key_cache_123_begin_0 = const()[name = string("key_cache_123_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_123_end_0 = const()[name = string("key_cache_123_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_123_end_mask_0 = const()[name = string("key_cache_123_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_123_cast_fp16 = slice_by_index(begin = key_cache_123_begin_0, end = key_cache_123_end_0, end_mask = key_cache_123_end_mask_0, x = layer_key_caches_25_cast_fp16)[name = string("key_cache_123_cast_fp16")]; + tensor value_cache_123_begin_0 = const()[name = string("value_cache_123_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_123_end_0 = const()[name = string("value_cache_123_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_123_end_mask_0 = const()[name = string("value_cache_123_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_123_cast_fp16 = slice_by_index(begin = value_cache_123_begin_0, end = value_cache_123_end_0, end_mask = value_cache_123_end_mask_0, x = layer_value_caches_25_cast_fp16)[name = string("value_cache_123_cast_fp16")]; + int32 var_15325 = const()[name = string("op_15325"), val = int32(2)]; + int32 var_15329 = const()[name = string("op_15329"), val = int32(3)]; + tensor var_15344_cast_fp16 = mul(x = x_463_cast_fp16, y = x_463_cast_fp16)[name = string("op_15344_cast_fp16")]; + tensor variance_511_axes_0 = const()[name = string("variance_511_axes_0"), val = tensor([1])]; + bool variance_511_keep_dims_0 = const()[name = string("variance_511_keep_dims_0"), val = bool(true)]; + tensor variance_511_cast_fp16 = reduce_mean(axes = variance_511_axes_0, keep_dims = variance_511_keep_dims_0, x = var_15344_cast_fp16)[name = string("variance_511_cast_fp16")]; + fp16 var_15347_to_fp16 = const()[name = string("op_15347_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15348_cast_fp16 = add(x = variance_511_cast_fp16, y = var_15347_to_fp16)[name = string("op_15348_cast_fp16")]; + fp32 var_15349_epsilon_0 = const()[name = string("op_15349_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15349_cast_fp16 = rsqrt(epsilon = var_15349_epsilon_0, x = var_15348_cast_fp16)[name = string("op_15349_cast_fp16")]; + tensor var_15350_cast_fp16 = mul(x = x_463_cast_fp16, y = var_15349_cast_fp16)[name = string("op_15350_cast_fp16")]; + tensor input_655_cast_fp16 = mul(x = var_15350_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_655_cast_fp16")]; + string q_367_pad_type_0 = const()[name = string("q_367_pad_type_0"), val = string("valid")]; + tensor q_367_strides_0 = const()[name = string("q_367_strides_0"), val = tensor([1, 1])]; + tensor q_367_pad_0 = const()[name = string("q_367_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_367_dilations_0 = const()[name = string("q_367_dilations_0"), val = tensor([1, 1])]; + int32 q_367_groups_0 = const()[name = string("q_367_groups_0"), val = int32(1)]; + tensor q_367_cast_fp16 = conv(dilations = q_367_dilations_0, groups = q_367_groups_0, pad = q_367_pad_0, pad_type = q_367_pad_type_0, strides = q_367_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_655_cast_fp16)[name = string("q_367_cast_fp16")]; + string k_367_pad_type_0 = const()[name = string("k_367_pad_type_0"), val = string("valid")]; + tensor k_367_strides_0 = const()[name = string("k_367_strides_0"), val = tensor([1, 1])]; + tensor k_367_pad_0 = const()[name = string("k_367_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_367_dilations_0 = const()[name = string("k_367_dilations_0"), val = tensor([1, 1])]; + int32 k_367_groups_0 = const()[name = string("k_367_groups_0"), val = int32(1)]; + tensor k_367_cast_fp16 = conv(dilations = k_367_dilations_0, groups = k_367_groups_0, pad = k_367_pad_0, pad_type = k_367_pad_type_0, strides = k_367_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_655_cast_fp16)[name = string("k_367_cast_fp16")]; + string v_123_pad_type_0 = const()[name = string("v_123_pad_type_0"), val = string("valid")]; + tensor v_123_strides_0 = const()[name = string("v_123_strides_0"), val = tensor([1, 1])]; + tensor v_123_pad_0 = const()[name = string("v_123_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_123_dilations_0 = const()[name = string("v_123_dilations_0"), val = tensor([1, 1])]; + int32 v_123_groups_0 = const()[name = string("v_123_groups_0"), val = int32(1)]; + tensor v_123_cast_fp16 = conv(dilations = v_123_dilations_0, groups = v_123_groups_0, pad = v_123_pad_0, pad_type = v_123_pad_type_0, strides = v_123_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_655_cast_fp16)[name = string("v_123_cast_fp16")]; + tensor var_15384 = const()[name = string("op_15384"), val = tensor([16, 128, 1, 1])]; + tensor x_465_cast_fp16 = reshape(shape = var_15384, x = q_367_cast_fp16)[name = string("x_465_cast_fp16")]; + tensor var_15387_cast_fp16 = mul(x = x_465_cast_fp16, y = x_465_cast_fp16)[name = string("op_15387_cast_fp16")]; + tensor variance_513_axes_0 = const()[name = string("variance_513_axes_0"), val = tensor([1])]; + bool variance_513_keep_dims_0 = const()[name = string("variance_513_keep_dims_0"), val = bool(true)]; + tensor variance_513_cast_fp16 = reduce_mean(axes = variance_513_axes_0, keep_dims = variance_513_keep_dims_0, x = var_15387_cast_fp16)[name = string("variance_513_cast_fp16")]; + fp16 var_15390_to_fp16 = const()[name = string("op_15390_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15391_cast_fp16 = add(x = variance_513_cast_fp16, y = var_15390_to_fp16)[name = string("op_15391_cast_fp16")]; + fp32 var_15392_epsilon_0 = const()[name = string("op_15392_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15392_cast_fp16 = rsqrt(epsilon = var_15392_epsilon_0, x = var_15391_cast_fp16)[name = string("op_15392_cast_fp16")]; + tensor var_15393_cast_fp16 = mul(x = x_465_cast_fp16, y = var_15392_cast_fp16)[name = string("op_15393_cast_fp16")]; + tensor q_369_cast_fp16 = mul(x = var_15393_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_369_cast_fp16")]; + tensor var_15395 = const()[name = string("op_15395"), val = tensor([8, 128, 1, 1])]; + tensor x_467_cast_fp16 = reshape(shape = var_15395, x = k_367_cast_fp16)[name = string("x_467_cast_fp16")]; + tensor var_15398_cast_fp16 = mul(x = x_467_cast_fp16, y = x_467_cast_fp16)[name = string("op_15398_cast_fp16")]; + tensor variance_515_axes_0 = const()[name = string("variance_515_axes_0"), val = tensor([1])]; + bool variance_515_keep_dims_0 = const()[name = string("variance_515_keep_dims_0"), val = bool(true)]; + tensor variance_515_cast_fp16 = reduce_mean(axes = variance_515_axes_0, keep_dims = variance_515_keep_dims_0, x = var_15398_cast_fp16)[name = string("variance_515_cast_fp16")]; + fp16 var_15401_to_fp16 = const()[name = string("op_15401_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15402_cast_fp16 = add(x = variance_515_cast_fp16, y = var_15401_to_fp16)[name = string("op_15402_cast_fp16")]; + fp32 var_15403_epsilon_0 = const()[name = string("op_15403_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15403_cast_fp16 = rsqrt(epsilon = var_15403_epsilon_0, x = var_15402_cast_fp16)[name = string("op_15403_cast_fp16")]; + tensor var_15404_cast_fp16 = mul(x = x_467_cast_fp16, y = var_15403_cast_fp16)[name = string("op_15404_cast_fp16")]; + tensor k_369_cast_fp16 = mul(x = var_15404_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_369_cast_fp16")]; + tensor var_15406 = const()[name = string("op_15406"), val = tensor([1, 16, 128, 1])]; + tensor z_245_cast_fp16 = reshape(shape = var_15406, x = q_369_cast_fp16)[name = string("z_245_cast_fp16")]; + tensor var_15408 = const()[name = string("op_15408"), val = tensor([1, 8, 128, 1])]; + tensor z_247_cast_fp16 = reshape(shape = var_15408, x = k_369_cast_fp16)[name = string("z_247_cast_fp16")]; + tensor z1_245_begin_0 = const()[name = string("z1_245_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_245_end_0 = const()[name = string("z1_245_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_245_end_mask_0 = const()[name = string("z1_245_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_245_cast_fp16 = slice_by_index(begin = z1_245_begin_0, end = z1_245_end_0, end_mask = z1_245_end_mask_0, x = z_245_cast_fp16)[name = string("z1_245_cast_fp16")]; + tensor z2_245_begin_0 = const()[name = string("z2_245_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_245_end_0 = const()[name = string("z2_245_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_245_end_mask_0 = const()[name = string("z2_245_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_245_cast_fp16 = slice_by_index(begin = z2_245_begin_0, end = z2_245_end_0, end_mask = z2_245_end_mask_0, x = z_245_cast_fp16)[name = string("z2_245_cast_fp16")]; + tensor var_15416_cast_fp16 = mul(x = z_245_cast_fp16, y = cos_121_to_fp16)[name = string("op_15416_cast_fp16")]; + fp16 const_135_promoted_to_fp16 = const()[name = string("const_135_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15417_cast_fp16 = mul(x = z2_245_cast_fp16, y = const_135_promoted_to_fp16)[name = string("op_15417_cast_fp16")]; + bool var_15419_interleave_0 = const()[name = string("op_15419_interleave_0"), val = bool(false)]; + tensor var_15419_cast_fp16 = concat(axis = var_15325, interleave = var_15419_interleave_0, values = (var_15417_cast_fp16, z1_245_cast_fp16))[name = string("op_15419_cast_fp16")]; + tensor var_15420_cast_fp16 = mul(x = var_15419_cast_fp16, y = sin_121_to_fp16)[name = string("op_15420_cast_fp16")]; + tensor q_371_cast_fp16 = add(x = var_15416_cast_fp16, y = var_15420_cast_fp16)[name = string("q_371_cast_fp16")]; + tensor z1_247_begin_0 = const()[name = string("z1_247_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_247_end_0 = const()[name = string("z1_247_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_247_end_mask_0 = const()[name = string("z1_247_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_247_cast_fp16 = slice_by_index(begin = z1_247_begin_0, end = z1_247_end_0, end_mask = z1_247_end_mask_0, x = z_247_cast_fp16)[name = string("z1_247_cast_fp16")]; + tensor z2_247_begin_0 = const()[name = string("z2_247_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_247_end_0 = const()[name = string("z2_247_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_247_end_mask_0 = const()[name = string("z2_247_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_247_cast_fp16 = slice_by_index(begin = z2_247_begin_0, end = z2_247_end_0, end_mask = z2_247_end_mask_0, x = z_247_cast_fp16)[name = string("z2_247_cast_fp16")]; + tensor var_15428_cast_fp16 = mul(x = z_247_cast_fp16, y = cos_121_to_fp16)[name = string("op_15428_cast_fp16")]; + fp16 const_136_promoted_to_fp16 = const()[name = string("const_136_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15429_cast_fp16 = mul(x = z2_247_cast_fp16, y = const_136_promoted_to_fp16)[name = string("op_15429_cast_fp16")]; + bool var_15431_interleave_0 = const()[name = string("op_15431_interleave_0"), val = bool(false)]; + tensor var_15431_cast_fp16 = concat(axis = var_15325, interleave = var_15431_interleave_0, values = (var_15429_cast_fp16, z1_247_cast_fp16))[name = string("op_15431_cast_fp16")]; + tensor var_15432_cast_fp16 = mul(x = var_15431_cast_fp16, y = sin_121_to_fp16)[name = string("op_15432_cast_fp16")]; + tensor k_371_cast_fp16 = add(x = var_15428_cast_fp16, y = var_15432_cast_fp16)[name = string("k_371_cast_fp16")]; + tensor var_15434 = const()[name = string("op_15434"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_123_cast_fp16 = reshape(shape = var_15434, x = k_371_cast_fp16)[name = string("cur_key_123_cast_fp16")]; + tensor var_15436_to_fp16 = const()[name = string("op_15436_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644288)))]; + tensor var_15437_cast_fp16 = mul(x = key_cache_123_cast_fp16, y = var_15436_to_fp16)[name = string("op_15437_cast_fp16")]; + tensor upd_123_to_fp16 = const()[name = string("upd_123_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644416)))]; + tensor var_15438_cast_fp16 = mul(x = cur_key_123_cast_fp16, y = upd_123_to_fp16)[name = string("op_15438_cast_fp16")]; + tensor key_123_cast_fp16 = add(x = var_15437_cast_fp16, y = var_15438_cast_fp16)[name = string("key_123_cast_fp16")]; + tensor var_15440_to_fp16 = const()[name = string("op_15440_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644288)))]; + tensor var_15441_cast_fp16 = mul(x = value_cache_123_cast_fp16, y = var_15440_to_fp16)[name = string("op_15441_cast_fp16")]; + tensor var_15442_cast_fp16 = mul(x = v_123_cast_fp16, y = upd_123_to_fp16)[name = string("op_15442_cast_fp16")]; + tensor value_123_cast_fp16 = add(x = var_15441_cast_fp16, y = var_15442_cast_fp16)[name = string("value_123_cast_fp16")]; + tensor var_15444 = const()[name = string("op_15444"), val = tensor([1, 8, 128, 16])]; + tensor kh_245_cast_fp16 = reshape(shape = var_15444, x = key_123_cast_fp16)[name = string("kh_245_cast_fp16")]; + tensor var_15446 = const()[name = string("op_15446"), val = tensor([1, 8, 128, 16])]; + tensor vh_245_cast_fp16 = reshape(shape = var_15446, x = value_123_cast_fp16)[name = string("vh_245_cast_fp16")]; + tensor transpose_244_perm_0 = const()[name = string("transpose_244_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_122_reps_0 = const()[name = string("tile_122_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_244_cast_fp16 = transpose(perm = transpose_244_perm_0, x = kh_245_cast_fp16)[name = string("transpose_113")]; + tensor tile_122_cast_fp16 = tile(reps = tile_122_reps_0, x = transpose_244_cast_fp16)[name = string("tile_122_cast_fp16")]; + tensor concat_307 = const()[name = string("concat_307"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_244_cast_fp16 = reshape(shape = concat_307, x = tile_122_cast_fp16)[name = string("reshape_244_cast_fp16")]; + tensor transpose_245_perm_0 = const()[name = string("transpose_245_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_308 = const()[name = string("concat_308"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_245_cast_fp16 = transpose(perm = transpose_245_perm_0, x = reshape_244_cast_fp16)[name = string("transpose_112")]; + tensor reshape_245_cast_fp16 = reshape(shape = concat_308, x = transpose_245_cast_fp16)[name = string("reshape_245_cast_fp16")]; + tensor transpose_246_perm_0 = const()[name = string("transpose_246_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_123_reps_0 = const()[name = string("tile_123_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_246_cast_fp16 = transpose(perm = transpose_246_perm_0, x = vh_245_cast_fp16)[name = string("transpose_111")]; + tensor tile_123_cast_fp16 = tile(reps = tile_123_reps_0, x = transpose_246_cast_fp16)[name = string("tile_123_cast_fp16")]; + tensor concat_309 = const()[name = string("concat_309"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_246_cast_fp16 = reshape(shape = concat_309, x = tile_123_cast_fp16)[name = string("reshape_246_cast_fp16")]; + tensor transpose_247_perm_0 = const()[name = string("transpose_247_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_310 = const()[name = string("concat_310"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_247_cast_fp16 = transpose(perm = transpose_247_perm_0, x = reshape_246_cast_fp16)[name = string("transpose_110")]; + tensor reshape_247_cast_fp16 = reshape(shape = concat_310, x = transpose_247_cast_fp16)[name = string("reshape_247_cast_fp16")]; + fp16 var_15450_to_fp16 = const()[name = string("op_15450_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_15451_cast_fp16 = mul(x = q_371_cast_fp16, y = var_15450_to_fp16)[name = string("op_15451_cast_fp16")]; + tensor transpose_561_perm_0 = const()[name = string("transpose_561_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_267_transpose_x_1 = const()[name = string("w_267_transpose_x_1"), val = bool(true)]; + bool w_267_transpose_y_1 = const()[name = string("w_267_transpose_y_1"), val = bool(false)]; + tensor transpose_561_cast_fp16 = transpose(perm = transpose_561_perm_0, x = reshape_245_cast_fp16)[name = string("transpose_109")]; + tensor w_267_cast_fp16 = matmul(transpose_x = w_267_transpose_x_1, transpose_y = w_267_transpose_y_1, x = var_15451_cast_fp16, y = transpose_561_cast_fp16)[name = string("w_267_cast_fp16")]; + tensor pad_123_to_fp16 = const()[name = string("pad_123_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644544)))]; + tensor var_15454_cast_fp16 = add(x = w_267_cast_fp16, y = pad_123_to_fp16)[name = string("op_15454_cast_fp16")]; + tensor w_269_cast_fp16 = softmax(axis = var_15329, x = var_15454_cast_fp16)[name = string("w_269_cast_fp16")]; + tensor transpose_562_perm_0 = const()[name = string("transpose_562_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_123_transpose_x_1 = const()[name = string("attn_123_transpose_x_1"), val = bool(false)]; + bool attn_123_transpose_y_1 = const()[name = string("attn_123_transpose_y_1"), val = bool(true)]; + tensor transpose_562_cast_fp16 = transpose(perm = transpose_562_perm_0, x = reshape_247_cast_fp16)[name = string("transpose_108")]; + tensor attn_123_cast_fp16 = matmul(transpose_x = attn_123_transpose_x_1, transpose_y = attn_123_transpose_y_1, x = transpose_562_cast_fp16, y = w_269_cast_fp16)[name = string("attn_123_cast_fp16")]; + tensor var_15458 = const()[name = string("op_15458"), val = tensor([1, 2048, 1, 1])]; + tensor input_657_cast_fp16 = reshape(shape = var_15458, x = attn_123_cast_fp16)[name = string("input_657_cast_fp16")]; + string attn_output_123_pad_type_0 = const()[name = string("attn_output_123_pad_type_0"), val = string("valid")]; + tensor attn_output_123_strides_0 = const()[name = string("attn_output_123_strides_0"), val = tensor([1, 1])]; + tensor attn_output_123_pad_0 = const()[name = string("attn_output_123_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_123_dilations_0 = const()[name = string("attn_output_123_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_123_groups_0 = const()[name = string("attn_output_123_groups_0"), val = int32(1)]; + tensor attn_output_123_cast_fp16 = conv(dilations = attn_output_123_dilations_0, groups = attn_output_123_groups_0, pad = attn_output_123_pad_0, pad_type = attn_output_123_pad_type_0, strides = attn_output_123_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_657_cast_fp16)[name = string("attn_output_123_cast_fp16")]; + tensor x_469_cast_fp16 = add(x = x_463_cast_fp16, y = attn_output_123_cast_fp16)[name = string("x_469_cast_fp16")]; + tensor var_15472_cast_fp16 = mul(x = x_469_cast_fp16, y = x_469_cast_fp16)[name = string("op_15472_cast_fp16")]; + tensor variance_517_axes_0 = const()[name = string("variance_517_axes_0"), val = tensor([1])]; + bool variance_517_keep_dims_0 = const()[name = string("variance_517_keep_dims_0"), val = bool(true)]; + tensor variance_517_cast_fp16 = reduce_mean(axes = variance_517_axes_0, keep_dims = variance_517_keep_dims_0, x = var_15472_cast_fp16)[name = string("variance_517_cast_fp16")]; + fp16 var_15475_to_fp16 = const()[name = string("op_15475_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15476_cast_fp16 = add(x = variance_517_cast_fp16, y = var_15475_to_fp16)[name = string("op_15476_cast_fp16")]; + fp32 var_15477_epsilon_0 = const()[name = string("op_15477_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15477_cast_fp16 = rsqrt(epsilon = var_15477_epsilon_0, x = var_15476_cast_fp16)[name = string("op_15477_cast_fp16")]; + tensor var_15478_cast_fp16 = mul(x = x_469_cast_fp16, y = var_15477_cast_fp16)[name = string("op_15478_cast_fp16")]; + tensor input_659_cast_fp16 = mul(x = var_15478_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_659_cast_fp16")]; + string input_661_pad_type_0 = const()[name = string("input_661_pad_type_0"), val = string("valid")]; + tensor input_661_strides_0 = const()[name = string("input_661_strides_0"), val = tensor([1, 1])]; + tensor input_661_pad_0 = const()[name = string("input_661_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_661_dilations_0 = const()[name = string("input_661_dilations_0"), val = tensor([1, 1])]; + int32 input_661_groups_0 = const()[name = string("input_661_groups_0"), val = int32(1)]; + tensor input_661_cast_fp16 = conv(dilations = input_661_dilations_0, groups = input_661_groups_0, pad = input_661_pad_0, pad_type = input_661_pad_type_0, strides = input_661_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_659_cast_fp16)[name = string("input_661_cast_fp16")]; + tensor var_15486_cast_fp16 = silu(x = input_661_cast_fp16)[name = string("op_15486_cast_fp16")]; + string var_15492_pad_type_0 = const()[name = string("op_15492_pad_type_0"), val = string("valid")]; + tensor var_15492_strides_0 = const()[name = string("op_15492_strides_0"), val = tensor([1, 1])]; + tensor var_15492_pad_0 = const()[name = string("op_15492_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15492_dilations_0 = const()[name = string("op_15492_dilations_0"), val = tensor([1, 1])]; + int32 var_15492_groups_0 = const()[name = string("op_15492_groups_0"), val = int32(1)]; + tensor var_15492_cast_fp16 = conv(dilations = var_15492_dilations_0, groups = var_15492_groups_0, pad = var_15492_pad_0, pad_type = var_15492_pad_type_0, strides = var_15492_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_659_cast_fp16)[name = string("op_15492_cast_fp16")]; + tensor input_663_cast_fp16 = mul(x = var_15486_cast_fp16, y = var_15492_cast_fp16)[name = string("input_663_cast_fp16")]; + string h_123_pad_type_0 = const()[name = string("h_123_pad_type_0"), val = string("valid")]; + tensor h_123_strides_0 = const()[name = string("h_123_strides_0"), val = tensor([1, 1])]; + tensor h_123_pad_0 = const()[name = string("h_123_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_123_dilations_0 = const()[name = string("h_123_dilations_0"), val = tensor([1, 1])]; + int32 h_123_groups_0 = const()[name = string("h_123_groups_0"), val = int32(1)]; + tensor h_123_cast_fp16 = conv(dilations = h_123_dilations_0, groups = h_123_groups_0, pad = h_123_pad_0, pad_type = h_123_pad_type_0, strides = h_123_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_663_cast_fp16)[name = string("h_123_cast_fp16")]; + tensor x_471_cast_fp16 = add(x = x_469_cast_fp16, y = h_123_cast_fp16)[name = string("x_471_cast_fp16")]; + tensor key_cache_125_begin_0 = const()[name = string("key_cache_125_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_125_end_0 = const()[name = string("key_cache_125_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_125_end_mask_0 = const()[name = string("key_cache_125_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_125_cast_fp16 = slice_by_index(begin = key_cache_125_begin_0, end = key_cache_125_end_0, end_mask = key_cache_125_end_mask_0, x = layer_key_caches_25_cast_fp16)[name = string("key_cache_125_cast_fp16")]; + tensor value_cache_125_begin_0 = const()[name = string("value_cache_125_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_125_end_0 = const()[name = string("value_cache_125_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_125_end_mask_0 = const()[name = string("value_cache_125_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_125_cast_fp16 = slice_by_index(begin = value_cache_125_begin_0, end = value_cache_125_end_0, end_mask = value_cache_125_end_mask_0, x = layer_value_caches_25_cast_fp16)[name = string("value_cache_125_cast_fp16")]; + int32 var_15545 = const()[name = string("op_15545"), val = int32(2)]; + int32 var_15549 = const()[name = string("op_15549"), val = int32(3)]; + tensor var_15564_cast_fp16 = mul(x = x_471_cast_fp16, y = x_471_cast_fp16)[name = string("op_15564_cast_fp16")]; + tensor variance_519_axes_0 = const()[name = string("variance_519_axes_0"), val = tensor([1])]; + bool variance_519_keep_dims_0 = const()[name = string("variance_519_keep_dims_0"), val = bool(true)]; + tensor variance_519_cast_fp16 = reduce_mean(axes = variance_519_axes_0, keep_dims = variance_519_keep_dims_0, x = var_15564_cast_fp16)[name = string("variance_519_cast_fp16")]; + fp16 var_15567_to_fp16 = const()[name = string("op_15567_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15568_cast_fp16 = add(x = variance_519_cast_fp16, y = var_15567_to_fp16)[name = string("op_15568_cast_fp16")]; + fp32 var_15569_epsilon_0 = const()[name = string("op_15569_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15569_cast_fp16 = rsqrt(epsilon = var_15569_epsilon_0, x = var_15568_cast_fp16)[name = string("op_15569_cast_fp16")]; + tensor var_15570_cast_fp16 = mul(x = x_471_cast_fp16, y = var_15569_cast_fp16)[name = string("op_15570_cast_fp16")]; + tensor input_665_cast_fp16 = mul(x = var_15570_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_665_cast_fp16")]; + string q_373_pad_type_0 = const()[name = string("q_373_pad_type_0"), val = string("valid")]; + tensor q_373_strides_0 = const()[name = string("q_373_strides_0"), val = tensor([1, 1])]; + tensor q_373_pad_0 = const()[name = string("q_373_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_373_dilations_0 = const()[name = string("q_373_dilations_0"), val = tensor([1, 1])]; + int32 q_373_groups_0 = const()[name = string("q_373_groups_0"), val = int32(1)]; + tensor q_373_cast_fp16 = conv(dilations = q_373_dilations_0, groups = q_373_groups_0, pad = q_373_pad_0, pad_type = q_373_pad_type_0, strides = q_373_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_665_cast_fp16)[name = string("q_373_cast_fp16")]; + string k_373_pad_type_0 = const()[name = string("k_373_pad_type_0"), val = string("valid")]; + tensor k_373_strides_0 = const()[name = string("k_373_strides_0"), val = tensor([1, 1])]; + tensor k_373_pad_0 = const()[name = string("k_373_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_373_dilations_0 = const()[name = string("k_373_dilations_0"), val = tensor([1, 1])]; + int32 k_373_groups_0 = const()[name = string("k_373_groups_0"), val = int32(1)]; + tensor k_373_cast_fp16 = conv(dilations = k_373_dilations_0, groups = k_373_groups_0, pad = k_373_pad_0, pad_type = k_373_pad_type_0, strides = k_373_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_665_cast_fp16)[name = string("k_373_cast_fp16")]; + string v_125_pad_type_0 = const()[name = string("v_125_pad_type_0"), val = string("valid")]; + tensor v_125_strides_0 = const()[name = string("v_125_strides_0"), val = tensor([1, 1])]; + tensor v_125_pad_0 = const()[name = string("v_125_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_125_dilations_0 = const()[name = string("v_125_dilations_0"), val = tensor([1, 1])]; + int32 v_125_groups_0 = const()[name = string("v_125_groups_0"), val = int32(1)]; + tensor v_125_cast_fp16 = conv(dilations = v_125_dilations_0, groups = v_125_groups_0, pad = v_125_pad_0, pad_type = v_125_pad_type_0, strides = v_125_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_665_cast_fp16)[name = string("v_125_cast_fp16")]; + tensor var_15604 = const()[name = string("op_15604"), val = tensor([16, 128, 1, 1])]; + tensor x_473_cast_fp16 = reshape(shape = var_15604, x = q_373_cast_fp16)[name = string("x_473_cast_fp16")]; + tensor var_15607_cast_fp16 = mul(x = x_473_cast_fp16, y = x_473_cast_fp16)[name = string("op_15607_cast_fp16")]; + tensor variance_521_axes_0 = const()[name = string("variance_521_axes_0"), val = tensor([1])]; + bool variance_521_keep_dims_0 = const()[name = string("variance_521_keep_dims_0"), val = bool(true)]; + tensor variance_521_cast_fp16 = reduce_mean(axes = variance_521_axes_0, keep_dims = variance_521_keep_dims_0, x = var_15607_cast_fp16)[name = string("variance_521_cast_fp16")]; + fp16 var_15610_to_fp16 = const()[name = string("op_15610_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15611_cast_fp16 = add(x = variance_521_cast_fp16, y = var_15610_to_fp16)[name = string("op_15611_cast_fp16")]; + fp32 var_15612_epsilon_0 = const()[name = string("op_15612_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15612_cast_fp16 = rsqrt(epsilon = var_15612_epsilon_0, x = var_15611_cast_fp16)[name = string("op_15612_cast_fp16")]; + tensor var_15613_cast_fp16 = mul(x = x_473_cast_fp16, y = var_15612_cast_fp16)[name = string("op_15613_cast_fp16")]; + tensor q_375_cast_fp16 = mul(x = var_15613_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_375_cast_fp16")]; + tensor var_15615 = const()[name = string("op_15615"), val = tensor([8, 128, 1, 1])]; + tensor x_475_cast_fp16 = reshape(shape = var_15615, x = k_373_cast_fp16)[name = string("x_475_cast_fp16")]; + tensor var_15618_cast_fp16 = mul(x = x_475_cast_fp16, y = x_475_cast_fp16)[name = string("op_15618_cast_fp16")]; + tensor variance_523_axes_0 = const()[name = string("variance_523_axes_0"), val = tensor([1])]; + bool variance_523_keep_dims_0 = const()[name = string("variance_523_keep_dims_0"), val = bool(true)]; + tensor variance_523_cast_fp16 = reduce_mean(axes = variance_523_axes_0, keep_dims = variance_523_keep_dims_0, x = var_15618_cast_fp16)[name = string("variance_523_cast_fp16")]; + fp16 var_15621_to_fp16 = const()[name = string("op_15621_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15622_cast_fp16 = add(x = variance_523_cast_fp16, y = var_15621_to_fp16)[name = string("op_15622_cast_fp16")]; + fp32 var_15623_epsilon_0 = const()[name = string("op_15623_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15623_cast_fp16 = rsqrt(epsilon = var_15623_epsilon_0, x = var_15622_cast_fp16)[name = string("op_15623_cast_fp16")]; + tensor var_15624_cast_fp16 = mul(x = x_475_cast_fp16, y = var_15623_cast_fp16)[name = string("op_15624_cast_fp16")]; + tensor k_375_cast_fp16 = mul(x = var_15624_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_375_cast_fp16")]; + tensor var_15626 = const()[name = string("op_15626"), val = tensor([1, 16, 128, 1])]; + tensor z_249_cast_fp16 = reshape(shape = var_15626, x = q_375_cast_fp16)[name = string("z_249_cast_fp16")]; + tensor var_15628 = const()[name = string("op_15628"), val = tensor([1, 8, 128, 1])]; + tensor z_251_cast_fp16 = reshape(shape = var_15628, x = k_375_cast_fp16)[name = string("z_251_cast_fp16")]; + tensor z1_249_begin_0 = const()[name = string("z1_249_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_249_end_0 = const()[name = string("z1_249_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_249_end_mask_0 = const()[name = string("z1_249_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_249_cast_fp16 = slice_by_index(begin = z1_249_begin_0, end = z1_249_end_0, end_mask = z1_249_end_mask_0, x = z_249_cast_fp16)[name = string("z1_249_cast_fp16")]; + tensor z2_249_begin_0 = const()[name = string("z2_249_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_249_end_0 = const()[name = string("z2_249_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_249_end_mask_0 = const()[name = string("z2_249_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_249_cast_fp16 = slice_by_index(begin = z2_249_begin_0, end = z2_249_end_0, end_mask = z2_249_end_mask_0, x = z_249_cast_fp16)[name = string("z2_249_cast_fp16")]; + tensor var_15636_cast_fp16 = mul(x = z_249_cast_fp16, y = cos_121_to_fp16)[name = string("op_15636_cast_fp16")]; + fp16 const_137_promoted_to_fp16 = const()[name = string("const_137_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15637_cast_fp16 = mul(x = z2_249_cast_fp16, y = const_137_promoted_to_fp16)[name = string("op_15637_cast_fp16")]; + bool var_15639_interleave_0 = const()[name = string("op_15639_interleave_0"), val = bool(false)]; + tensor var_15639_cast_fp16 = concat(axis = var_15545, interleave = var_15639_interleave_0, values = (var_15637_cast_fp16, z1_249_cast_fp16))[name = string("op_15639_cast_fp16")]; + tensor var_15640_cast_fp16 = mul(x = var_15639_cast_fp16, y = sin_121_to_fp16)[name = string("op_15640_cast_fp16")]; + tensor q_377_cast_fp16 = add(x = var_15636_cast_fp16, y = var_15640_cast_fp16)[name = string("q_377_cast_fp16")]; + tensor z1_251_begin_0 = const()[name = string("z1_251_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_251_end_0 = const()[name = string("z1_251_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_251_end_mask_0 = const()[name = string("z1_251_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_251_cast_fp16 = slice_by_index(begin = z1_251_begin_0, end = z1_251_end_0, end_mask = z1_251_end_mask_0, x = z_251_cast_fp16)[name = string("z1_251_cast_fp16")]; + tensor z2_251_begin_0 = const()[name = string("z2_251_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_251_end_0 = const()[name = string("z2_251_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_251_end_mask_0 = const()[name = string("z2_251_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_251_cast_fp16 = slice_by_index(begin = z2_251_begin_0, end = z2_251_end_0, end_mask = z2_251_end_mask_0, x = z_251_cast_fp16)[name = string("z2_251_cast_fp16")]; + tensor var_15648_cast_fp16 = mul(x = z_251_cast_fp16, y = cos_121_to_fp16)[name = string("op_15648_cast_fp16")]; + fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15649_cast_fp16 = mul(x = z2_251_cast_fp16, y = const_138_promoted_to_fp16)[name = string("op_15649_cast_fp16")]; + bool var_15651_interleave_0 = const()[name = string("op_15651_interleave_0"), val = bool(false)]; + tensor var_15651_cast_fp16 = concat(axis = var_15545, interleave = var_15651_interleave_0, values = (var_15649_cast_fp16, z1_251_cast_fp16))[name = string("op_15651_cast_fp16")]; + tensor var_15652_cast_fp16 = mul(x = var_15651_cast_fp16, y = sin_121_to_fp16)[name = string("op_15652_cast_fp16")]; + tensor k_377_cast_fp16 = add(x = var_15648_cast_fp16, y = var_15652_cast_fp16)[name = string("k_377_cast_fp16")]; + tensor var_15654 = const()[name = string("op_15654"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_125_cast_fp16 = reshape(shape = var_15654, x = k_377_cast_fp16)[name = string("cur_key_125_cast_fp16")]; + tensor var_15656_to_fp16 = const()[name = string("op_15656_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644288)))]; + tensor var_15657_cast_fp16 = mul(x = key_cache_125_cast_fp16, y = var_15656_to_fp16)[name = string("op_15657_cast_fp16")]; + tensor upd_125_to_fp16 = const()[name = string("upd_125_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644416)))]; + tensor var_15658_cast_fp16 = mul(x = cur_key_125_cast_fp16, y = upd_125_to_fp16)[name = string("op_15658_cast_fp16")]; + tensor key_125_cast_fp16 = add(x = var_15657_cast_fp16, y = var_15658_cast_fp16)[name = string("key_125_cast_fp16")]; + tensor var_15660_to_fp16 = const()[name = string("op_15660_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644288)))]; + tensor var_15661_cast_fp16 = mul(x = value_cache_125_cast_fp16, y = var_15660_to_fp16)[name = string("op_15661_cast_fp16")]; + tensor var_15662_cast_fp16 = mul(x = v_125_cast_fp16, y = upd_125_to_fp16)[name = string("op_15662_cast_fp16")]; + tensor value_125_cast_fp16 = add(x = var_15661_cast_fp16, y = var_15662_cast_fp16)[name = string("value_125_cast_fp16")]; + tensor var_15664 = const()[name = string("op_15664"), val = tensor([1, 8, 128, 16])]; + tensor kh_249_cast_fp16 = reshape(shape = var_15664, x = key_125_cast_fp16)[name = string("kh_249_cast_fp16")]; + tensor var_15666 = const()[name = string("op_15666"), val = tensor([1, 8, 128, 16])]; + tensor vh_249_cast_fp16 = reshape(shape = var_15666, x = value_125_cast_fp16)[name = string("vh_249_cast_fp16")]; + tensor transpose_248_perm_0 = const()[name = string("transpose_248_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_124_reps_0 = const()[name = string("tile_124_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_248_cast_fp16 = transpose(perm = transpose_248_perm_0, x = kh_249_cast_fp16)[name = string("transpose_107")]; + tensor tile_124_cast_fp16 = tile(reps = tile_124_reps_0, x = transpose_248_cast_fp16)[name = string("tile_124_cast_fp16")]; + tensor concat_311 = const()[name = string("concat_311"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_248_cast_fp16 = reshape(shape = concat_311, x = tile_124_cast_fp16)[name = string("reshape_248_cast_fp16")]; + tensor transpose_249_perm_0 = const()[name = string("transpose_249_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_312 = const()[name = string("concat_312"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_249_cast_fp16 = transpose(perm = transpose_249_perm_0, x = reshape_248_cast_fp16)[name = string("transpose_106")]; + tensor reshape_249_cast_fp16 = reshape(shape = concat_312, x = transpose_249_cast_fp16)[name = string("reshape_249_cast_fp16")]; + tensor transpose_250_perm_0 = const()[name = string("transpose_250_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_125_reps_0 = const()[name = string("tile_125_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_250_cast_fp16 = transpose(perm = transpose_250_perm_0, x = vh_249_cast_fp16)[name = string("transpose_105")]; + tensor tile_125_cast_fp16 = tile(reps = tile_125_reps_0, x = transpose_250_cast_fp16)[name = string("tile_125_cast_fp16")]; + tensor concat_313 = const()[name = string("concat_313"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_250_cast_fp16 = reshape(shape = concat_313, x = tile_125_cast_fp16)[name = string("reshape_250_cast_fp16")]; + tensor transpose_251_perm_0 = const()[name = string("transpose_251_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_314 = const()[name = string("concat_314"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_251_cast_fp16 = transpose(perm = transpose_251_perm_0, x = reshape_250_cast_fp16)[name = string("transpose_104")]; + tensor reshape_251_cast_fp16 = reshape(shape = concat_314, x = transpose_251_cast_fp16)[name = string("reshape_251_cast_fp16")]; + fp16 var_15670_to_fp16 = const()[name = string("op_15670_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_15671_cast_fp16 = mul(x = q_377_cast_fp16, y = var_15670_to_fp16)[name = string("op_15671_cast_fp16")]; + tensor transpose_565_perm_0 = const()[name = string("transpose_565_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_271_transpose_x_1 = const()[name = string("w_271_transpose_x_1"), val = bool(true)]; + bool w_271_transpose_y_1 = const()[name = string("w_271_transpose_y_1"), val = bool(false)]; + tensor transpose_565_cast_fp16 = transpose(perm = transpose_565_perm_0, x = reshape_249_cast_fp16)[name = string("transpose_103")]; + tensor w_271_cast_fp16 = matmul(transpose_x = w_271_transpose_x_1, transpose_y = w_271_transpose_y_1, x = var_15671_cast_fp16, y = transpose_565_cast_fp16)[name = string("w_271_cast_fp16")]; + tensor pad_125_to_fp16 = const()[name = string("pad_125_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644544)))]; + tensor var_15674_cast_fp16 = add(x = w_271_cast_fp16, y = pad_125_to_fp16)[name = string("op_15674_cast_fp16")]; + tensor w_273_cast_fp16 = softmax(axis = var_15549, x = var_15674_cast_fp16)[name = string("w_273_cast_fp16")]; + tensor transpose_566_perm_0 = const()[name = string("transpose_566_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_125_transpose_x_1 = const()[name = string("attn_125_transpose_x_1"), val = bool(false)]; + bool attn_125_transpose_y_1 = const()[name = string("attn_125_transpose_y_1"), val = bool(true)]; + tensor transpose_566_cast_fp16 = transpose(perm = transpose_566_perm_0, x = reshape_251_cast_fp16)[name = string("transpose_102")]; + tensor attn_125_cast_fp16 = matmul(transpose_x = attn_125_transpose_x_1, transpose_y = attn_125_transpose_y_1, x = transpose_566_cast_fp16, y = w_273_cast_fp16)[name = string("attn_125_cast_fp16")]; + tensor var_15678 = const()[name = string("op_15678"), val = tensor([1, 2048, 1, 1])]; + tensor input_667_cast_fp16 = reshape(shape = var_15678, x = attn_125_cast_fp16)[name = string("input_667_cast_fp16")]; + string attn_output_125_pad_type_0 = const()[name = string("attn_output_125_pad_type_0"), val = string("valid")]; + tensor attn_output_125_strides_0 = const()[name = string("attn_output_125_strides_0"), val = tensor([1, 1])]; + tensor attn_output_125_pad_0 = const()[name = string("attn_output_125_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_125_dilations_0 = const()[name = string("attn_output_125_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_125_groups_0 = const()[name = string("attn_output_125_groups_0"), val = int32(1)]; + tensor attn_output_125_cast_fp16 = conv(dilations = attn_output_125_dilations_0, groups = attn_output_125_groups_0, pad = attn_output_125_pad_0, pad_type = attn_output_125_pad_type_0, strides = attn_output_125_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_667_cast_fp16)[name = string("attn_output_125_cast_fp16")]; + tensor x_477_cast_fp16 = add(x = x_471_cast_fp16, y = attn_output_125_cast_fp16)[name = string("x_477_cast_fp16")]; + tensor var_15692_cast_fp16 = mul(x = x_477_cast_fp16, y = x_477_cast_fp16)[name = string("op_15692_cast_fp16")]; + tensor variance_525_axes_0 = const()[name = string("variance_525_axes_0"), val = tensor([1])]; + bool variance_525_keep_dims_0 = const()[name = string("variance_525_keep_dims_0"), val = bool(true)]; + tensor variance_525_cast_fp16 = reduce_mean(axes = variance_525_axes_0, keep_dims = variance_525_keep_dims_0, x = var_15692_cast_fp16)[name = string("variance_525_cast_fp16")]; + fp16 var_15695_to_fp16 = const()[name = string("op_15695_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15696_cast_fp16 = add(x = variance_525_cast_fp16, y = var_15695_to_fp16)[name = string("op_15696_cast_fp16")]; + fp32 var_15697_epsilon_0 = const()[name = string("op_15697_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15697_cast_fp16 = rsqrt(epsilon = var_15697_epsilon_0, x = var_15696_cast_fp16)[name = string("op_15697_cast_fp16")]; + tensor var_15698_cast_fp16 = mul(x = x_477_cast_fp16, y = var_15697_cast_fp16)[name = string("op_15698_cast_fp16")]; + tensor input_669_cast_fp16 = mul(x = var_15698_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_669_cast_fp16")]; + string input_671_pad_type_0 = const()[name = string("input_671_pad_type_0"), val = string("valid")]; + tensor input_671_strides_0 = const()[name = string("input_671_strides_0"), val = tensor([1, 1])]; + tensor input_671_pad_0 = const()[name = string("input_671_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_671_dilations_0 = const()[name = string("input_671_dilations_0"), val = tensor([1, 1])]; + int32 input_671_groups_0 = const()[name = string("input_671_groups_0"), val = int32(1)]; + tensor input_671_cast_fp16 = conv(dilations = input_671_dilations_0, groups = input_671_groups_0, pad = input_671_pad_0, pad_type = input_671_pad_type_0, strides = input_671_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_669_cast_fp16)[name = string("input_671_cast_fp16")]; + tensor var_15706_cast_fp16 = silu(x = input_671_cast_fp16)[name = string("op_15706_cast_fp16")]; + string var_15712_pad_type_0 = const()[name = string("op_15712_pad_type_0"), val = string("valid")]; + tensor var_15712_strides_0 = const()[name = string("op_15712_strides_0"), val = tensor([1, 1])]; + tensor var_15712_pad_0 = const()[name = string("op_15712_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15712_dilations_0 = const()[name = string("op_15712_dilations_0"), val = tensor([1, 1])]; + int32 var_15712_groups_0 = const()[name = string("op_15712_groups_0"), val = int32(1)]; + tensor var_15712_cast_fp16 = conv(dilations = var_15712_dilations_0, groups = var_15712_groups_0, pad = var_15712_pad_0, pad_type = var_15712_pad_type_0, strides = var_15712_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_669_cast_fp16)[name = string("op_15712_cast_fp16")]; + tensor input_673_cast_fp16 = mul(x = var_15706_cast_fp16, y = var_15712_cast_fp16)[name = string("input_673_cast_fp16")]; + string h_125_pad_type_0 = const()[name = string("h_125_pad_type_0"), val = string("valid")]; + tensor h_125_strides_0 = const()[name = string("h_125_strides_0"), val = tensor([1, 1])]; + tensor h_125_pad_0 = const()[name = string("h_125_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_125_dilations_0 = const()[name = string("h_125_dilations_0"), val = tensor([1, 1])]; + int32 h_125_groups_0 = const()[name = string("h_125_groups_0"), val = int32(1)]; + tensor h_125_cast_fp16 = conv(dilations = h_125_dilations_0, groups = h_125_groups_0, pad = h_125_pad_0, pad_type = h_125_pad_type_0, strides = h_125_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_673_cast_fp16)[name = string("h_125_cast_fp16")]; + tensor x_479_cast_fp16 = add(x = x_477_cast_fp16, y = h_125_cast_fp16)[name = string("x_479_cast_fp16")]; + tensor key_cache_127_begin_0 = const()[name = string("key_cache_127_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_127_end_0 = const()[name = string("key_cache_127_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_127_end_mask_0 = const()[name = string("key_cache_127_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_127_cast_fp16 = slice_by_index(begin = key_cache_127_begin_0, end = key_cache_127_end_0, end_mask = key_cache_127_end_mask_0, x = layer_key_caches_25_cast_fp16)[name = string("key_cache_127_cast_fp16")]; + tensor value_cache_127_begin_0 = const()[name = string("value_cache_127_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_127_end_0 = const()[name = string("value_cache_127_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_127_end_mask_0 = const()[name = string("value_cache_127_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_127_cast_fp16 = slice_by_index(begin = value_cache_127_begin_0, end = value_cache_127_end_0, end_mask = value_cache_127_end_mask_0, x = layer_value_caches_25_cast_fp16)[name = string("value_cache_127_cast_fp16")]; + int32 var_15765 = const()[name = string("op_15765"), val = int32(2)]; + int32 var_15769 = const()[name = string("op_15769"), val = int32(3)]; + tensor var_15784_cast_fp16 = mul(x = x_479_cast_fp16, y = x_479_cast_fp16)[name = string("op_15784_cast_fp16")]; + tensor variance_527_axes_0 = const()[name = string("variance_527_axes_0"), val = tensor([1])]; + bool variance_527_keep_dims_0 = const()[name = string("variance_527_keep_dims_0"), val = bool(true)]; + tensor variance_527_cast_fp16 = reduce_mean(axes = variance_527_axes_0, keep_dims = variance_527_keep_dims_0, x = var_15784_cast_fp16)[name = string("variance_527_cast_fp16")]; + fp16 var_15787_to_fp16 = const()[name = string("op_15787_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15788_cast_fp16 = add(x = variance_527_cast_fp16, y = var_15787_to_fp16)[name = string("op_15788_cast_fp16")]; + fp32 var_15789_epsilon_0 = const()[name = string("op_15789_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15789_cast_fp16 = rsqrt(epsilon = var_15789_epsilon_0, x = var_15788_cast_fp16)[name = string("op_15789_cast_fp16")]; + tensor var_15790_cast_fp16 = mul(x = x_479_cast_fp16, y = var_15789_cast_fp16)[name = string("op_15790_cast_fp16")]; + tensor input_675_cast_fp16 = mul(x = var_15790_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_675_cast_fp16")]; + string q_379_pad_type_0 = const()[name = string("q_379_pad_type_0"), val = string("valid")]; + tensor q_379_strides_0 = const()[name = string("q_379_strides_0"), val = tensor([1, 1])]; + tensor q_379_pad_0 = const()[name = string("q_379_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_379_dilations_0 = const()[name = string("q_379_dilations_0"), val = tensor([1, 1])]; + int32 q_379_groups_0 = const()[name = string("q_379_groups_0"), val = int32(1)]; + tensor q_379_cast_fp16 = conv(dilations = q_379_dilations_0, groups = q_379_groups_0, pad = q_379_pad_0, pad_type = q_379_pad_type_0, strides = q_379_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_675_cast_fp16)[name = string("q_379_cast_fp16")]; + string k_379_pad_type_0 = const()[name = string("k_379_pad_type_0"), val = string("valid")]; + tensor k_379_strides_0 = const()[name = string("k_379_strides_0"), val = tensor([1, 1])]; + tensor k_379_pad_0 = const()[name = string("k_379_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_379_dilations_0 = const()[name = string("k_379_dilations_0"), val = tensor([1, 1])]; + int32 k_379_groups_0 = const()[name = string("k_379_groups_0"), val = int32(1)]; + tensor k_379_cast_fp16 = conv(dilations = k_379_dilations_0, groups = k_379_groups_0, pad = k_379_pad_0, pad_type = k_379_pad_type_0, strides = k_379_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_675_cast_fp16)[name = string("k_379_cast_fp16")]; + string v_127_pad_type_0 = const()[name = string("v_127_pad_type_0"), val = string("valid")]; + tensor v_127_strides_0 = const()[name = string("v_127_strides_0"), val = tensor([1, 1])]; + tensor v_127_pad_0 = const()[name = string("v_127_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_127_dilations_0 = const()[name = string("v_127_dilations_0"), val = tensor([1, 1])]; + int32 v_127_groups_0 = const()[name = string("v_127_groups_0"), val = int32(1)]; + tensor v_127_cast_fp16 = conv(dilations = v_127_dilations_0, groups = v_127_groups_0, pad = v_127_pad_0, pad_type = v_127_pad_type_0, strides = v_127_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_675_cast_fp16)[name = string("v_127_cast_fp16")]; + tensor var_15824 = const()[name = string("op_15824"), val = tensor([16, 128, 1, 1])]; + tensor x_481_cast_fp16 = reshape(shape = var_15824, x = q_379_cast_fp16)[name = string("x_481_cast_fp16")]; + tensor var_15827_cast_fp16 = mul(x = x_481_cast_fp16, y = x_481_cast_fp16)[name = string("op_15827_cast_fp16")]; + tensor variance_529_axes_0 = const()[name = string("variance_529_axes_0"), val = tensor([1])]; + bool variance_529_keep_dims_0 = const()[name = string("variance_529_keep_dims_0"), val = bool(true)]; + tensor variance_529_cast_fp16 = reduce_mean(axes = variance_529_axes_0, keep_dims = variance_529_keep_dims_0, x = var_15827_cast_fp16)[name = string("variance_529_cast_fp16")]; + fp16 var_15830_to_fp16 = const()[name = string("op_15830_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15831_cast_fp16 = add(x = variance_529_cast_fp16, y = var_15830_to_fp16)[name = string("op_15831_cast_fp16")]; + fp32 var_15832_epsilon_0 = const()[name = string("op_15832_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15832_cast_fp16 = rsqrt(epsilon = var_15832_epsilon_0, x = var_15831_cast_fp16)[name = string("op_15832_cast_fp16")]; + tensor var_15833_cast_fp16 = mul(x = x_481_cast_fp16, y = var_15832_cast_fp16)[name = string("op_15833_cast_fp16")]; + tensor q_381_cast_fp16 = mul(x = var_15833_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_381_cast_fp16")]; + tensor var_15835 = const()[name = string("op_15835"), val = tensor([8, 128, 1, 1])]; + tensor x_483_cast_fp16 = reshape(shape = var_15835, x = k_379_cast_fp16)[name = string("x_483_cast_fp16")]; + tensor var_15838_cast_fp16 = mul(x = x_483_cast_fp16, y = x_483_cast_fp16)[name = string("op_15838_cast_fp16")]; + tensor variance_531_axes_0 = const()[name = string("variance_531_axes_0"), val = tensor([1])]; + bool variance_531_keep_dims_0 = const()[name = string("variance_531_keep_dims_0"), val = bool(true)]; + tensor variance_531_cast_fp16 = reduce_mean(axes = variance_531_axes_0, keep_dims = variance_531_keep_dims_0, x = var_15838_cast_fp16)[name = string("variance_531_cast_fp16")]; + fp16 var_15841_to_fp16 = const()[name = string("op_15841_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15842_cast_fp16 = add(x = variance_531_cast_fp16, y = var_15841_to_fp16)[name = string("op_15842_cast_fp16")]; + fp32 var_15843_epsilon_0 = const()[name = string("op_15843_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15843_cast_fp16 = rsqrt(epsilon = var_15843_epsilon_0, x = var_15842_cast_fp16)[name = string("op_15843_cast_fp16")]; + tensor var_15844_cast_fp16 = mul(x = x_483_cast_fp16, y = var_15843_cast_fp16)[name = string("op_15844_cast_fp16")]; + tensor k_381_cast_fp16 = mul(x = var_15844_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_381_cast_fp16")]; + tensor var_15846 = const()[name = string("op_15846"), val = tensor([1, 16, 128, 1])]; + tensor z_253_cast_fp16 = reshape(shape = var_15846, x = q_381_cast_fp16)[name = string("z_253_cast_fp16")]; + tensor var_15848 = const()[name = string("op_15848"), val = tensor([1, 8, 128, 1])]; + tensor z_255_cast_fp16 = reshape(shape = var_15848, x = k_381_cast_fp16)[name = string("z_255_cast_fp16")]; + tensor z1_253_begin_0 = const()[name = string("z1_253_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_253_end_0 = const()[name = string("z1_253_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_253_end_mask_0 = const()[name = string("z1_253_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_253_cast_fp16 = slice_by_index(begin = z1_253_begin_0, end = z1_253_end_0, end_mask = z1_253_end_mask_0, x = z_253_cast_fp16)[name = string("z1_253_cast_fp16")]; + tensor z2_253_begin_0 = const()[name = string("z2_253_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_253_end_0 = const()[name = string("z2_253_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_253_end_mask_0 = const()[name = string("z2_253_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_253_cast_fp16 = slice_by_index(begin = z2_253_begin_0, end = z2_253_end_0, end_mask = z2_253_end_mask_0, x = z_253_cast_fp16)[name = string("z2_253_cast_fp16")]; + tensor var_15856_cast_fp16 = mul(x = z_253_cast_fp16, y = cos_121_to_fp16)[name = string("op_15856_cast_fp16")]; + fp16 const_139_promoted_to_fp16 = const()[name = string("const_139_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15857_cast_fp16 = mul(x = z2_253_cast_fp16, y = const_139_promoted_to_fp16)[name = string("op_15857_cast_fp16")]; + bool var_15859_interleave_0 = const()[name = string("op_15859_interleave_0"), val = bool(false)]; + tensor var_15859_cast_fp16 = concat(axis = var_15765, interleave = var_15859_interleave_0, values = (var_15857_cast_fp16, z1_253_cast_fp16))[name = string("op_15859_cast_fp16")]; + tensor var_15860_cast_fp16 = mul(x = var_15859_cast_fp16, y = sin_121_to_fp16)[name = string("op_15860_cast_fp16")]; + tensor q_383_cast_fp16 = add(x = var_15856_cast_fp16, y = var_15860_cast_fp16)[name = string("q_383_cast_fp16")]; + tensor z1_255_begin_0 = const()[name = string("z1_255_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_255_end_0 = const()[name = string("z1_255_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_255_end_mask_0 = const()[name = string("z1_255_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_255_cast_fp16 = slice_by_index(begin = z1_255_begin_0, end = z1_255_end_0, end_mask = z1_255_end_mask_0, x = z_255_cast_fp16)[name = string("z1_255_cast_fp16")]; + tensor z2_255_begin_0 = const()[name = string("z2_255_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_255_end_0 = const()[name = string("z2_255_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_255_end_mask_0 = const()[name = string("z2_255_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_255_cast_fp16 = slice_by_index(begin = z2_255_begin_0, end = z2_255_end_0, end_mask = z2_255_end_mask_0, x = z_255_cast_fp16)[name = string("z2_255_cast_fp16")]; + tensor var_15868_cast_fp16 = mul(x = z_255_cast_fp16, y = cos_121_to_fp16)[name = string("op_15868_cast_fp16")]; + fp16 const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15869_cast_fp16 = mul(x = z2_255_cast_fp16, y = const_140_promoted_to_fp16)[name = string("op_15869_cast_fp16")]; + bool var_15871_interleave_0 = const()[name = string("op_15871_interleave_0"), val = bool(false)]; + tensor var_15871_cast_fp16 = concat(axis = var_15765, interleave = var_15871_interleave_0, values = (var_15869_cast_fp16, z1_255_cast_fp16))[name = string("op_15871_cast_fp16")]; + tensor var_15872_cast_fp16 = mul(x = var_15871_cast_fp16, y = sin_121_to_fp16)[name = string("op_15872_cast_fp16")]; + tensor k_383_cast_fp16 = add(x = var_15868_cast_fp16, y = var_15872_cast_fp16)[name = string("k_383_cast_fp16")]; + tensor var_15874 = const()[name = string("op_15874"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_127_cast_fp16 = reshape(shape = var_15874, x = k_383_cast_fp16)[name = string("cur_key_127_cast_fp16")]; + tensor var_15876_to_fp16 = const()[name = string("op_15876_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644288)))]; + tensor var_15877_cast_fp16 = mul(x = key_cache_127_cast_fp16, y = var_15876_to_fp16)[name = string("op_15877_cast_fp16")]; + tensor upd_127_to_fp16 = const()[name = string("upd_127_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644416)))]; + tensor var_15878_cast_fp16 = mul(x = cur_key_127_cast_fp16, y = upd_127_to_fp16)[name = string("op_15878_cast_fp16")]; + tensor key_127_cast_fp16 = add(x = var_15877_cast_fp16, y = var_15878_cast_fp16)[name = string("key_127_cast_fp16")]; + tensor var_15880_to_fp16 = const()[name = string("op_15880_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644288)))]; + tensor var_15881_cast_fp16 = mul(x = value_cache_127_cast_fp16, y = var_15880_to_fp16)[name = string("op_15881_cast_fp16")]; + tensor var_15882_cast_fp16 = mul(x = v_127_cast_fp16, y = upd_127_to_fp16)[name = string("op_15882_cast_fp16")]; + tensor value_127_cast_fp16 = add(x = var_15881_cast_fp16, y = var_15882_cast_fp16)[name = string("value_127_cast_fp16")]; + tensor var_15884 = const()[name = string("op_15884"), val = tensor([1, 8, 128, 16])]; + tensor kh_253_cast_fp16 = reshape(shape = var_15884, x = key_127_cast_fp16)[name = string("kh_253_cast_fp16")]; + tensor var_15886 = const()[name = string("op_15886"), val = tensor([1, 8, 128, 16])]; + tensor vh_253_cast_fp16 = reshape(shape = var_15886, x = value_127_cast_fp16)[name = string("vh_253_cast_fp16")]; + tensor transpose_252_perm_0 = const()[name = string("transpose_252_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_126_reps_0 = const()[name = string("tile_126_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_252_cast_fp16 = transpose(perm = transpose_252_perm_0, x = kh_253_cast_fp16)[name = string("transpose_101")]; + tensor tile_126_cast_fp16 = tile(reps = tile_126_reps_0, x = transpose_252_cast_fp16)[name = string("tile_126_cast_fp16")]; + tensor concat_315 = const()[name = string("concat_315"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_252_cast_fp16 = reshape(shape = concat_315, x = tile_126_cast_fp16)[name = string("reshape_252_cast_fp16")]; + tensor transpose_253_perm_0 = const()[name = string("transpose_253_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_316 = const()[name = string("concat_316"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_253_cast_fp16 = transpose(perm = transpose_253_perm_0, x = reshape_252_cast_fp16)[name = string("transpose_100")]; + tensor reshape_253_cast_fp16 = reshape(shape = concat_316, x = transpose_253_cast_fp16)[name = string("reshape_253_cast_fp16")]; + tensor transpose_254_perm_0 = const()[name = string("transpose_254_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_127_reps_0 = const()[name = string("tile_127_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_254_cast_fp16 = transpose(perm = transpose_254_perm_0, x = vh_253_cast_fp16)[name = string("transpose_99")]; + tensor tile_127_cast_fp16 = tile(reps = tile_127_reps_0, x = transpose_254_cast_fp16)[name = string("tile_127_cast_fp16")]; + tensor concat_317 = const()[name = string("concat_317"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_254_cast_fp16 = reshape(shape = concat_317, x = tile_127_cast_fp16)[name = string("reshape_254_cast_fp16")]; + tensor transpose_255_perm_0 = const()[name = string("transpose_255_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_318 = const()[name = string("concat_318"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_255_cast_fp16 = transpose(perm = transpose_255_perm_0, x = reshape_254_cast_fp16)[name = string("transpose_98")]; + tensor reshape_255_cast_fp16 = reshape(shape = concat_318, x = transpose_255_cast_fp16)[name = string("reshape_255_cast_fp16")]; + fp16 var_15890_to_fp16 = const()[name = string("op_15890_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_15891_cast_fp16 = mul(x = q_383_cast_fp16, y = var_15890_to_fp16)[name = string("op_15891_cast_fp16")]; + tensor transpose_569_perm_0 = const()[name = string("transpose_569_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_275_transpose_x_1 = const()[name = string("w_275_transpose_x_1"), val = bool(true)]; + bool w_275_transpose_y_1 = const()[name = string("w_275_transpose_y_1"), val = bool(false)]; + tensor transpose_569_cast_fp16 = transpose(perm = transpose_569_perm_0, x = reshape_253_cast_fp16)[name = string("transpose_97")]; + tensor w_275_cast_fp16 = matmul(transpose_x = w_275_transpose_x_1, transpose_y = w_275_transpose_y_1, x = var_15891_cast_fp16, y = transpose_569_cast_fp16)[name = string("w_275_cast_fp16")]; + tensor pad_127_to_fp16 = const()[name = string("pad_127_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644544)))]; + tensor var_15894_cast_fp16 = add(x = w_275_cast_fp16, y = pad_127_to_fp16)[name = string("op_15894_cast_fp16")]; + tensor w_277_cast_fp16 = softmax(axis = var_15769, x = var_15894_cast_fp16)[name = string("w_277_cast_fp16")]; + tensor transpose_570_perm_0 = const()[name = string("transpose_570_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_127_transpose_x_1 = const()[name = string("attn_127_transpose_x_1"), val = bool(false)]; + bool attn_127_transpose_y_1 = const()[name = string("attn_127_transpose_y_1"), val = bool(true)]; + tensor transpose_570_cast_fp16 = transpose(perm = transpose_570_perm_0, x = reshape_255_cast_fp16)[name = string("transpose_96")]; + tensor attn_127_cast_fp16 = matmul(transpose_x = attn_127_transpose_x_1, transpose_y = attn_127_transpose_y_1, x = transpose_570_cast_fp16, y = w_277_cast_fp16)[name = string("attn_127_cast_fp16")]; + tensor var_15898 = const()[name = string("op_15898"), val = tensor([1, 2048, 1, 1])]; + tensor input_677_cast_fp16 = reshape(shape = var_15898, x = attn_127_cast_fp16)[name = string("input_677_cast_fp16")]; + string attn_output_127_pad_type_0 = const()[name = string("attn_output_127_pad_type_0"), val = string("valid")]; + tensor attn_output_127_strides_0 = const()[name = string("attn_output_127_strides_0"), val = tensor([1, 1])]; + tensor attn_output_127_pad_0 = const()[name = string("attn_output_127_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_127_dilations_0 = const()[name = string("attn_output_127_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_127_groups_0 = const()[name = string("attn_output_127_groups_0"), val = int32(1)]; + tensor attn_output_127_cast_fp16 = conv(dilations = attn_output_127_dilations_0, groups = attn_output_127_groups_0, pad = attn_output_127_pad_0, pad_type = attn_output_127_pad_type_0, strides = attn_output_127_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_677_cast_fp16)[name = string("attn_output_127_cast_fp16")]; + tensor x_485_cast_fp16 = add(x = x_479_cast_fp16, y = attn_output_127_cast_fp16)[name = string("x_485_cast_fp16")]; + tensor var_15912_cast_fp16 = mul(x = x_485_cast_fp16, y = x_485_cast_fp16)[name = string("op_15912_cast_fp16")]; + tensor variance_533_axes_0 = const()[name = string("variance_533_axes_0"), val = tensor([1])]; + bool variance_533_keep_dims_0 = const()[name = string("variance_533_keep_dims_0"), val = bool(true)]; + tensor variance_533_cast_fp16 = reduce_mean(axes = variance_533_axes_0, keep_dims = variance_533_keep_dims_0, x = var_15912_cast_fp16)[name = string("variance_533_cast_fp16")]; + fp16 var_15915_to_fp16 = const()[name = string("op_15915_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15916_cast_fp16 = add(x = variance_533_cast_fp16, y = var_15915_to_fp16)[name = string("op_15916_cast_fp16")]; + fp32 var_15917_epsilon_0 = const()[name = string("op_15917_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15917_cast_fp16 = rsqrt(epsilon = var_15917_epsilon_0, x = var_15916_cast_fp16)[name = string("op_15917_cast_fp16")]; + tensor var_15918_cast_fp16 = mul(x = x_485_cast_fp16, y = var_15917_cast_fp16)[name = string("op_15918_cast_fp16")]; + tensor input_679_cast_fp16 = mul(x = var_15918_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_679_cast_fp16")]; + string input_681_pad_type_0 = const()[name = string("input_681_pad_type_0"), val = string("valid")]; + tensor input_681_strides_0 = const()[name = string("input_681_strides_0"), val = tensor([1, 1])]; + tensor input_681_pad_0 = const()[name = string("input_681_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_681_dilations_0 = const()[name = string("input_681_dilations_0"), val = tensor([1, 1])]; + int32 input_681_groups_0 = const()[name = string("input_681_groups_0"), val = int32(1)]; + tensor input_681_cast_fp16 = conv(dilations = input_681_dilations_0, groups = input_681_groups_0, pad = input_681_pad_0, pad_type = input_681_pad_type_0, strides = input_681_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_679_cast_fp16)[name = string("input_681_cast_fp16")]; + tensor var_15926_cast_fp16 = silu(x = input_681_cast_fp16)[name = string("op_15926_cast_fp16")]; + string var_15932_pad_type_0 = const()[name = string("op_15932_pad_type_0"), val = string("valid")]; + tensor var_15932_strides_0 = const()[name = string("op_15932_strides_0"), val = tensor([1, 1])]; + tensor var_15932_pad_0 = const()[name = string("op_15932_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15932_dilations_0 = const()[name = string("op_15932_dilations_0"), val = tensor([1, 1])]; + int32 var_15932_groups_0 = const()[name = string("op_15932_groups_0"), val = int32(1)]; + tensor var_15932_cast_fp16 = conv(dilations = var_15932_dilations_0, groups = var_15932_groups_0, pad = var_15932_pad_0, pad_type = var_15932_pad_type_0, strides = var_15932_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_679_cast_fp16)[name = string("op_15932_cast_fp16")]; + tensor input_683_cast_fp16 = mul(x = var_15926_cast_fp16, y = var_15932_cast_fp16)[name = string("input_683_cast_fp16")]; + string h_127_pad_type_0 = const()[name = string("h_127_pad_type_0"), val = string("valid")]; + tensor h_127_strides_0 = const()[name = string("h_127_strides_0"), val = tensor([1, 1])]; + tensor h_127_pad_0 = const()[name = string("h_127_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_127_dilations_0 = const()[name = string("h_127_dilations_0"), val = tensor([1, 1])]; + int32 h_127_groups_0 = const()[name = string("h_127_groups_0"), val = int32(1)]; + tensor h_127_cast_fp16 = conv(dilations = h_127_dilations_0, groups = h_127_groups_0, pad = h_127_pad_0, pad_type = h_127_pad_type_0, strides = h_127_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_683_cast_fp16)[name = string("h_127_cast_fp16")]; + tensor x_487_cast_fp16 = add(x = x_485_cast_fp16, y = h_127_cast_fp16)[name = string("x_487_cast_fp16")]; + tensor key_cache_129_begin_0 = const()[name = string("key_cache_129_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_129_end_0 = const()[name = string("key_cache_129_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_129_end_mask_0 = const()[name = string("key_cache_129_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_129_cast_fp16 = slice_by_index(begin = key_cache_129_begin_0, end = key_cache_129_end_0, end_mask = key_cache_129_end_mask_0, x = layer_key_caches_25_cast_fp16)[name = string("key_cache_129_cast_fp16")]; + tensor value_cache_129_begin_0 = const()[name = string("value_cache_129_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_129_end_0 = const()[name = string("value_cache_129_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_129_end_mask_0 = const()[name = string("value_cache_129_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_129_cast_fp16 = slice_by_index(begin = value_cache_129_begin_0, end = value_cache_129_end_0, end_mask = value_cache_129_end_mask_0, x = layer_value_caches_25_cast_fp16)[name = string("value_cache_129_cast_fp16")]; + int32 var_15985 = const()[name = string("op_15985"), val = int32(2)]; + int32 var_15989 = const()[name = string("op_15989"), val = int32(3)]; + tensor var_16004_cast_fp16 = mul(x = x_487_cast_fp16, y = x_487_cast_fp16)[name = string("op_16004_cast_fp16")]; + tensor variance_535_axes_0 = const()[name = string("variance_535_axes_0"), val = tensor([1])]; + bool variance_535_keep_dims_0 = const()[name = string("variance_535_keep_dims_0"), val = bool(true)]; + tensor variance_535_cast_fp16 = reduce_mean(axes = variance_535_axes_0, keep_dims = variance_535_keep_dims_0, x = var_16004_cast_fp16)[name = string("variance_535_cast_fp16")]; + fp16 var_16007_to_fp16 = const()[name = string("op_16007_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16008_cast_fp16 = add(x = variance_535_cast_fp16, y = var_16007_to_fp16)[name = string("op_16008_cast_fp16")]; + fp32 var_16009_epsilon_0 = const()[name = string("op_16009_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16009_cast_fp16 = rsqrt(epsilon = var_16009_epsilon_0, x = var_16008_cast_fp16)[name = string("op_16009_cast_fp16")]; + tensor var_16010_cast_fp16 = mul(x = x_487_cast_fp16, y = var_16009_cast_fp16)[name = string("op_16010_cast_fp16")]; + tensor input_685_cast_fp16 = mul(x = var_16010_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_685_cast_fp16")]; + string q_385_pad_type_0 = const()[name = string("q_385_pad_type_0"), val = string("valid")]; + tensor q_385_strides_0 = const()[name = string("q_385_strides_0"), val = tensor([1, 1])]; + tensor q_385_pad_0 = const()[name = string("q_385_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_385_dilations_0 = const()[name = string("q_385_dilations_0"), val = tensor([1, 1])]; + int32 q_385_groups_0 = const()[name = string("q_385_groups_0"), val = int32(1)]; + tensor q_385_cast_fp16 = conv(dilations = q_385_dilations_0, groups = q_385_groups_0, pad = q_385_pad_0, pad_type = q_385_pad_type_0, strides = q_385_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_685_cast_fp16)[name = string("q_385_cast_fp16")]; + string k_385_pad_type_0 = const()[name = string("k_385_pad_type_0"), val = string("valid")]; + tensor k_385_strides_0 = const()[name = string("k_385_strides_0"), val = tensor([1, 1])]; + tensor k_385_pad_0 = const()[name = string("k_385_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_385_dilations_0 = const()[name = string("k_385_dilations_0"), val = tensor([1, 1])]; + int32 k_385_groups_0 = const()[name = string("k_385_groups_0"), val = int32(1)]; + tensor k_385_cast_fp16 = conv(dilations = k_385_dilations_0, groups = k_385_groups_0, pad = k_385_pad_0, pad_type = k_385_pad_type_0, strides = k_385_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_685_cast_fp16)[name = string("k_385_cast_fp16")]; + string v_129_pad_type_0 = const()[name = string("v_129_pad_type_0"), val = string("valid")]; + tensor v_129_strides_0 = const()[name = string("v_129_strides_0"), val = tensor([1, 1])]; + tensor v_129_pad_0 = const()[name = string("v_129_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_129_dilations_0 = const()[name = string("v_129_dilations_0"), val = tensor([1, 1])]; + int32 v_129_groups_0 = const()[name = string("v_129_groups_0"), val = int32(1)]; + tensor v_129_cast_fp16 = conv(dilations = v_129_dilations_0, groups = v_129_groups_0, pad = v_129_pad_0, pad_type = v_129_pad_type_0, strides = v_129_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_685_cast_fp16)[name = string("v_129_cast_fp16")]; + tensor var_16044 = const()[name = string("op_16044"), val = tensor([16, 128, 1, 1])]; + tensor x_489_cast_fp16 = reshape(shape = var_16044, x = q_385_cast_fp16)[name = string("x_489_cast_fp16")]; + tensor var_16047_cast_fp16 = mul(x = x_489_cast_fp16, y = x_489_cast_fp16)[name = string("op_16047_cast_fp16")]; + tensor variance_537_axes_0 = const()[name = string("variance_537_axes_0"), val = tensor([1])]; + bool variance_537_keep_dims_0 = const()[name = string("variance_537_keep_dims_0"), val = bool(true)]; + tensor variance_537_cast_fp16 = reduce_mean(axes = variance_537_axes_0, keep_dims = variance_537_keep_dims_0, x = var_16047_cast_fp16)[name = string("variance_537_cast_fp16")]; + fp16 var_16050_to_fp16 = const()[name = string("op_16050_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16051_cast_fp16 = add(x = variance_537_cast_fp16, y = var_16050_to_fp16)[name = string("op_16051_cast_fp16")]; + fp32 var_16052_epsilon_0 = const()[name = string("op_16052_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16052_cast_fp16 = rsqrt(epsilon = var_16052_epsilon_0, x = var_16051_cast_fp16)[name = string("op_16052_cast_fp16")]; + tensor var_16053_cast_fp16 = mul(x = x_489_cast_fp16, y = var_16052_cast_fp16)[name = string("op_16053_cast_fp16")]; + tensor q_387_cast_fp16 = mul(x = var_16053_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_387_cast_fp16")]; + tensor var_16055 = const()[name = string("op_16055"), val = tensor([8, 128, 1, 1])]; + tensor x_491_cast_fp16 = reshape(shape = var_16055, x = k_385_cast_fp16)[name = string("x_491_cast_fp16")]; + tensor var_16058_cast_fp16 = mul(x = x_491_cast_fp16, y = x_491_cast_fp16)[name = string("op_16058_cast_fp16")]; + tensor variance_539_axes_0 = const()[name = string("variance_539_axes_0"), val = tensor([1])]; + bool variance_539_keep_dims_0 = const()[name = string("variance_539_keep_dims_0"), val = bool(true)]; + tensor variance_539_cast_fp16 = reduce_mean(axes = variance_539_axes_0, keep_dims = variance_539_keep_dims_0, x = var_16058_cast_fp16)[name = string("variance_539_cast_fp16")]; + fp16 var_16061_to_fp16 = const()[name = string("op_16061_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16062_cast_fp16 = add(x = variance_539_cast_fp16, y = var_16061_to_fp16)[name = string("op_16062_cast_fp16")]; + fp32 var_16063_epsilon_0 = const()[name = string("op_16063_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16063_cast_fp16 = rsqrt(epsilon = var_16063_epsilon_0, x = var_16062_cast_fp16)[name = string("op_16063_cast_fp16")]; + tensor var_16064_cast_fp16 = mul(x = x_491_cast_fp16, y = var_16063_cast_fp16)[name = string("op_16064_cast_fp16")]; + tensor k_387_cast_fp16 = mul(x = var_16064_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_387_cast_fp16")]; + tensor var_16066 = const()[name = string("op_16066"), val = tensor([1, 16, 128, 1])]; + tensor z_257_cast_fp16 = reshape(shape = var_16066, x = q_387_cast_fp16)[name = string("z_257_cast_fp16")]; + tensor var_16068 = const()[name = string("op_16068"), val = tensor([1, 8, 128, 1])]; + tensor z_259_cast_fp16 = reshape(shape = var_16068, x = k_387_cast_fp16)[name = string("z_259_cast_fp16")]; + tensor z1_257_begin_0 = const()[name = string("z1_257_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_257_end_0 = const()[name = string("z1_257_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_257_end_mask_0 = const()[name = string("z1_257_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_257_cast_fp16 = slice_by_index(begin = z1_257_begin_0, end = z1_257_end_0, end_mask = z1_257_end_mask_0, x = z_257_cast_fp16)[name = string("z1_257_cast_fp16")]; + tensor z2_257_begin_0 = const()[name = string("z2_257_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_257_end_0 = const()[name = string("z2_257_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_257_end_mask_0 = const()[name = string("z2_257_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_257_cast_fp16 = slice_by_index(begin = z2_257_begin_0, end = z2_257_end_0, end_mask = z2_257_end_mask_0, x = z_257_cast_fp16)[name = string("z2_257_cast_fp16")]; + tensor var_16076_cast_fp16 = mul(x = z_257_cast_fp16, y = cos_121_to_fp16)[name = string("op_16076_cast_fp16")]; + fp16 const_141_promoted_to_fp16 = const()[name = string("const_141_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16077_cast_fp16 = mul(x = z2_257_cast_fp16, y = const_141_promoted_to_fp16)[name = string("op_16077_cast_fp16")]; + bool var_16079_interleave_0 = const()[name = string("op_16079_interleave_0"), val = bool(false)]; + tensor var_16079_cast_fp16 = concat(axis = var_15985, interleave = var_16079_interleave_0, values = (var_16077_cast_fp16, z1_257_cast_fp16))[name = string("op_16079_cast_fp16")]; + tensor var_16080_cast_fp16 = mul(x = var_16079_cast_fp16, y = sin_121_to_fp16)[name = string("op_16080_cast_fp16")]; + tensor q_389_cast_fp16 = add(x = var_16076_cast_fp16, y = var_16080_cast_fp16)[name = string("q_389_cast_fp16")]; + tensor z1_259_begin_0 = const()[name = string("z1_259_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_259_end_0 = const()[name = string("z1_259_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_259_end_mask_0 = const()[name = string("z1_259_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_259_cast_fp16 = slice_by_index(begin = z1_259_begin_0, end = z1_259_end_0, end_mask = z1_259_end_mask_0, x = z_259_cast_fp16)[name = string("z1_259_cast_fp16")]; + tensor z2_259_begin_0 = const()[name = string("z2_259_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_259_end_0 = const()[name = string("z2_259_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_259_end_mask_0 = const()[name = string("z2_259_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_259_cast_fp16 = slice_by_index(begin = z2_259_begin_0, end = z2_259_end_0, end_mask = z2_259_end_mask_0, x = z_259_cast_fp16)[name = string("z2_259_cast_fp16")]; + tensor var_16088_cast_fp16 = mul(x = z_259_cast_fp16, y = cos_121_to_fp16)[name = string("op_16088_cast_fp16")]; + fp16 const_142_promoted_to_fp16 = const()[name = string("const_142_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16089_cast_fp16 = mul(x = z2_259_cast_fp16, y = const_142_promoted_to_fp16)[name = string("op_16089_cast_fp16")]; + bool var_16091_interleave_0 = const()[name = string("op_16091_interleave_0"), val = bool(false)]; + tensor var_16091_cast_fp16 = concat(axis = var_15985, interleave = var_16091_interleave_0, values = (var_16089_cast_fp16, z1_259_cast_fp16))[name = string("op_16091_cast_fp16")]; + tensor var_16092_cast_fp16 = mul(x = var_16091_cast_fp16, y = sin_121_to_fp16)[name = string("op_16092_cast_fp16")]; + tensor k_389_cast_fp16 = add(x = var_16088_cast_fp16, y = var_16092_cast_fp16)[name = string("k_389_cast_fp16")]; + tensor var_16094 = const()[name = string("op_16094"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_129_cast_fp16 = reshape(shape = var_16094, x = k_389_cast_fp16)[name = string("cur_key_129_cast_fp16")]; + tensor var_16096_to_fp16 = const()[name = string("op_16096_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644288)))]; + tensor var_16097_cast_fp16 = mul(x = key_cache_129_cast_fp16, y = var_16096_to_fp16)[name = string("op_16097_cast_fp16")]; + tensor upd_129_to_fp16 = const()[name = string("upd_129_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644416)))]; + tensor var_16098_cast_fp16 = mul(x = cur_key_129_cast_fp16, y = upd_129_to_fp16)[name = string("op_16098_cast_fp16")]; + tensor key_129_cast_fp16 = add(x = var_16097_cast_fp16, y = var_16098_cast_fp16)[name = string("key_129_cast_fp16")]; + tensor var_16100_to_fp16 = const()[name = string("op_16100_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644288)))]; + tensor var_16101_cast_fp16 = mul(x = value_cache_129_cast_fp16, y = var_16100_to_fp16)[name = string("op_16101_cast_fp16")]; + tensor var_16102_cast_fp16 = mul(x = v_129_cast_fp16, y = upd_129_to_fp16)[name = string("op_16102_cast_fp16")]; + tensor value_129_cast_fp16 = add(x = var_16101_cast_fp16, y = var_16102_cast_fp16)[name = string("value_129_cast_fp16")]; + tensor var_16104 = const()[name = string("op_16104"), val = tensor([1, 8, 128, 16])]; + tensor kh_257_cast_fp16 = reshape(shape = var_16104, x = key_129_cast_fp16)[name = string("kh_257_cast_fp16")]; + tensor var_16106 = const()[name = string("op_16106"), val = tensor([1, 8, 128, 16])]; + tensor vh_257_cast_fp16 = reshape(shape = var_16106, x = value_129_cast_fp16)[name = string("vh_257_cast_fp16")]; + tensor transpose_256_perm_0 = const()[name = string("transpose_256_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_128_reps_0 = const()[name = string("tile_128_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_256_cast_fp16 = transpose(perm = transpose_256_perm_0, x = kh_257_cast_fp16)[name = string("transpose_95")]; + tensor tile_128_cast_fp16 = tile(reps = tile_128_reps_0, x = transpose_256_cast_fp16)[name = string("tile_128_cast_fp16")]; + tensor concat_319 = const()[name = string("concat_319"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_256_cast_fp16 = reshape(shape = concat_319, x = tile_128_cast_fp16)[name = string("reshape_256_cast_fp16")]; + tensor transpose_257_perm_0 = const()[name = string("transpose_257_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_320 = const()[name = string("concat_320"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_257_cast_fp16 = transpose(perm = transpose_257_perm_0, x = reshape_256_cast_fp16)[name = string("transpose_94")]; + tensor reshape_257_cast_fp16 = reshape(shape = concat_320, x = transpose_257_cast_fp16)[name = string("reshape_257_cast_fp16")]; + tensor transpose_258_perm_0 = const()[name = string("transpose_258_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_129_reps_0 = const()[name = string("tile_129_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_258_cast_fp16 = transpose(perm = transpose_258_perm_0, x = vh_257_cast_fp16)[name = string("transpose_93")]; + tensor tile_129_cast_fp16 = tile(reps = tile_129_reps_0, x = transpose_258_cast_fp16)[name = string("tile_129_cast_fp16")]; + tensor concat_321 = const()[name = string("concat_321"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_258_cast_fp16 = reshape(shape = concat_321, x = tile_129_cast_fp16)[name = string("reshape_258_cast_fp16")]; + tensor transpose_259_perm_0 = const()[name = string("transpose_259_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_322 = const()[name = string("concat_322"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_259_cast_fp16 = transpose(perm = transpose_259_perm_0, x = reshape_258_cast_fp16)[name = string("transpose_92")]; + tensor reshape_259_cast_fp16 = reshape(shape = concat_322, x = transpose_259_cast_fp16)[name = string("reshape_259_cast_fp16")]; + fp16 var_16110_to_fp16 = const()[name = string("op_16110_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_16111_cast_fp16 = mul(x = q_389_cast_fp16, y = var_16110_to_fp16)[name = string("op_16111_cast_fp16")]; + tensor transpose_573_perm_0 = const()[name = string("transpose_573_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_279_transpose_x_1 = const()[name = string("w_279_transpose_x_1"), val = bool(true)]; + bool w_279_transpose_y_1 = const()[name = string("w_279_transpose_y_1"), val = bool(false)]; + tensor transpose_573_cast_fp16 = transpose(perm = transpose_573_perm_0, x = reshape_257_cast_fp16)[name = string("transpose_91")]; + tensor w_279_cast_fp16 = matmul(transpose_x = w_279_transpose_x_1, transpose_y = w_279_transpose_y_1, x = var_16111_cast_fp16, y = transpose_573_cast_fp16)[name = string("w_279_cast_fp16")]; + tensor pad_129_to_fp16 = const()[name = string("pad_129_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644544)))]; + tensor var_16114_cast_fp16 = add(x = w_279_cast_fp16, y = pad_129_to_fp16)[name = string("op_16114_cast_fp16")]; + tensor w_281_cast_fp16 = softmax(axis = var_15989, x = var_16114_cast_fp16)[name = string("w_281_cast_fp16")]; + tensor transpose_574_perm_0 = const()[name = string("transpose_574_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_129_transpose_x_1 = const()[name = string("attn_129_transpose_x_1"), val = bool(false)]; + bool attn_129_transpose_y_1 = const()[name = string("attn_129_transpose_y_1"), val = bool(true)]; + tensor transpose_574_cast_fp16 = transpose(perm = transpose_574_perm_0, x = reshape_259_cast_fp16)[name = string("transpose_90")]; + tensor attn_129_cast_fp16 = matmul(transpose_x = attn_129_transpose_x_1, transpose_y = attn_129_transpose_y_1, x = transpose_574_cast_fp16, y = w_281_cast_fp16)[name = string("attn_129_cast_fp16")]; + tensor var_16118 = const()[name = string("op_16118"), val = tensor([1, 2048, 1, 1])]; + tensor input_687_cast_fp16 = reshape(shape = var_16118, x = attn_129_cast_fp16)[name = string("input_687_cast_fp16")]; + string attn_output_129_pad_type_0 = const()[name = string("attn_output_129_pad_type_0"), val = string("valid")]; + tensor attn_output_129_strides_0 = const()[name = string("attn_output_129_strides_0"), val = tensor([1, 1])]; + tensor attn_output_129_pad_0 = const()[name = string("attn_output_129_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_129_dilations_0 = const()[name = string("attn_output_129_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_129_groups_0 = const()[name = string("attn_output_129_groups_0"), val = int32(1)]; + tensor attn_output_129_cast_fp16 = conv(dilations = attn_output_129_dilations_0, groups = attn_output_129_groups_0, pad = attn_output_129_pad_0, pad_type = attn_output_129_pad_type_0, strides = attn_output_129_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_687_cast_fp16)[name = string("attn_output_129_cast_fp16")]; + tensor x_493_cast_fp16 = add(x = x_487_cast_fp16, y = attn_output_129_cast_fp16)[name = string("x_493_cast_fp16")]; + tensor var_16132_cast_fp16 = mul(x = x_493_cast_fp16, y = x_493_cast_fp16)[name = string("op_16132_cast_fp16")]; + tensor variance_541_axes_0 = const()[name = string("variance_541_axes_0"), val = tensor([1])]; + bool variance_541_keep_dims_0 = const()[name = string("variance_541_keep_dims_0"), val = bool(true)]; + tensor variance_541_cast_fp16 = reduce_mean(axes = variance_541_axes_0, keep_dims = variance_541_keep_dims_0, x = var_16132_cast_fp16)[name = string("variance_541_cast_fp16")]; + fp16 var_16135_to_fp16 = const()[name = string("op_16135_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16136_cast_fp16 = add(x = variance_541_cast_fp16, y = var_16135_to_fp16)[name = string("op_16136_cast_fp16")]; + fp32 var_16137_epsilon_0 = const()[name = string("op_16137_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16137_cast_fp16 = rsqrt(epsilon = var_16137_epsilon_0, x = var_16136_cast_fp16)[name = string("op_16137_cast_fp16")]; + tensor var_16138_cast_fp16 = mul(x = x_493_cast_fp16, y = var_16137_cast_fp16)[name = string("op_16138_cast_fp16")]; + tensor input_689_cast_fp16 = mul(x = var_16138_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_689_cast_fp16")]; + string input_691_pad_type_0 = const()[name = string("input_691_pad_type_0"), val = string("valid")]; + tensor input_691_strides_0 = const()[name = string("input_691_strides_0"), val = tensor([1, 1])]; + tensor input_691_pad_0 = const()[name = string("input_691_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_691_dilations_0 = const()[name = string("input_691_dilations_0"), val = tensor([1, 1])]; + int32 input_691_groups_0 = const()[name = string("input_691_groups_0"), val = int32(1)]; + tensor input_691_cast_fp16 = conv(dilations = input_691_dilations_0, groups = input_691_groups_0, pad = input_691_pad_0, pad_type = input_691_pad_type_0, strides = input_691_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_689_cast_fp16)[name = string("input_691_cast_fp16")]; + tensor var_16146_cast_fp16 = silu(x = input_691_cast_fp16)[name = string("op_16146_cast_fp16")]; + string var_16152_pad_type_0 = const()[name = string("op_16152_pad_type_0"), val = string("valid")]; + tensor var_16152_strides_0 = const()[name = string("op_16152_strides_0"), val = tensor([1, 1])]; + tensor var_16152_pad_0 = const()[name = string("op_16152_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16152_dilations_0 = const()[name = string("op_16152_dilations_0"), val = tensor([1, 1])]; + int32 var_16152_groups_0 = const()[name = string("op_16152_groups_0"), val = int32(1)]; + tensor var_16152_cast_fp16 = conv(dilations = var_16152_dilations_0, groups = var_16152_groups_0, pad = var_16152_pad_0, pad_type = var_16152_pad_type_0, strides = var_16152_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_689_cast_fp16)[name = string("op_16152_cast_fp16")]; + tensor input_693_cast_fp16 = mul(x = var_16146_cast_fp16, y = var_16152_cast_fp16)[name = string("input_693_cast_fp16")]; + string h_129_pad_type_0 = const()[name = string("h_129_pad_type_0"), val = string("valid")]; + tensor h_129_strides_0 = const()[name = string("h_129_strides_0"), val = tensor([1, 1])]; + tensor h_129_pad_0 = const()[name = string("h_129_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_129_dilations_0 = const()[name = string("h_129_dilations_0"), val = tensor([1, 1])]; + int32 h_129_groups_0 = const()[name = string("h_129_groups_0"), val = int32(1)]; + tensor h_129_cast_fp16 = conv(dilations = h_129_dilations_0, groups = h_129_groups_0, pad = h_129_pad_0, pad_type = h_129_pad_type_0, strides = h_129_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_693_cast_fp16)[name = string("h_129_cast_fp16")]; + tensor inputs_23_cast_fp16 = add(x = x_493_cast_fp16, y = h_129_cast_fp16)[name = string("inputs_23_cast_fp16")]; + int32 var_16180 = const()[name = string("op_16180"), val = int32(1)]; + bool layer_key_caches_27_interleave_0 = const()[name = string("layer_key_caches_27_interleave_0"), val = bool(false)]; + tensor layer_key_caches_27_cast_fp16 = concat(axis = var_16180, interleave = layer_key_caches_27_interleave_0, values = (key_121_cast_fp16, key_123_cast_fp16, key_125_cast_fp16, key_127_cast_fp16, key_129_cast_fp16))[name = string("layer_key_caches_27_cast_fp16")]; + int32 var_16183 = const()[name = string("op_16183"), val = int32(1)]; + bool layer_value_caches_27_interleave_0 = const()[name = string("layer_value_caches_27_interleave_0"), val = bool(false)]; + tensor layer_value_caches_27_cast_fp16 = concat(axis = var_16183, interleave = layer_value_caches_27_interleave_0, values = (value_121_cast_fp16, value_123_cast_fp16, value_125_cast_fp16, value_127_cast_fp16, value_129_cast_fp16))[name = string("layer_value_caches_27_cast_fp16")]; + tensor inputs_sq_23_cast_fp16 = mul(x = inputs_23_cast_fp16, y = inputs_23_cast_fp16)[name = string("inputs_sq_23_cast_fp16")]; + tensor variance_543_axes_0 = const()[name = string("variance_543_axes_0"), val = tensor([1])]; + bool variance_543_keep_dims_0 = const()[name = string("variance_543_keep_dims_0"), val = bool(true)]; + tensor variance_543_cast_fp16 = reduce_mean(axes = variance_543_axes_0, keep_dims = variance_543_keep_dims_0, x = inputs_sq_23_cast_fp16)[name = string("variance_543_cast_fp16")]; + fp16 var_16193_to_fp16 = const()[name = string("op_16193_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16194_cast_fp16 = add(x = variance_543_cast_fp16, y = var_16193_to_fp16)[name = string("op_16194_cast_fp16")]; + fp32 var_16195_epsilon_0 = const()[name = string("op_16195_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16195_cast_fp16 = rsqrt(epsilon = var_16195_epsilon_0, x = var_16194_cast_fp16)[name = string("op_16195_cast_fp16")]; + tensor hidden_states_23_cast_fp16 = mul(x = inputs_23_cast_fp16, y = var_16195_cast_fp16)[name = string("hidden_states_23_cast_fp16")]; + tensor input_695_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_23_cast_fp16)[name = string("input_695_cast_fp16")]; + string logits_45_pad_type_0 = const()[name = string("logits_45_pad_type_0"), val = string("valid")]; + tensor logits_45_strides_0 = const()[name = string("logits_45_strides_0"), val = tensor([1, 1])]; + tensor logits_45_pad_0 = const()[name = string("logits_45_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_45_dilations_0 = const()[name = string("logits_45_dilations_0"), val = tensor([1, 1])]; + int32 logits_45_groups_0 = const()[name = string("logits_45_groups_0"), val = int32(1)]; + tensor lm_heads_11_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101782400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103879616))))[name = string("lm_heads_11_weight_to_fp16_palettized")]; + tensor logits_45_cast_fp16 = conv(dilations = logits_45_dilations_0, groups = logits_45_groups_0, pad = logits_45_pad_0, pad_type = logits_45_pad_type_0, strides = logits_45_strides_0, weight = lm_heads_11_weight_to_fp16_palettized, x = input_695_cast_fp16)[name = string("logits_45_cast_fp16")]; + tensor var_16213 = const()[name = string("op_16213"), val = tensor([1, 2048])]; + tensor logits_47_cast_fp16 = reshape(shape = var_16213, x = logits_45_cast_fp16)[name = string("logits_47_cast_fp16")]; + tensor scaled_logits_23_cast_fp16 = real_div(x = logits_47_cast_fp16, y = temperature)[name = string("scaled_logits_23_cast_fp16")]; + int32 var_16223 = const()[name = string("op_16223"), val = int32(100)]; + int32 top_values_23_axis_0 = const()[name = string("top_values_23_axis_0"), val = int32(1)]; + bool top_values_23_ascending_0 = const()[name = string("top_values_23_ascending_0"), val = bool(false)]; + bool top_values_23_sort_0 = const()[name = string("top_values_23_sort_0"), val = bool(true)]; + bool top_values_23_return_indices_0 = const()[name = string("top_values_23_return_indices_0"), val = bool(true)]; + string top_values_23_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_23_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_23_cast_fp16_cast_uint16_0, tensor top_values_23_cast_fp16_cast_uint16_1 = topk(ascending = top_values_23_ascending_0, axis = top_values_23_axis_0, k = var_16223, output_indices_dtype = top_values_23_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_23_return_indices_0, sort = top_values_23_sort_0, x = scaled_logits_23_cast_fp16)[name = string("top_values_23_cast_fp16_cast_uint16")]; + tensor var_16229_cast_fp16 = mul(x = top_values_23_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_16229_cast_fp16")]; + tensor var_16233_cast_fp16 = add(x = var_16229_cast_fp16, y = var_2438_cast_fp16)[name = string("op_16233_cast_fp16")]; + tensor reduce_min_11_axes_0 = const()[name = string("reduce_min_11_axes_0"), val = tensor([1])]; + bool reduce_min_11_keep_dims_0 = const()[name = string("reduce_min_11_keep_dims_0"), val = bool(true)]; + tensor reduce_min_11_cast_fp16 = reduce_min(axes = reduce_min_11_axes_0, keep_dims = reduce_min_11_keep_dims_0, x = var_16233_cast_fp16)[name = string("reduce_min_11_cast_fp16")]; + tensor var_16236_cast_fp16 = greater_equal(x = scaled_logits_23_cast_fp16, y = reduce_min_11_cast_fp16)[name = string("op_16236_cast_fp16")]; + fp16 var_16237_value_0_to_fp16 = const()[name = string("op_16237_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_16237_cast_fp16 = fill_like(ref_tensor = scaled_logits_23_cast_fp16, value = var_16237_value_0_to_fp16)[name = string("op_16237_cast_fp16")]; + tensor masked_logits_23_cast_fp16 = select(a = scaled_logits_23_cast_fp16, b = var_16237_cast_fp16, cond = var_16236_cast_fp16)[name = string("masked_logits_23_cast_fp16")]; + tensor var_16241_begin_0 = const()[name = string("op_16241_begin_0"), val = tensor([11, 0])]; + tensor var_16241_end_0 = const()[name = string("op_16241_end_0"), val = tensor([12, 2048])]; + tensor var_16241_end_mask_0 = const()[name = string("op_16241_end_mask_0"), val = tensor([false, true])]; + tensor var_16241_squeeze_mask_0 = const()[name = string("op_16241_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_16241_cast_fp16 = slice_by_index(begin = var_16241_begin_0, end = var_16241_end_0, end_mask = var_16241_end_mask_0, squeeze_mask = var_16241_squeeze_mask_0, x = gumbel)[name = string("op_16241_cast_fp16")]; + tensor var_16244 = const()[name = string("op_16244"), val = tensor([1, 2048])]; + tensor var_16245_cast_fp16 = reshape(shape = var_16244, x = var_16241_cast_fp16)[name = string("op_16245_cast_fp16")]; + tensor noisy_logits_23_cast_fp16 = add(x = masked_logits_23_cast_fp16, y = var_16245_cast_fp16)[name = string("noisy_logits_23_cast_fp16")]; + int32 code_23_axis_0 = const()[name = string("code_23_axis_0"), val = int32(1)]; + bool code_23_keep_dims_0 = const()[name = string("code_23_keep_dims_0"), val = bool(false)]; + string code_23_output_dtype_0 = const()[name = string("code_23_output_dtype_0"), val = string("int32")]; + tensor code_23_cast_fp16 = reduce_argmax(axis = code_23_axis_0, keep_dims = code_23_keep_dims_0, output_dtype = code_23_output_dtype_0, x = noisy_logits_23_cast_fp16)[name = string("code_23_cast_fp16")]; + int32 var_16256 = const()[name = string("op_16256"), val = int32(22528)]; + tensor input_697 = add(x = code_23_cast_fp16, y = var_16256)[name = string("input_697")]; + int32 code_embed_45_axis_0 = const()[name = string("code_embed_45_axis_0"), val = int32(0)]; + int32 code_embed_45_batch_dims_0 = const()[name = string("code_embed_45_batch_dims_0"), val = int32(0)]; + bool code_embed_45_validate_indices_0 = const()[name = string("code_embed_45_validate_indices_0"), val = bool(false)]; + string input_697_to_uint16_dtype_0 = const()[name = string("input_697_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_697_to_uint16 = cast(dtype = input_697_to_uint16_dtype_0, x = input_697)[name = string("cast_3")]; + tensor code_embed_45_cast_fp16_cast_uint16 = gather(axis = code_embed_45_axis_0, batch_dims = code_embed_45_batch_dims_0, indices = input_697_to_uint16, validate_indices = code_embed_45_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_45_cast_fp16_cast_uint16")]; + tensor var_16260 = const()[name = string("op_16260"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_47_cast_fp16 = reshape(shape = var_16260, x = code_embed_45_cast_fp16_cast_uint16)[name = string("code_embed_47_cast_fp16")]; + tensor embed_sum_25_cast_fp16 = add(x = embed_sum_23_cast_fp16, y = code_embed_47_cast_fp16)[name = string("embed_sum_25_cast_fp16")]; + tensor key_cache_131_begin_0 = const()[name = string("key_cache_131_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_131_end_0 = const()[name = string("key_cache_131_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_131_end_mask_0 = const()[name = string("key_cache_131_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_131_cast_fp16 = slice_by_index(begin = key_cache_131_begin_0, end = key_cache_131_end_0, end_mask = key_cache_131_end_mask_0, x = layer_key_caches_27_cast_fp16)[name = string("key_cache_131_cast_fp16")]; + tensor value_cache_131_begin_0 = const()[name = string("value_cache_131_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_131_end_0 = const()[name = string("value_cache_131_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_131_end_mask_0 = const()[name = string("value_cache_131_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_131_cast_fp16 = slice_by_index(begin = value_cache_131_begin_0, end = value_cache_131_end_0, end_mask = value_cache_131_end_mask_0, x = layer_value_caches_27_cast_fp16)[name = string("value_cache_131_cast_fp16")]; + int32 var_16359 = const()[name = string("op_16359"), val = int32(2)]; + int32 var_16363 = const()[name = string("op_16363"), val = int32(3)]; + tensor var_16378_cast_fp16 = mul(x = code_embed_47_cast_fp16, y = code_embed_47_cast_fp16)[name = string("op_16378_cast_fp16")]; + tensor variance_545_axes_0 = const()[name = string("variance_545_axes_0"), val = tensor([1])]; + bool variance_545_keep_dims_0 = const()[name = string("variance_545_keep_dims_0"), val = bool(true)]; + tensor variance_545_cast_fp16 = reduce_mean(axes = variance_545_axes_0, keep_dims = variance_545_keep_dims_0, x = var_16378_cast_fp16)[name = string("variance_545_cast_fp16")]; + fp16 var_16381_to_fp16 = const()[name = string("op_16381_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16382_cast_fp16 = add(x = variance_545_cast_fp16, y = var_16381_to_fp16)[name = string("op_16382_cast_fp16")]; + fp32 var_16383_epsilon_0 = const()[name = string("op_16383_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16383_cast_fp16 = rsqrt(epsilon = var_16383_epsilon_0, x = var_16382_cast_fp16)[name = string("op_16383_cast_fp16")]; + tensor var_16384_cast_fp16 = mul(x = code_embed_47_cast_fp16, y = var_16383_cast_fp16)[name = string("op_16384_cast_fp16")]; + tensor input_699_cast_fp16 = mul(x = var_16384_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_699_cast_fp16")]; + string q_391_pad_type_0 = const()[name = string("q_391_pad_type_0"), val = string("valid")]; + tensor q_391_strides_0 = const()[name = string("q_391_strides_0"), val = tensor([1, 1])]; + tensor q_391_pad_0 = const()[name = string("q_391_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_391_dilations_0 = const()[name = string("q_391_dilations_0"), val = tensor([1, 1])]; + int32 q_391_groups_0 = const()[name = string("q_391_groups_0"), val = int32(1)]; + tensor q_391_cast_fp16 = conv(dilations = q_391_dilations_0, groups = q_391_groups_0, pad = q_391_pad_0, pad_type = q_391_pad_type_0, strides = q_391_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_699_cast_fp16)[name = string("q_391_cast_fp16")]; + string k_391_pad_type_0 = const()[name = string("k_391_pad_type_0"), val = string("valid")]; + tensor k_391_strides_0 = const()[name = string("k_391_strides_0"), val = tensor([1, 1])]; + tensor k_391_pad_0 = const()[name = string("k_391_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_391_dilations_0 = const()[name = string("k_391_dilations_0"), val = tensor([1, 1])]; + int32 k_391_groups_0 = const()[name = string("k_391_groups_0"), val = int32(1)]; + tensor k_391_cast_fp16 = conv(dilations = k_391_dilations_0, groups = k_391_groups_0, pad = k_391_pad_0, pad_type = k_391_pad_type_0, strides = k_391_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_699_cast_fp16)[name = string("k_391_cast_fp16")]; + string v_131_pad_type_0 = const()[name = string("v_131_pad_type_0"), val = string("valid")]; + tensor v_131_strides_0 = const()[name = string("v_131_strides_0"), val = tensor([1, 1])]; + tensor v_131_pad_0 = const()[name = string("v_131_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_131_dilations_0 = const()[name = string("v_131_dilations_0"), val = tensor([1, 1])]; + int32 v_131_groups_0 = const()[name = string("v_131_groups_0"), val = int32(1)]; + tensor v_131_cast_fp16 = conv(dilations = v_131_dilations_0, groups = v_131_groups_0, pad = v_131_pad_0, pad_type = v_131_pad_type_0, strides = v_131_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_699_cast_fp16)[name = string("v_131_cast_fp16")]; + tensor var_16418 = const()[name = string("op_16418"), val = tensor([16, 128, 1, 1])]; + tensor x_495_cast_fp16 = reshape(shape = var_16418, x = q_391_cast_fp16)[name = string("x_495_cast_fp16")]; + tensor var_16421_cast_fp16 = mul(x = x_495_cast_fp16, y = x_495_cast_fp16)[name = string("op_16421_cast_fp16")]; + tensor variance_547_axes_0 = const()[name = string("variance_547_axes_0"), val = tensor([1])]; + bool variance_547_keep_dims_0 = const()[name = string("variance_547_keep_dims_0"), val = bool(true)]; + tensor variance_547_cast_fp16 = reduce_mean(axes = variance_547_axes_0, keep_dims = variance_547_keep_dims_0, x = var_16421_cast_fp16)[name = string("variance_547_cast_fp16")]; + fp16 var_16424_to_fp16 = const()[name = string("op_16424_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16425_cast_fp16 = add(x = variance_547_cast_fp16, y = var_16424_to_fp16)[name = string("op_16425_cast_fp16")]; + fp32 var_16426_epsilon_0 = const()[name = string("op_16426_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16426_cast_fp16 = rsqrt(epsilon = var_16426_epsilon_0, x = var_16425_cast_fp16)[name = string("op_16426_cast_fp16")]; + tensor var_16427_cast_fp16 = mul(x = x_495_cast_fp16, y = var_16426_cast_fp16)[name = string("op_16427_cast_fp16")]; + tensor q_393_cast_fp16 = mul(x = var_16427_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_393_cast_fp16")]; + tensor var_16429 = const()[name = string("op_16429"), val = tensor([8, 128, 1, 1])]; + tensor x_497_cast_fp16 = reshape(shape = var_16429, x = k_391_cast_fp16)[name = string("x_497_cast_fp16")]; + tensor var_16432_cast_fp16 = mul(x = x_497_cast_fp16, y = x_497_cast_fp16)[name = string("op_16432_cast_fp16")]; + tensor variance_549_axes_0 = const()[name = string("variance_549_axes_0"), val = tensor([1])]; + bool variance_549_keep_dims_0 = const()[name = string("variance_549_keep_dims_0"), val = bool(true)]; + tensor variance_549_cast_fp16 = reduce_mean(axes = variance_549_axes_0, keep_dims = variance_549_keep_dims_0, x = var_16432_cast_fp16)[name = string("variance_549_cast_fp16")]; + fp16 var_16435_to_fp16 = const()[name = string("op_16435_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16436_cast_fp16 = add(x = variance_549_cast_fp16, y = var_16435_to_fp16)[name = string("op_16436_cast_fp16")]; + fp32 var_16437_epsilon_0 = const()[name = string("op_16437_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16437_cast_fp16 = rsqrt(epsilon = var_16437_epsilon_0, x = var_16436_cast_fp16)[name = string("op_16437_cast_fp16")]; + tensor var_16438_cast_fp16 = mul(x = x_497_cast_fp16, y = var_16437_cast_fp16)[name = string("op_16438_cast_fp16")]; + tensor k_393_cast_fp16 = mul(x = var_16438_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_393_cast_fp16")]; + tensor var_16440 = const()[name = string("op_16440"), val = tensor([1, 16, 128, 1])]; + tensor z_261_cast_fp16 = reshape(shape = var_16440, x = q_393_cast_fp16)[name = string("z_261_cast_fp16")]; + tensor var_16442 = const()[name = string("op_16442"), val = tensor([1, 8, 128, 1])]; + tensor z_263_cast_fp16 = reshape(shape = var_16442, x = k_393_cast_fp16)[name = string("z_263_cast_fp16")]; + tensor z1_261_begin_0 = const()[name = string("z1_261_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_261_end_0 = const()[name = string("z1_261_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_261_end_mask_0 = const()[name = string("z1_261_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_261_cast_fp16 = slice_by_index(begin = z1_261_begin_0, end = z1_261_end_0, end_mask = z1_261_end_mask_0, x = z_261_cast_fp16)[name = string("z1_261_cast_fp16")]; + tensor z2_261_begin_0 = const()[name = string("z2_261_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_261_end_0 = const()[name = string("z2_261_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_261_end_mask_0 = const()[name = string("z2_261_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_261_cast_fp16 = slice_by_index(begin = z2_261_begin_0, end = z2_261_end_0, end_mask = z2_261_end_mask_0, x = z_261_cast_fp16)[name = string("z2_261_cast_fp16")]; + tensor cos_131_to_fp16 = const()[name = string("cos_131_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644672)))]; + tensor var_16450_cast_fp16 = mul(x = z_261_cast_fp16, y = cos_131_to_fp16)[name = string("op_16450_cast_fp16")]; + fp16 const_144_promoted_to_fp16 = const()[name = string("const_144_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16451_cast_fp16 = mul(x = z2_261_cast_fp16, y = const_144_promoted_to_fp16)[name = string("op_16451_cast_fp16")]; + bool var_16453_interleave_0 = const()[name = string("op_16453_interleave_0"), val = bool(false)]; + tensor var_16453_cast_fp16 = concat(axis = var_16359, interleave = var_16453_interleave_0, values = (var_16451_cast_fp16, z1_261_cast_fp16))[name = string("op_16453_cast_fp16")]; + tensor sin_131_to_fp16 = const()[name = string("sin_131_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141644992)))]; + tensor var_16454_cast_fp16 = mul(x = var_16453_cast_fp16, y = sin_131_to_fp16)[name = string("op_16454_cast_fp16")]; + tensor q_395_cast_fp16 = add(x = var_16450_cast_fp16, y = var_16454_cast_fp16)[name = string("q_395_cast_fp16")]; + tensor z1_263_begin_0 = const()[name = string("z1_263_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_263_end_0 = const()[name = string("z1_263_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_263_end_mask_0 = const()[name = string("z1_263_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_263_cast_fp16 = slice_by_index(begin = z1_263_begin_0, end = z1_263_end_0, end_mask = z1_263_end_mask_0, x = z_263_cast_fp16)[name = string("z1_263_cast_fp16")]; + tensor z2_263_begin_0 = const()[name = string("z2_263_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_263_end_0 = const()[name = string("z2_263_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_263_end_mask_0 = const()[name = string("z2_263_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_263_cast_fp16 = slice_by_index(begin = z2_263_begin_0, end = z2_263_end_0, end_mask = z2_263_end_mask_0, x = z_263_cast_fp16)[name = string("z2_263_cast_fp16")]; + tensor var_16462_cast_fp16 = mul(x = z_263_cast_fp16, y = cos_131_to_fp16)[name = string("op_16462_cast_fp16")]; + fp16 const_145_promoted_to_fp16 = const()[name = string("const_145_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16463_cast_fp16 = mul(x = z2_263_cast_fp16, y = const_145_promoted_to_fp16)[name = string("op_16463_cast_fp16")]; + bool var_16465_interleave_0 = const()[name = string("op_16465_interleave_0"), val = bool(false)]; + tensor var_16465_cast_fp16 = concat(axis = var_16359, interleave = var_16465_interleave_0, values = (var_16463_cast_fp16, z1_263_cast_fp16))[name = string("op_16465_cast_fp16")]; + tensor var_16466_cast_fp16 = mul(x = var_16465_cast_fp16, y = sin_131_to_fp16)[name = string("op_16466_cast_fp16")]; + tensor k_395_cast_fp16 = add(x = var_16462_cast_fp16, y = var_16466_cast_fp16)[name = string("k_395_cast_fp16")]; + tensor var_16468 = const()[name = string("op_16468"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_131_cast_fp16 = reshape(shape = var_16468, x = k_395_cast_fp16)[name = string("cur_key_131_cast_fp16")]; + tensor var_16470_to_fp16 = const()[name = string("op_16470_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645312)))]; + tensor var_16471_cast_fp16 = mul(x = key_cache_131_cast_fp16, y = var_16470_to_fp16)[name = string("op_16471_cast_fp16")]; + tensor upd_131_to_fp16 = const()[name = string("upd_131_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645440)))]; + tensor var_16472_cast_fp16 = mul(x = cur_key_131_cast_fp16, y = upd_131_to_fp16)[name = string("op_16472_cast_fp16")]; + tensor key_131_cast_fp16 = add(x = var_16471_cast_fp16, y = var_16472_cast_fp16)[name = string("key_131_cast_fp16")]; + tensor var_16474_to_fp16 = const()[name = string("op_16474_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645312)))]; + tensor var_16475_cast_fp16 = mul(x = value_cache_131_cast_fp16, y = var_16474_to_fp16)[name = string("op_16475_cast_fp16")]; + tensor var_16476_cast_fp16 = mul(x = v_131_cast_fp16, y = upd_131_to_fp16)[name = string("op_16476_cast_fp16")]; + tensor value_131_cast_fp16 = add(x = var_16475_cast_fp16, y = var_16476_cast_fp16)[name = string("value_131_cast_fp16")]; + tensor var_16478 = const()[name = string("op_16478"), val = tensor([1, 8, 128, 16])]; + tensor kh_261_cast_fp16 = reshape(shape = var_16478, x = key_131_cast_fp16)[name = string("kh_261_cast_fp16")]; + tensor var_16480 = const()[name = string("op_16480"), val = tensor([1, 8, 128, 16])]; + tensor vh_261_cast_fp16 = reshape(shape = var_16480, x = value_131_cast_fp16)[name = string("vh_261_cast_fp16")]; + tensor transpose_260_perm_0 = const()[name = string("transpose_260_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_130_reps_0 = const()[name = string("tile_130_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_260_cast_fp16 = transpose(perm = transpose_260_perm_0, x = kh_261_cast_fp16)[name = string("transpose_89")]; + tensor tile_130_cast_fp16 = tile(reps = tile_130_reps_0, x = transpose_260_cast_fp16)[name = string("tile_130_cast_fp16")]; + tensor concat_328 = const()[name = string("concat_328"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_260_cast_fp16 = reshape(shape = concat_328, x = tile_130_cast_fp16)[name = string("reshape_260_cast_fp16")]; + tensor transpose_261_perm_0 = const()[name = string("transpose_261_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_329 = const()[name = string("concat_329"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_261_cast_fp16 = transpose(perm = transpose_261_perm_0, x = reshape_260_cast_fp16)[name = string("transpose_88")]; + tensor reshape_261_cast_fp16 = reshape(shape = concat_329, x = transpose_261_cast_fp16)[name = string("reshape_261_cast_fp16")]; + tensor transpose_262_perm_0 = const()[name = string("transpose_262_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_131_reps_0 = const()[name = string("tile_131_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_262_cast_fp16 = transpose(perm = transpose_262_perm_0, x = vh_261_cast_fp16)[name = string("transpose_87")]; + tensor tile_131_cast_fp16 = tile(reps = tile_131_reps_0, x = transpose_262_cast_fp16)[name = string("tile_131_cast_fp16")]; + tensor concat_330 = const()[name = string("concat_330"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_262_cast_fp16 = reshape(shape = concat_330, x = tile_131_cast_fp16)[name = string("reshape_262_cast_fp16")]; + tensor transpose_263_perm_0 = const()[name = string("transpose_263_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_331 = const()[name = string("concat_331"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_263_cast_fp16 = transpose(perm = transpose_263_perm_0, x = reshape_262_cast_fp16)[name = string("transpose_86")]; + tensor reshape_263_cast_fp16 = reshape(shape = concat_331, x = transpose_263_cast_fp16)[name = string("reshape_263_cast_fp16")]; + fp16 var_16484_to_fp16 = const()[name = string("op_16484_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_16485_cast_fp16 = mul(x = q_395_cast_fp16, y = var_16484_to_fp16)[name = string("op_16485_cast_fp16")]; + tensor transpose_577_perm_0 = const()[name = string("transpose_577_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_285_transpose_x_1 = const()[name = string("w_285_transpose_x_1"), val = bool(true)]; + bool w_285_transpose_y_1 = const()[name = string("w_285_transpose_y_1"), val = bool(false)]; + tensor transpose_577_cast_fp16 = transpose(perm = transpose_577_perm_0, x = reshape_261_cast_fp16)[name = string("transpose_85")]; + tensor w_285_cast_fp16 = matmul(transpose_x = w_285_transpose_x_1, transpose_y = w_285_transpose_y_1, x = var_16485_cast_fp16, y = transpose_577_cast_fp16)[name = string("w_285_cast_fp16")]; + tensor pad_131_to_fp16 = const()[name = string("pad_131_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645568)))]; + tensor var_16488_cast_fp16 = add(x = w_285_cast_fp16, y = pad_131_to_fp16)[name = string("op_16488_cast_fp16")]; + tensor w_287_cast_fp16 = softmax(axis = var_16363, x = var_16488_cast_fp16)[name = string("w_287_cast_fp16")]; + tensor transpose_578_perm_0 = const()[name = string("transpose_578_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_131_transpose_x_1 = const()[name = string("attn_131_transpose_x_1"), val = bool(false)]; + bool attn_131_transpose_y_1 = const()[name = string("attn_131_transpose_y_1"), val = bool(true)]; + tensor transpose_578_cast_fp16 = transpose(perm = transpose_578_perm_0, x = reshape_263_cast_fp16)[name = string("transpose_84")]; + tensor attn_131_cast_fp16 = matmul(transpose_x = attn_131_transpose_x_1, transpose_y = attn_131_transpose_y_1, x = transpose_578_cast_fp16, y = w_287_cast_fp16)[name = string("attn_131_cast_fp16")]; + tensor var_16492 = const()[name = string("op_16492"), val = tensor([1, 2048, 1, 1])]; + tensor input_701_cast_fp16 = reshape(shape = var_16492, x = attn_131_cast_fp16)[name = string("input_701_cast_fp16")]; + string attn_output_131_pad_type_0 = const()[name = string("attn_output_131_pad_type_0"), val = string("valid")]; + tensor attn_output_131_strides_0 = const()[name = string("attn_output_131_strides_0"), val = tensor([1, 1])]; + tensor attn_output_131_pad_0 = const()[name = string("attn_output_131_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_131_dilations_0 = const()[name = string("attn_output_131_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_131_groups_0 = const()[name = string("attn_output_131_groups_0"), val = int32(1)]; + tensor attn_output_131_cast_fp16 = conv(dilations = attn_output_131_dilations_0, groups = attn_output_131_groups_0, pad = attn_output_131_pad_0, pad_type = attn_output_131_pad_type_0, strides = attn_output_131_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_701_cast_fp16)[name = string("attn_output_131_cast_fp16")]; + tensor x_499_cast_fp16 = add(x = code_embed_47_cast_fp16, y = attn_output_131_cast_fp16)[name = string("x_499_cast_fp16")]; + tensor var_16506_cast_fp16 = mul(x = x_499_cast_fp16, y = x_499_cast_fp16)[name = string("op_16506_cast_fp16")]; + tensor variance_551_axes_0 = const()[name = string("variance_551_axes_0"), val = tensor([1])]; + bool variance_551_keep_dims_0 = const()[name = string("variance_551_keep_dims_0"), val = bool(true)]; + tensor variance_551_cast_fp16 = reduce_mean(axes = variance_551_axes_0, keep_dims = variance_551_keep_dims_0, x = var_16506_cast_fp16)[name = string("variance_551_cast_fp16")]; + fp16 var_16509_to_fp16 = const()[name = string("op_16509_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16510_cast_fp16 = add(x = variance_551_cast_fp16, y = var_16509_to_fp16)[name = string("op_16510_cast_fp16")]; + fp32 var_16511_epsilon_0 = const()[name = string("op_16511_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16511_cast_fp16 = rsqrt(epsilon = var_16511_epsilon_0, x = var_16510_cast_fp16)[name = string("op_16511_cast_fp16")]; + tensor var_16512_cast_fp16 = mul(x = x_499_cast_fp16, y = var_16511_cast_fp16)[name = string("op_16512_cast_fp16")]; + tensor input_703_cast_fp16 = mul(x = var_16512_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_703_cast_fp16")]; + string input_705_pad_type_0 = const()[name = string("input_705_pad_type_0"), val = string("valid")]; + tensor input_705_strides_0 = const()[name = string("input_705_strides_0"), val = tensor([1, 1])]; + tensor input_705_pad_0 = const()[name = string("input_705_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_705_dilations_0 = const()[name = string("input_705_dilations_0"), val = tensor([1, 1])]; + int32 input_705_groups_0 = const()[name = string("input_705_groups_0"), val = int32(1)]; + tensor input_705_cast_fp16 = conv(dilations = input_705_dilations_0, groups = input_705_groups_0, pad = input_705_pad_0, pad_type = input_705_pad_type_0, strides = input_705_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_703_cast_fp16)[name = string("input_705_cast_fp16")]; + tensor var_16520_cast_fp16 = silu(x = input_705_cast_fp16)[name = string("op_16520_cast_fp16")]; + string var_16526_pad_type_0 = const()[name = string("op_16526_pad_type_0"), val = string("valid")]; + tensor var_16526_strides_0 = const()[name = string("op_16526_strides_0"), val = tensor([1, 1])]; + tensor var_16526_pad_0 = const()[name = string("op_16526_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16526_dilations_0 = const()[name = string("op_16526_dilations_0"), val = tensor([1, 1])]; + int32 var_16526_groups_0 = const()[name = string("op_16526_groups_0"), val = int32(1)]; + tensor var_16526_cast_fp16 = conv(dilations = var_16526_dilations_0, groups = var_16526_groups_0, pad = var_16526_pad_0, pad_type = var_16526_pad_type_0, strides = var_16526_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_703_cast_fp16)[name = string("op_16526_cast_fp16")]; + tensor input_707_cast_fp16 = mul(x = var_16520_cast_fp16, y = var_16526_cast_fp16)[name = string("input_707_cast_fp16")]; + string h_131_pad_type_0 = const()[name = string("h_131_pad_type_0"), val = string("valid")]; + tensor h_131_strides_0 = const()[name = string("h_131_strides_0"), val = tensor([1, 1])]; + tensor h_131_pad_0 = const()[name = string("h_131_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_131_dilations_0 = const()[name = string("h_131_dilations_0"), val = tensor([1, 1])]; + int32 h_131_groups_0 = const()[name = string("h_131_groups_0"), val = int32(1)]; + tensor h_131_cast_fp16 = conv(dilations = h_131_dilations_0, groups = h_131_groups_0, pad = h_131_pad_0, pad_type = h_131_pad_type_0, strides = h_131_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_707_cast_fp16)[name = string("h_131_cast_fp16")]; + tensor x_501_cast_fp16 = add(x = x_499_cast_fp16, y = h_131_cast_fp16)[name = string("x_501_cast_fp16")]; + tensor key_cache_133_begin_0 = const()[name = string("key_cache_133_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_133_end_0 = const()[name = string("key_cache_133_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_133_end_mask_0 = const()[name = string("key_cache_133_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_133_cast_fp16 = slice_by_index(begin = key_cache_133_begin_0, end = key_cache_133_end_0, end_mask = key_cache_133_end_mask_0, x = layer_key_caches_27_cast_fp16)[name = string("key_cache_133_cast_fp16")]; + tensor value_cache_133_begin_0 = const()[name = string("value_cache_133_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_133_end_0 = const()[name = string("value_cache_133_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_133_end_mask_0 = const()[name = string("value_cache_133_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_133_cast_fp16 = slice_by_index(begin = value_cache_133_begin_0, end = value_cache_133_end_0, end_mask = value_cache_133_end_mask_0, x = layer_value_caches_27_cast_fp16)[name = string("value_cache_133_cast_fp16")]; + int32 var_16579 = const()[name = string("op_16579"), val = int32(2)]; + int32 var_16583 = const()[name = string("op_16583"), val = int32(3)]; + tensor var_16598_cast_fp16 = mul(x = x_501_cast_fp16, y = x_501_cast_fp16)[name = string("op_16598_cast_fp16")]; + tensor variance_553_axes_0 = const()[name = string("variance_553_axes_0"), val = tensor([1])]; + bool variance_553_keep_dims_0 = const()[name = string("variance_553_keep_dims_0"), val = bool(true)]; + tensor variance_553_cast_fp16 = reduce_mean(axes = variance_553_axes_0, keep_dims = variance_553_keep_dims_0, x = var_16598_cast_fp16)[name = string("variance_553_cast_fp16")]; + fp16 var_16601_to_fp16 = const()[name = string("op_16601_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16602_cast_fp16 = add(x = variance_553_cast_fp16, y = var_16601_to_fp16)[name = string("op_16602_cast_fp16")]; + fp32 var_16603_epsilon_0 = const()[name = string("op_16603_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16603_cast_fp16 = rsqrt(epsilon = var_16603_epsilon_0, x = var_16602_cast_fp16)[name = string("op_16603_cast_fp16")]; + tensor var_16604_cast_fp16 = mul(x = x_501_cast_fp16, y = var_16603_cast_fp16)[name = string("op_16604_cast_fp16")]; + tensor input_709_cast_fp16 = mul(x = var_16604_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_709_cast_fp16")]; + string q_397_pad_type_0 = const()[name = string("q_397_pad_type_0"), val = string("valid")]; + tensor q_397_strides_0 = const()[name = string("q_397_strides_0"), val = tensor([1, 1])]; + tensor q_397_pad_0 = const()[name = string("q_397_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_397_dilations_0 = const()[name = string("q_397_dilations_0"), val = tensor([1, 1])]; + int32 q_397_groups_0 = const()[name = string("q_397_groups_0"), val = int32(1)]; + tensor q_397_cast_fp16 = conv(dilations = q_397_dilations_0, groups = q_397_groups_0, pad = q_397_pad_0, pad_type = q_397_pad_type_0, strides = q_397_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_709_cast_fp16)[name = string("q_397_cast_fp16")]; + string k_397_pad_type_0 = const()[name = string("k_397_pad_type_0"), val = string("valid")]; + tensor k_397_strides_0 = const()[name = string("k_397_strides_0"), val = tensor([1, 1])]; + tensor k_397_pad_0 = const()[name = string("k_397_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_397_dilations_0 = const()[name = string("k_397_dilations_0"), val = tensor([1, 1])]; + int32 k_397_groups_0 = const()[name = string("k_397_groups_0"), val = int32(1)]; + tensor k_397_cast_fp16 = conv(dilations = k_397_dilations_0, groups = k_397_groups_0, pad = k_397_pad_0, pad_type = k_397_pad_type_0, strides = k_397_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_709_cast_fp16)[name = string("k_397_cast_fp16")]; + string v_133_pad_type_0 = const()[name = string("v_133_pad_type_0"), val = string("valid")]; + tensor v_133_strides_0 = const()[name = string("v_133_strides_0"), val = tensor([1, 1])]; + tensor v_133_pad_0 = const()[name = string("v_133_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_133_dilations_0 = const()[name = string("v_133_dilations_0"), val = tensor([1, 1])]; + int32 v_133_groups_0 = const()[name = string("v_133_groups_0"), val = int32(1)]; + tensor v_133_cast_fp16 = conv(dilations = v_133_dilations_0, groups = v_133_groups_0, pad = v_133_pad_0, pad_type = v_133_pad_type_0, strides = v_133_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_709_cast_fp16)[name = string("v_133_cast_fp16")]; + tensor var_16638 = const()[name = string("op_16638"), val = tensor([16, 128, 1, 1])]; + tensor x_503_cast_fp16 = reshape(shape = var_16638, x = q_397_cast_fp16)[name = string("x_503_cast_fp16")]; + tensor var_16641_cast_fp16 = mul(x = x_503_cast_fp16, y = x_503_cast_fp16)[name = string("op_16641_cast_fp16")]; + tensor variance_555_axes_0 = const()[name = string("variance_555_axes_0"), val = tensor([1])]; + bool variance_555_keep_dims_0 = const()[name = string("variance_555_keep_dims_0"), val = bool(true)]; + tensor variance_555_cast_fp16 = reduce_mean(axes = variance_555_axes_0, keep_dims = variance_555_keep_dims_0, x = var_16641_cast_fp16)[name = string("variance_555_cast_fp16")]; + fp16 var_16644_to_fp16 = const()[name = string("op_16644_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16645_cast_fp16 = add(x = variance_555_cast_fp16, y = var_16644_to_fp16)[name = string("op_16645_cast_fp16")]; + fp32 var_16646_epsilon_0 = const()[name = string("op_16646_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16646_cast_fp16 = rsqrt(epsilon = var_16646_epsilon_0, x = var_16645_cast_fp16)[name = string("op_16646_cast_fp16")]; + tensor var_16647_cast_fp16 = mul(x = x_503_cast_fp16, y = var_16646_cast_fp16)[name = string("op_16647_cast_fp16")]; + tensor q_399_cast_fp16 = mul(x = var_16647_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_399_cast_fp16")]; + tensor var_16649 = const()[name = string("op_16649"), val = tensor([8, 128, 1, 1])]; + tensor x_505_cast_fp16 = reshape(shape = var_16649, x = k_397_cast_fp16)[name = string("x_505_cast_fp16")]; + tensor var_16652_cast_fp16 = mul(x = x_505_cast_fp16, y = x_505_cast_fp16)[name = string("op_16652_cast_fp16")]; + tensor variance_557_axes_0 = const()[name = string("variance_557_axes_0"), val = tensor([1])]; + bool variance_557_keep_dims_0 = const()[name = string("variance_557_keep_dims_0"), val = bool(true)]; + tensor variance_557_cast_fp16 = reduce_mean(axes = variance_557_axes_0, keep_dims = variance_557_keep_dims_0, x = var_16652_cast_fp16)[name = string("variance_557_cast_fp16")]; + fp16 var_16655_to_fp16 = const()[name = string("op_16655_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16656_cast_fp16 = add(x = variance_557_cast_fp16, y = var_16655_to_fp16)[name = string("op_16656_cast_fp16")]; + fp32 var_16657_epsilon_0 = const()[name = string("op_16657_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16657_cast_fp16 = rsqrt(epsilon = var_16657_epsilon_0, x = var_16656_cast_fp16)[name = string("op_16657_cast_fp16")]; + tensor var_16658_cast_fp16 = mul(x = x_505_cast_fp16, y = var_16657_cast_fp16)[name = string("op_16658_cast_fp16")]; + tensor k_399_cast_fp16 = mul(x = var_16658_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_399_cast_fp16")]; + tensor var_16660 = const()[name = string("op_16660"), val = tensor([1, 16, 128, 1])]; + tensor z_265_cast_fp16 = reshape(shape = var_16660, x = q_399_cast_fp16)[name = string("z_265_cast_fp16")]; + tensor var_16662 = const()[name = string("op_16662"), val = tensor([1, 8, 128, 1])]; + tensor z_267_cast_fp16 = reshape(shape = var_16662, x = k_399_cast_fp16)[name = string("z_267_cast_fp16")]; + tensor z1_265_begin_0 = const()[name = string("z1_265_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_265_end_0 = const()[name = string("z1_265_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_265_end_mask_0 = const()[name = string("z1_265_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_265_cast_fp16 = slice_by_index(begin = z1_265_begin_0, end = z1_265_end_0, end_mask = z1_265_end_mask_0, x = z_265_cast_fp16)[name = string("z1_265_cast_fp16")]; + tensor z2_265_begin_0 = const()[name = string("z2_265_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_265_end_0 = const()[name = string("z2_265_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_265_end_mask_0 = const()[name = string("z2_265_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_265_cast_fp16 = slice_by_index(begin = z2_265_begin_0, end = z2_265_end_0, end_mask = z2_265_end_mask_0, x = z_265_cast_fp16)[name = string("z2_265_cast_fp16")]; + tensor var_16670_cast_fp16 = mul(x = z_265_cast_fp16, y = cos_131_to_fp16)[name = string("op_16670_cast_fp16")]; + fp16 const_146_promoted_to_fp16 = const()[name = string("const_146_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16671_cast_fp16 = mul(x = z2_265_cast_fp16, y = const_146_promoted_to_fp16)[name = string("op_16671_cast_fp16")]; + bool var_16673_interleave_0 = const()[name = string("op_16673_interleave_0"), val = bool(false)]; + tensor var_16673_cast_fp16 = concat(axis = var_16579, interleave = var_16673_interleave_0, values = (var_16671_cast_fp16, z1_265_cast_fp16))[name = string("op_16673_cast_fp16")]; + tensor var_16674_cast_fp16 = mul(x = var_16673_cast_fp16, y = sin_131_to_fp16)[name = string("op_16674_cast_fp16")]; + tensor q_401_cast_fp16 = add(x = var_16670_cast_fp16, y = var_16674_cast_fp16)[name = string("q_401_cast_fp16")]; + tensor z1_267_begin_0 = const()[name = string("z1_267_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_267_end_0 = const()[name = string("z1_267_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_267_end_mask_0 = const()[name = string("z1_267_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_267_cast_fp16 = slice_by_index(begin = z1_267_begin_0, end = z1_267_end_0, end_mask = z1_267_end_mask_0, x = z_267_cast_fp16)[name = string("z1_267_cast_fp16")]; + tensor z2_267_begin_0 = const()[name = string("z2_267_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_267_end_0 = const()[name = string("z2_267_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_267_end_mask_0 = const()[name = string("z2_267_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_267_cast_fp16 = slice_by_index(begin = z2_267_begin_0, end = z2_267_end_0, end_mask = z2_267_end_mask_0, x = z_267_cast_fp16)[name = string("z2_267_cast_fp16")]; + tensor var_16682_cast_fp16 = mul(x = z_267_cast_fp16, y = cos_131_to_fp16)[name = string("op_16682_cast_fp16")]; + fp16 const_147_promoted_to_fp16 = const()[name = string("const_147_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16683_cast_fp16 = mul(x = z2_267_cast_fp16, y = const_147_promoted_to_fp16)[name = string("op_16683_cast_fp16")]; + bool var_16685_interleave_0 = const()[name = string("op_16685_interleave_0"), val = bool(false)]; + tensor var_16685_cast_fp16 = concat(axis = var_16579, interleave = var_16685_interleave_0, values = (var_16683_cast_fp16, z1_267_cast_fp16))[name = string("op_16685_cast_fp16")]; + tensor var_16686_cast_fp16 = mul(x = var_16685_cast_fp16, y = sin_131_to_fp16)[name = string("op_16686_cast_fp16")]; + tensor k_401_cast_fp16 = add(x = var_16682_cast_fp16, y = var_16686_cast_fp16)[name = string("k_401_cast_fp16")]; + tensor var_16688 = const()[name = string("op_16688"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_133_cast_fp16 = reshape(shape = var_16688, x = k_401_cast_fp16)[name = string("cur_key_133_cast_fp16")]; + tensor var_16690_to_fp16 = const()[name = string("op_16690_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645312)))]; + tensor var_16691_cast_fp16 = mul(x = key_cache_133_cast_fp16, y = var_16690_to_fp16)[name = string("op_16691_cast_fp16")]; + tensor upd_133_to_fp16 = const()[name = string("upd_133_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645440)))]; + tensor var_16692_cast_fp16 = mul(x = cur_key_133_cast_fp16, y = upd_133_to_fp16)[name = string("op_16692_cast_fp16")]; + tensor key_133_cast_fp16 = add(x = var_16691_cast_fp16, y = var_16692_cast_fp16)[name = string("key_133_cast_fp16")]; + tensor var_16694_to_fp16 = const()[name = string("op_16694_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645312)))]; + tensor var_16695_cast_fp16 = mul(x = value_cache_133_cast_fp16, y = var_16694_to_fp16)[name = string("op_16695_cast_fp16")]; + tensor var_16696_cast_fp16 = mul(x = v_133_cast_fp16, y = upd_133_to_fp16)[name = string("op_16696_cast_fp16")]; + tensor value_133_cast_fp16 = add(x = var_16695_cast_fp16, y = var_16696_cast_fp16)[name = string("value_133_cast_fp16")]; + tensor var_16698 = const()[name = string("op_16698"), val = tensor([1, 8, 128, 16])]; + tensor kh_265_cast_fp16 = reshape(shape = var_16698, x = key_133_cast_fp16)[name = string("kh_265_cast_fp16")]; + tensor var_16700 = const()[name = string("op_16700"), val = tensor([1, 8, 128, 16])]; + tensor vh_265_cast_fp16 = reshape(shape = var_16700, x = value_133_cast_fp16)[name = string("vh_265_cast_fp16")]; + tensor transpose_264_perm_0 = const()[name = string("transpose_264_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_132_reps_0 = const()[name = string("tile_132_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_264_cast_fp16 = transpose(perm = transpose_264_perm_0, x = kh_265_cast_fp16)[name = string("transpose_83")]; + tensor tile_132_cast_fp16 = tile(reps = tile_132_reps_0, x = transpose_264_cast_fp16)[name = string("tile_132_cast_fp16")]; + tensor concat_332 = const()[name = string("concat_332"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_264_cast_fp16 = reshape(shape = concat_332, x = tile_132_cast_fp16)[name = string("reshape_264_cast_fp16")]; + tensor transpose_265_perm_0 = const()[name = string("transpose_265_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_333 = const()[name = string("concat_333"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_265_cast_fp16 = transpose(perm = transpose_265_perm_0, x = reshape_264_cast_fp16)[name = string("transpose_82")]; + tensor reshape_265_cast_fp16 = reshape(shape = concat_333, x = transpose_265_cast_fp16)[name = string("reshape_265_cast_fp16")]; + tensor transpose_266_perm_0 = const()[name = string("transpose_266_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_133_reps_0 = const()[name = string("tile_133_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_266_cast_fp16 = transpose(perm = transpose_266_perm_0, x = vh_265_cast_fp16)[name = string("transpose_81")]; + tensor tile_133_cast_fp16 = tile(reps = tile_133_reps_0, x = transpose_266_cast_fp16)[name = string("tile_133_cast_fp16")]; + tensor concat_334 = const()[name = string("concat_334"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_266_cast_fp16 = reshape(shape = concat_334, x = tile_133_cast_fp16)[name = string("reshape_266_cast_fp16")]; + tensor transpose_267_perm_0 = const()[name = string("transpose_267_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_335 = const()[name = string("concat_335"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_267_cast_fp16 = transpose(perm = transpose_267_perm_0, x = reshape_266_cast_fp16)[name = string("transpose_80")]; + tensor reshape_267_cast_fp16 = reshape(shape = concat_335, x = transpose_267_cast_fp16)[name = string("reshape_267_cast_fp16")]; + fp16 var_16704_to_fp16 = const()[name = string("op_16704_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_16705_cast_fp16 = mul(x = q_401_cast_fp16, y = var_16704_to_fp16)[name = string("op_16705_cast_fp16")]; + tensor transpose_581_perm_0 = const()[name = string("transpose_581_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_289_transpose_x_1 = const()[name = string("w_289_transpose_x_1"), val = bool(true)]; + bool w_289_transpose_y_1 = const()[name = string("w_289_transpose_y_1"), val = bool(false)]; + tensor transpose_581_cast_fp16 = transpose(perm = transpose_581_perm_0, x = reshape_265_cast_fp16)[name = string("transpose_79")]; + tensor w_289_cast_fp16 = matmul(transpose_x = w_289_transpose_x_1, transpose_y = w_289_transpose_y_1, x = var_16705_cast_fp16, y = transpose_581_cast_fp16)[name = string("w_289_cast_fp16")]; + tensor pad_133_to_fp16 = const()[name = string("pad_133_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645568)))]; + tensor var_16708_cast_fp16 = add(x = w_289_cast_fp16, y = pad_133_to_fp16)[name = string("op_16708_cast_fp16")]; + tensor w_291_cast_fp16 = softmax(axis = var_16583, x = var_16708_cast_fp16)[name = string("w_291_cast_fp16")]; + tensor transpose_582_perm_0 = const()[name = string("transpose_582_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_133_transpose_x_1 = const()[name = string("attn_133_transpose_x_1"), val = bool(false)]; + bool attn_133_transpose_y_1 = const()[name = string("attn_133_transpose_y_1"), val = bool(true)]; + tensor transpose_582_cast_fp16 = transpose(perm = transpose_582_perm_0, x = reshape_267_cast_fp16)[name = string("transpose_78")]; + tensor attn_133_cast_fp16 = matmul(transpose_x = attn_133_transpose_x_1, transpose_y = attn_133_transpose_y_1, x = transpose_582_cast_fp16, y = w_291_cast_fp16)[name = string("attn_133_cast_fp16")]; + tensor var_16712 = const()[name = string("op_16712"), val = tensor([1, 2048, 1, 1])]; + tensor input_711_cast_fp16 = reshape(shape = var_16712, x = attn_133_cast_fp16)[name = string("input_711_cast_fp16")]; + string attn_output_133_pad_type_0 = const()[name = string("attn_output_133_pad_type_0"), val = string("valid")]; + tensor attn_output_133_strides_0 = const()[name = string("attn_output_133_strides_0"), val = tensor([1, 1])]; + tensor attn_output_133_pad_0 = const()[name = string("attn_output_133_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_133_dilations_0 = const()[name = string("attn_output_133_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_133_groups_0 = const()[name = string("attn_output_133_groups_0"), val = int32(1)]; + tensor attn_output_133_cast_fp16 = conv(dilations = attn_output_133_dilations_0, groups = attn_output_133_groups_0, pad = attn_output_133_pad_0, pad_type = attn_output_133_pad_type_0, strides = attn_output_133_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_711_cast_fp16)[name = string("attn_output_133_cast_fp16")]; + tensor x_507_cast_fp16 = add(x = x_501_cast_fp16, y = attn_output_133_cast_fp16)[name = string("x_507_cast_fp16")]; + tensor var_16726_cast_fp16 = mul(x = x_507_cast_fp16, y = x_507_cast_fp16)[name = string("op_16726_cast_fp16")]; + tensor variance_559_axes_0 = const()[name = string("variance_559_axes_0"), val = tensor([1])]; + bool variance_559_keep_dims_0 = const()[name = string("variance_559_keep_dims_0"), val = bool(true)]; + tensor variance_559_cast_fp16 = reduce_mean(axes = variance_559_axes_0, keep_dims = variance_559_keep_dims_0, x = var_16726_cast_fp16)[name = string("variance_559_cast_fp16")]; + fp16 var_16729_to_fp16 = const()[name = string("op_16729_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16730_cast_fp16 = add(x = variance_559_cast_fp16, y = var_16729_to_fp16)[name = string("op_16730_cast_fp16")]; + fp32 var_16731_epsilon_0 = const()[name = string("op_16731_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16731_cast_fp16 = rsqrt(epsilon = var_16731_epsilon_0, x = var_16730_cast_fp16)[name = string("op_16731_cast_fp16")]; + tensor var_16732_cast_fp16 = mul(x = x_507_cast_fp16, y = var_16731_cast_fp16)[name = string("op_16732_cast_fp16")]; + tensor input_713_cast_fp16 = mul(x = var_16732_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_713_cast_fp16")]; + string input_715_pad_type_0 = const()[name = string("input_715_pad_type_0"), val = string("valid")]; + tensor input_715_strides_0 = const()[name = string("input_715_strides_0"), val = tensor([1, 1])]; + tensor input_715_pad_0 = const()[name = string("input_715_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_715_dilations_0 = const()[name = string("input_715_dilations_0"), val = tensor([1, 1])]; + int32 input_715_groups_0 = const()[name = string("input_715_groups_0"), val = int32(1)]; + tensor input_715_cast_fp16 = conv(dilations = input_715_dilations_0, groups = input_715_groups_0, pad = input_715_pad_0, pad_type = input_715_pad_type_0, strides = input_715_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_713_cast_fp16)[name = string("input_715_cast_fp16")]; + tensor var_16740_cast_fp16 = silu(x = input_715_cast_fp16)[name = string("op_16740_cast_fp16")]; + string var_16746_pad_type_0 = const()[name = string("op_16746_pad_type_0"), val = string("valid")]; + tensor var_16746_strides_0 = const()[name = string("op_16746_strides_0"), val = tensor([1, 1])]; + tensor var_16746_pad_0 = const()[name = string("op_16746_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16746_dilations_0 = const()[name = string("op_16746_dilations_0"), val = tensor([1, 1])]; + int32 var_16746_groups_0 = const()[name = string("op_16746_groups_0"), val = int32(1)]; + tensor var_16746_cast_fp16 = conv(dilations = var_16746_dilations_0, groups = var_16746_groups_0, pad = var_16746_pad_0, pad_type = var_16746_pad_type_0, strides = var_16746_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_713_cast_fp16)[name = string("op_16746_cast_fp16")]; + tensor input_717_cast_fp16 = mul(x = var_16740_cast_fp16, y = var_16746_cast_fp16)[name = string("input_717_cast_fp16")]; + string h_133_pad_type_0 = const()[name = string("h_133_pad_type_0"), val = string("valid")]; + tensor h_133_strides_0 = const()[name = string("h_133_strides_0"), val = tensor([1, 1])]; + tensor h_133_pad_0 = const()[name = string("h_133_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_133_dilations_0 = const()[name = string("h_133_dilations_0"), val = tensor([1, 1])]; + int32 h_133_groups_0 = const()[name = string("h_133_groups_0"), val = int32(1)]; + tensor h_133_cast_fp16 = conv(dilations = h_133_dilations_0, groups = h_133_groups_0, pad = h_133_pad_0, pad_type = h_133_pad_type_0, strides = h_133_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_717_cast_fp16)[name = string("h_133_cast_fp16")]; + tensor x_509_cast_fp16 = add(x = x_507_cast_fp16, y = h_133_cast_fp16)[name = string("x_509_cast_fp16")]; + tensor key_cache_135_begin_0 = const()[name = string("key_cache_135_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_135_end_0 = const()[name = string("key_cache_135_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_135_end_mask_0 = const()[name = string("key_cache_135_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_135_cast_fp16 = slice_by_index(begin = key_cache_135_begin_0, end = key_cache_135_end_0, end_mask = key_cache_135_end_mask_0, x = layer_key_caches_27_cast_fp16)[name = string("key_cache_135_cast_fp16")]; + tensor value_cache_135_begin_0 = const()[name = string("value_cache_135_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_135_end_0 = const()[name = string("value_cache_135_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_135_end_mask_0 = const()[name = string("value_cache_135_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_135_cast_fp16 = slice_by_index(begin = value_cache_135_begin_0, end = value_cache_135_end_0, end_mask = value_cache_135_end_mask_0, x = layer_value_caches_27_cast_fp16)[name = string("value_cache_135_cast_fp16")]; + int32 var_16799 = const()[name = string("op_16799"), val = int32(2)]; + int32 var_16803 = const()[name = string("op_16803"), val = int32(3)]; + tensor var_16818_cast_fp16 = mul(x = x_509_cast_fp16, y = x_509_cast_fp16)[name = string("op_16818_cast_fp16")]; + tensor variance_561_axes_0 = const()[name = string("variance_561_axes_0"), val = tensor([1])]; + bool variance_561_keep_dims_0 = const()[name = string("variance_561_keep_dims_0"), val = bool(true)]; + tensor variance_561_cast_fp16 = reduce_mean(axes = variance_561_axes_0, keep_dims = variance_561_keep_dims_0, x = var_16818_cast_fp16)[name = string("variance_561_cast_fp16")]; + fp16 var_16821_to_fp16 = const()[name = string("op_16821_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16822_cast_fp16 = add(x = variance_561_cast_fp16, y = var_16821_to_fp16)[name = string("op_16822_cast_fp16")]; + fp32 var_16823_epsilon_0 = const()[name = string("op_16823_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16823_cast_fp16 = rsqrt(epsilon = var_16823_epsilon_0, x = var_16822_cast_fp16)[name = string("op_16823_cast_fp16")]; + tensor var_16824_cast_fp16 = mul(x = x_509_cast_fp16, y = var_16823_cast_fp16)[name = string("op_16824_cast_fp16")]; + tensor input_719_cast_fp16 = mul(x = var_16824_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_719_cast_fp16")]; + string q_403_pad_type_0 = const()[name = string("q_403_pad_type_0"), val = string("valid")]; + tensor q_403_strides_0 = const()[name = string("q_403_strides_0"), val = tensor([1, 1])]; + tensor q_403_pad_0 = const()[name = string("q_403_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_403_dilations_0 = const()[name = string("q_403_dilations_0"), val = tensor([1, 1])]; + int32 q_403_groups_0 = const()[name = string("q_403_groups_0"), val = int32(1)]; + tensor q_403_cast_fp16 = conv(dilations = q_403_dilations_0, groups = q_403_groups_0, pad = q_403_pad_0, pad_type = q_403_pad_type_0, strides = q_403_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_719_cast_fp16)[name = string("q_403_cast_fp16")]; + string k_403_pad_type_0 = const()[name = string("k_403_pad_type_0"), val = string("valid")]; + tensor k_403_strides_0 = const()[name = string("k_403_strides_0"), val = tensor([1, 1])]; + tensor k_403_pad_0 = const()[name = string("k_403_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_403_dilations_0 = const()[name = string("k_403_dilations_0"), val = tensor([1, 1])]; + int32 k_403_groups_0 = const()[name = string("k_403_groups_0"), val = int32(1)]; + tensor k_403_cast_fp16 = conv(dilations = k_403_dilations_0, groups = k_403_groups_0, pad = k_403_pad_0, pad_type = k_403_pad_type_0, strides = k_403_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_719_cast_fp16)[name = string("k_403_cast_fp16")]; + string v_135_pad_type_0 = const()[name = string("v_135_pad_type_0"), val = string("valid")]; + tensor v_135_strides_0 = const()[name = string("v_135_strides_0"), val = tensor([1, 1])]; + tensor v_135_pad_0 = const()[name = string("v_135_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_135_dilations_0 = const()[name = string("v_135_dilations_0"), val = tensor([1, 1])]; + int32 v_135_groups_0 = const()[name = string("v_135_groups_0"), val = int32(1)]; + tensor v_135_cast_fp16 = conv(dilations = v_135_dilations_0, groups = v_135_groups_0, pad = v_135_pad_0, pad_type = v_135_pad_type_0, strides = v_135_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_719_cast_fp16)[name = string("v_135_cast_fp16")]; + tensor var_16858 = const()[name = string("op_16858"), val = tensor([16, 128, 1, 1])]; + tensor x_511_cast_fp16 = reshape(shape = var_16858, x = q_403_cast_fp16)[name = string("x_511_cast_fp16")]; + tensor var_16861_cast_fp16 = mul(x = x_511_cast_fp16, y = x_511_cast_fp16)[name = string("op_16861_cast_fp16")]; + tensor variance_563_axes_0 = const()[name = string("variance_563_axes_0"), val = tensor([1])]; + bool variance_563_keep_dims_0 = const()[name = string("variance_563_keep_dims_0"), val = bool(true)]; + tensor variance_563_cast_fp16 = reduce_mean(axes = variance_563_axes_0, keep_dims = variance_563_keep_dims_0, x = var_16861_cast_fp16)[name = string("variance_563_cast_fp16")]; + fp16 var_16864_to_fp16 = const()[name = string("op_16864_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16865_cast_fp16 = add(x = variance_563_cast_fp16, y = var_16864_to_fp16)[name = string("op_16865_cast_fp16")]; + fp32 var_16866_epsilon_0 = const()[name = string("op_16866_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16866_cast_fp16 = rsqrt(epsilon = var_16866_epsilon_0, x = var_16865_cast_fp16)[name = string("op_16866_cast_fp16")]; + tensor var_16867_cast_fp16 = mul(x = x_511_cast_fp16, y = var_16866_cast_fp16)[name = string("op_16867_cast_fp16")]; + tensor q_405_cast_fp16 = mul(x = var_16867_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_405_cast_fp16")]; + tensor var_16869 = const()[name = string("op_16869"), val = tensor([8, 128, 1, 1])]; + tensor x_513_cast_fp16 = reshape(shape = var_16869, x = k_403_cast_fp16)[name = string("x_513_cast_fp16")]; + tensor var_16872_cast_fp16 = mul(x = x_513_cast_fp16, y = x_513_cast_fp16)[name = string("op_16872_cast_fp16")]; + tensor variance_565_axes_0 = const()[name = string("variance_565_axes_0"), val = tensor([1])]; + bool variance_565_keep_dims_0 = const()[name = string("variance_565_keep_dims_0"), val = bool(true)]; + tensor variance_565_cast_fp16 = reduce_mean(axes = variance_565_axes_0, keep_dims = variance_565_keep_dims_0, x = var_16872_cast_fp16)[name = string("variance_565_cast_fp16")]; + fp16 var_16875_to_fp16 = const()[name = string("op_16875_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16876_cast_fp16 = add(x = variance_565_cast_fp16, y = var_16875_to_fp16)[name = string("op_16876_cast_fp16")]; + fp32 var_16877_epsilon_0 = const()[name = string("op_16877_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16877_cast_fp16 = rsqrt(epsilon = var_16877_epsilon_0, x = var_16876_cast_fp16)[name = string("op_16877_cast_fp16")]; + tensor var_16878_cast_fp16 = mul(x = x_513_cast_fp16, y = var_16877_cast_fp16)[name = string("op_16878_cast_fp16")]; + tensor k_405_cast_fp16 = mul(x = var_16878_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_405_cast_fp16")]; + tensor var_16880 = const()[name = string("op_16880"), val = tensor([1, 16, 128, 1])]; + tensor z_269_cast_fp16 = reshape(shape = var_16880, x = q_405_cast_fp16)[name = string("z_269_cast_fp16")]; + tensor var_16882 = const()[name = string("op_16882"), val = tensor([1, 8, 128, 1])]; + tensor z_271_cast_fp16 = reshape(shape = var_16882, x = k_405_cast_fp16)[name = string("z_271_cast_fp16")]; + tensor z1_269_begin_0 = const()[name = string("z1_269_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_269_end_0 = const()[name = string("z1_269_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_269_end_mask_0 = const()[name = string("z1_269_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_269_cast_fp16 = slice_by_index(begin = z1_269_begin_0, end = z1_269_end_0, end_mask = z1_269_end_mask_0, x = z_269_cast_fp16)[name = string("z1_269_cast_fp16")]; + tensor z2_269_begin_0 = const()[name = string("z2_269_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_269_end_0 = const()[name = string("z2_269_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_269_end_mask_0 = const()[name = string("z2_269_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_269_cast_fp16 = slice_by_index(begin = z2_269_begin_0, end = z2_269_end_0, end_mask = z2_269_end_mask_0, x = z_269_cast_fp16)[name = string("z2_269_cast_fp16")]; + tensor var_16890_cast_fp16 = mul(x = z_269_cast_fp16, y = cos_131_to_fp16)[name = string("op_16890_cast_fp16")]; + fp16 const_148_promoted_to_fp16 = const()[name = string("const_148_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16891_cast_fp16 = mul(x = z2_269_cast_fp16, y = const_148_promoted_to_fp16)[name = string("op_16891_cast_fp16")]; + bool var_16893_interleave_0 = const()[name = string("op_16893_interleave_0"), val = bool(false)]; + tensor var_16893_cast_fp16 = concat(axis = var_16799, interleave = var_16893_interleave_0, values = (var_16891_cast_fp16, z1_269_cast_fp16))[name = string("op_16893_cast_fp16")]; + tensor var_16894_cast_fp16 = mul(x = var_16893_cast_fp16, y = sin_131_to_fp16)[name = string("op_16894_cast_fp16")]; + tensor q_407_cast_fp16 = add(x = var_16890_cast_fp16, y = var_16894_cast_fp16)[name = string("q_407_cast_fp16")]; + tensor z1_271_begin_0 = const()[name = string("z1_271_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_271_end_0 = const()[name = string("z1_271_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_271_end_mask_0 = const()[name = string("z1_271_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_271_cast_fp16 = slice_by_index(begin = z1_271_begin_0, end = z1_271_end_0, end_mask = z1_271_end_mask_0, x = z_271_cast_fp16)[name = string("z1_271_cast_fp16")]; + tensor z2_271_begin_0 = const()[name = string("z2_271_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_271_end_0 = const()[name = string("z2_271_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_271_end_mask_0 = const()[name = string("z2_271_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_271_cast_fp16 = slice_by_index(begin = z2_271_begin_0, end = z2_271_end_0, end_mask = z2_271_end_mask_0, x = z_271_cast_fp16)[name = string("z2_271_cast_fp16")]; + tensor var_16902_cast_fp16 = mul(x = z_271_cast_fp16, y = cos_131_to_fp16)[name = string("op_16902_cast_fp16")]; + fp16 const_149_promoted_to_fp16 = const()[name = string("const_149_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16903_cast_fp16 = mul(x = z2_271_cast_fp16, y = const_149_promoted_to_fp16)[name = string("op_16903_cast_fp16")]; + bool var_16905_interleave_0 = const()[name = string("op_16905_interleave_0"), val = bool(false)]; + tensor var_16905_cast_fp16 = concat(axis = var_16799, interleave = var_16905_interleave_0, values = (var_16903_cast_fp16, z1_271_cast_fp16))[name = string("op_16905_cast_fp16")]; + tensor var_16906_cast_fp16 = mul(x = var_16905_cast_fp16, y = sin_131_to_fp16)[name = string("op_16906_cast_fp16")]; + tensor k_407_cast_fp16 = add(x = var_16902_cast_fp16, y = var_16906_cast_fp16)[name = string("k_407_cast_fp16")]; + tensor var_16908 = const()[name = string("op_16908"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_135_cast_fp16 = reshape(shape = var_16908, x = k_407_cast_fp16)[name = string("cur_key_135_cast_fp16")]; + tensor var_16910_to_fp16 = const()[name = string("op_16910_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645312)))]; + tensor var_16911_cast_fp16 = mul(x = key_cache_135_cast_fp16, y = var_16910_to_fp16)[name = string("op_16911_cast_fp16")]; + tensor upd_135_to_fp16 = const()[name = string("upd_135_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645440)))]; + tensor var_16912_cast_fp16 = mul(x = cur_key_135_cast_fp16, y = upd_135_to_fp16)[name = string("op_16912_cast_fp16")]; + tensor key_135_cast_fp16 = add(x = var_16911_cast_fp16, y = var_16912_cast_fp16)[name = string("key_135_cast_fp16")]; + tensor var_16914_to_fp16 = const()[name = string("op_16914_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645312)))]; + tensor var_16915_cast_fp16 = mul(x = value_cache_135_cast_fp16, y = var_16914_to_fp16)[name = string("op_16915_cast_fp16")]; + tensor var_16916_cast_fp16 = mul(x = v_135_cast_fp16, y = upd_135_to_fp16)[name = string("op_16916_cast_fp16")]; + tensor value_135_cast_fp16 = add(x = var_16915_cast_fp16, y = var_16916_cast_fp16)[name = string("value_135_cast_fp16")]; + tensor var_16918 = const()[name = string("op_16918"), val = tensor([1, 8, 128, 16])]; + tensor kh_269_cast_fp16 = reshape(shape = var_16918, x = key_135_cast_fp16)[name = string("kh_269_cast_fp16")]; + tensor var_16920 = const()[name = string("op_16920"), val = tensor([1, 8, 128, 16])]; + tensor vh_269_cast_fp16 = reshape(shape = var_16920, x = value_135_cast_fp16)[name = string("vh_269_cast_fp16")]; + tensor transpose_268_perm_0 = const()[name = string("transpose_268_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_134_reps_0 = const()[name = string("tile_134_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_268_cast_fp16 = transpose(perm = transpose_268_perm_0, x = kh_269_cast_fp16)[name = string("transpose_77")]; + tensor tile_134_cast_fp16 = tile(reps = tile_134_reps_0, x = transpose_268_cast_fp16)[name = string("tile_134_cast_fp16")]; + tensor concat_336 = const()[name = string("concat_336"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_268_cast_fp16 = reshape(shape = concat_336, x = tile_134_cast_fp16)[name = string("reshape_268_cast_fp16")]; + tensor transpose_269_perm_0 = const()[name = string("transpose_269_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_337 = const()[name = string("concat_337"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_269_cast_fp16 = transpose(perm = transpose_269_perm_0, x = reshape_268_cast_fp16)[name = string("transpose_76")]; + tensor reshape_269_cast_fp16 = reshape(shape = concat_337, x = transpose_269_cast_fp16)[name = string("reshape_269_cast_fp16")]; + tensor transpose_270_perm_0 = const()[name = string("transpose_270_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_135_reps_0 = const()[name = string("tile_135_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_270_cast_fp16 = transpose(perm = transpose_270_perm_0, x = vh_269_cast_fp16)[name = string("transpose_75")]; + tensor tile_135_cast_fp16 = tile(reps = tile_135_reps_0, x = transpose_270_cast_fp16)[name = string("tile_135_cast_fp16")]; + tensor concat_338 = const()[name = string("concat_338"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_270_cast_fp16 = reshape(shape = concat_338, x = tile_135_cast_fp16)[name = string("reshape_270_cast_fp16")]; + tensor transpose_271_perm_0 = const()[name = string("transpose_271_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_339 = const()[name = string("concat_339"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_271_cast_fp16 = transpose(perm = transpose_271_perm_0, x = reshape_270_cast_fp16)[name = string("transpose_74")]; + tensor reshape_271_cast_fp16 = reshape(shape = concat_339, x = transpose_271_cast_fp16)[name = string("reshape_271_cast_fp16")]; + fp16 var_16924_to_fp16 = const()[name = string("op_16924_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_16925_cast_fp16 = mul(x = q_407_cast_fp16, y = var_16924_to_fp16)[name = string("op_16925_cast_fp16")]; + tensor transpose_585_perm_0 = const()[name = string("transpose_585_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_293_transpose_x_1 = const()[name = string("w_293_transpose_x_1"), val = bool(true)]; + bool w_293_transpose_y_1 = const()[name = string("w_293_transpose_y_1"), val = bool(false)]; + tensor transpose_585_cast_fp16 = transpose(perm = transpose_585_perm_0, x = reshape_269_cast_fp16)[name = string("transpose_73")]; + tensor w_293_cast_fp16 = matmul(transpose_x = w_293_transpose_x_1, transpose_y = w_293_transpose_y_1, x = var_16925_cast_fp16, y = transpose_585_cast_fp16)[name = string("w_293_cast_fp16")]; + tensor pad_135_to_fp16 = const()[name = string("pad_135_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645568)))]; + tensor var_16928_cast_fp16 = add(x = w_293_cast_fp16, y = pad_135_to_fp16)[name = string("op_16928_cast_fp16")]; + tensor w_295_cast_fp16 = softmax(axis = var_16803, x = var_16928_cast_fp16)[name = string("w_295_cast_fp16")]; + tensor transpose_586_perm_0 = const()[name = string("transpose_586_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_135_transpose_x_1 = const()[name = string("attn_135_transpose_x_1"), val = bool(false)]; + bool attn_135_transpose_y_1 = const()[name = string("attn_135_transpose_y_1"), val = bool(true)]; + tensor transpose_586_cast_fp16 = transpose(perm = transpose_586_perm_0, x = reshape_271_cast_fp16)[name = string("transpose_72")]; + tensor attn_135_cast_fp16 = matmul(transpose_x = attn_135_transpose_x_1, transpose_y = attn_135_transpose_y_1, x = transpose_586_cast_fp16, y = w_295_cast_fp16)[name = string("attn_135_cast_fp16")]; + tensor var_16932 = const()[name = string("op_16932"), val = tensor([1, 2048, 1, 1])]; + tensor input_721_cast_fp16 = reshape(shape = var_16932, x = attn_135_cast_fp16)[name = string("input_721_cast_fp16")]; + string attn_output_135_pad_type_0 = const()[name = string("attn_output_135_pad_type_0"), val = string("valid")]; + tensor attn_output_135_strides_0 = const()[name = string("attn_output_135_strides_0"), val = tensor([1, 1])]; + tensor attn_output_135_pad_0 = const()[name = string("attn_output_135_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_135_dilations_0 = const()[name = string("attn_output_135_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_135_groups_0 = const()[name = string("attn_output_135_groups_0"), val = int32(1)]; + tensor attn_output_135_cast_fp16 = conv(dilations = attn_output_135_dilations_0, groups = attn_output_135_groups_0, pad = attn_output_135_pad_0, pad_type = attn_output_135_pad_type_0, strides = attn_output_135_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_721_cast_fp16)[name = string("attn_output_135_cast_fp16")]; + tensor x_515_cast_fp16 = add(x = x_509_cast_fp16, y = attn_output_135_cast_fp16)[name = string("x_515_cast_fp16")]; + tensor var_16946_cast_fp16 = mul(x = x_515_cast_fp16, y = x_515_cast_fp16)[name = string("op_16946_cast_fp16")]; + tensor variance_567_axes_0 = const()[name = string("variance_567_axes_0"), val = tensor([1])]; + bool variance_567_keep_dims_0 = const()[name = string("variance_567_keep_dims_0"), val = bool(true)]; + tensor variance_567_cast_fp16 = reduce_mean(axes = variance_567_axes_0, keep_dims = variance_567_keep_dims_0, x = var_16946_cast_fp16)[name = string("variance_567_cast_fp16")]; + fp16 var_16949_to_fp16 = const()[name = string("op_16949_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16950_cast_fp16 = add(x = variance_567_cast_fp16, y = var_16949_to_fp16)[name = string("op_16950_cast_fp16")]; + fp32 var_16951_epsilon_0 = const()[name = string("op_16951_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16951_cast_fp16 = rsqrt(epsilon = var_16951_epsilon_0, x = var_16950_cast_fp16)[name = string("op_16951_cast_fp16")]; + tensor var_16952_cast_fp16 = mul(x = x_515_cast_fp16, y = var_16951_cast_fp16)[name = string("op_16952_cast_fp16")]; + tensor input_723_cast_fp16 = mul(x = var_16952_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_723_cast_fp16")]; + string input_725_pad_type_0 = const()[name = string("input_725_pad_type_0"), val = string("valid")]; + tensor input_725_strides_0 = const()[name = string("input_725_strides_0"), val = tensor([1, 1])]; + tensor input_725_pad_0 = const()[name = string("input_725_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_725_dilations_0 = const()[name = string("input_725_dilations_0"), val = tensor([1, 1])]; + int32 input_725_groups_0 = const()[name = string("input_725_groups_0"), val = int32(1)]; + tensor input_725_cast_fp16 = conv(dilations = input_725_dilations_0, groups = input_725_groups_0, pad = input_725_pad_0, pad_type = input_725_pad_type_0, strides = input_725_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_723_cast_fp16)[name = string("input_725_cast_fp16")]; + tensor var_16960_cast_fp16 = silu(x = input_725_cast_fp16)[name = string("op_16960_cast_fp16")]; + string var_16966_pad_type_0 = const()[name = string("op_16966_pad_type_0"), val = string("valid")]; + tensor var_16966_strides_0 = const()[name = string("op_16966_strides_0"), val = tensor([1, 1])]; + tensor var_16966_pad_0 = const()[name = string("op_16966_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16966_dilations_0 = const()[name = string("op_16966_dilations_0"), val = tensor([1, 1])]; + int32 var_16966_groups_0 = const()[name = string("op_16966_groups_0"), val = int32(1)]; + tensor var_16966_cast_fp16 = conv(dilations = var_16966_dilations_0, groups = var_16966_groups_0, pad = var_16966_pad_0, pad_type = var_16966_pad_type_0, strides = var_16966_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_723_cast_fp16)[name = string("op_16966_cast_fp16")]; + tensor input_727_cast_fp16 = mul(x = var_16960_cast_fp16, y = var_16966_cast_fp16)[name = string("input_727_cast_fp16")]; + string h_135_pad_type_0 = const()[name = string("h_135_pad_type_0"), val = string("valid")]; + tensor h_135_strides_0 = const()[name = string("h_135_strides_0"), val = tensor([1, 1])]; + tensor h_135_pad_0 = const()[name = string("h_135_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_135_dilations_0 = const()[name = string("h_135_dilations_0"), val = tensor([1, 1])]; + int32 h_135_groups_0 = const()[name = string("h_135_groups_0"), val = int32(1)]; + tensor h_135_cast_fp16 = conv(dilations = h_135_dilations_0, groups = h_135_groups_0, pad = h_135_pad_0, pad_type = h_135_pad_type_0, strides = h_135_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_727_cast_fp16)[name = string("h_135_cast_fp16")]; + tensor x_517_cast_fp16 = add(x = x_515_cast_fp16, y = h_135_cast_fp16)[name = string("x_517_cast_fp16")]; + tensor key_cache_137_begin_0 = const()[name = string("key_cache_137_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_137_end_0 = const()[name = string("key_cache_137_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_137_end_mask_0 = const()[name = string("key_cache_137_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_137_cast_fp16 = slice_by_index(begin = key_cache_137_begin_0, end = key_cache_137_end_0, end_mask = key_cache_137_end_mask_0, x = layer_key_caches_27_cast_fp16)[name = string("key_cache_137_cast_fp16")]; + tensor value_cache_137_begin_0 = const()[name = string("value_cache_137_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_137_end_0 = const()[name = string("value_cache_137_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_137_end_mask_0 = const()[name = string("value_cache_137_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_137_cast_fp16 = slice_by_index(begin = value_cache_137_begin_0, end = value_cache_137_end_0, end_mask = value_cache_137_end_mask_0, x = layer_value_caches_27_cast_fp16)[name = string("value_cache_137_cast_fp16")]; + int32 var_17019 = const()[name = string("op_17019"), val = int32(2)]; + int32 var_17023 = const()[name = string("op_17023"), val = int32(3)]; + tensor var_17038_cast_fp16 = mul(x = x_517_cast_fp16, y = x_517_cast_fp16)[name = string("op_17038_cast_fp16")]; + tensor variance_569_axes_0 = const()[name = string("variance_569_axes_0"), val = tensor([1])]; + bool variance_569_keep_dims_0 = const()[name = string("variance_569_keep_dims_0"), val = bool(true)]; + tensor variance_569_cast_fp16 = reduce_mean(axes = variance_569_axes_0, keep_dims = variance_569_keep_dims_0, x = var_17038_cast_fp16)[name = string("variance_569_cast_fp16")]; + fp16 var_17041_to_fp16 = const()[name = string("op_17041_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17042_cast_fp16 = add(x = variance_569_cast_fp16, y = var_17041_to_fp16)[name = string("op_17042_cast_fp16")]; + fp32 var_17043_epsilon_0 = const()[name = string("op_17043_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17043_cast_fp16 = rsqrt(epsilon = var_17043_epsilon_0, x = var_17042_cast_fp16)[name = string("op_17043_cast_fp16")]; + tensor var_17044_cast_fp16 = mul(x = x_517_cast_fp16, y = var_17043_cast_fp16)[name = string("op_17044_cast_fp16")]; + tensor input_729_cast_fp16 = mul(x = var_17044_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_729_cast_fp16")]; + string q_409_pad_type_0 = const()[name = string("q_409_pad_type_0"), val = string("valid")]; + tensor q_409_strides_0 = const()[name = string("q_409_strides_0"), val = tensor([1, 1])]; + tensor q_409_pad_0 = const()[name = string("q_409_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_409_dilations_0 = const()[name = string("q_409_dilations_0"), val = tensor([1, 1])]; + int32 q_409_groups_0 = const()[name = string("q_409_groups_0"), val = int32(1)]; + tensor q_409_cast_fp16 = conv(dilations = q_409_dilations_0, groups = q_409_groups_0, pad = q_409_pad_0, pad_type = q_409_pad_type_0, strides = q_409_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_729_cast_fp16)[name = string("q_409_cast_fp16")]; + string k_409_pad_type_0 = const()[name = string("k_409_pad_type_0"), val = string("valid")]; + tensor k_409_strides_0 = const()[name = string("k_409_strides_0"), val = tensor([1, 1])]; + tensor k_409_pad_0 = const()[name = string("k_409_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_409_dilations_0 = const()[name = string("k_409_dilations_0"), val = tensor([1, 1])]; + int32 k_409_groups_0 = const()[name = string("k_409_groups_0"), val = int32(1)]; + tensor k_409_cast_fp16 = conv(dilations = k_409_dilations_0, groups = k_409_groups_0, pad = k_409_pad_0, pad_type = k_409_pad_type_0, strides = k_409_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_729_cast_fp16)[name = string("k_409_cast_fp16")]; + string v_137_pad_type_0 = const()[name = string("v_137_pad_type_0"), val = string("valid")]; + tensor v_137_strides_0 = const()[name = string("v_137_strides_0"), val = tensor([1, 1])]; + tensor v_137_pad_0 = const()[name = string("v_137_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_137_dilations_0 = const()[name = string("v_137_dilations_0"), val = tensor([1, 1])]; + int32 v_137_groups_0 = const()[name = string("v_137_groups_0"), val = int32(1)]; + tensor v_137_cast_fp16 = conv(dilations = v_137_dilations_0, groups = v_137_groups_0, pad = v_137_pad_0, pad_type = v_137_pad_type_0, strides = v_137_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_729_cast_fp16)[name = string("v_137_cast_fp16")]; + tensor var_17078 = const()[name = string("op_17078"), val = tensor([16, 128, 1, 1])]; + tensor x_519_cast_fp16 = reshape(shape = var_17078, x = q_409_cast_fp16)[name = string("x_519_cast_fp16")]; + tensor var_17081_cast_fp16 = mul(x = x_519_cast_fp16, y = x_519_cast_fp16)[name = string("op_17081_cast_fp16")]; + tensor variance_571_axes_0 = const()[name = string("variance_571_axes_0"), val = tensor([1])]; + bool variance_571_keep_dims_0 = const()[name = string("variance_571_keep_dims_0"), val = bool(true)]; + tensor variance_571_cast_fp16 = reduce_mean(axes = variance_571_axes_0, keep_dims = variance_571_keep_dims_0, x = var_17081_cast_fp16)[name = string("variance_571_cast_fp16")]; + fp16 var_17084_to_fp16 = const()[name = string("op_17084_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17085_cast_fp16 = add(x = variance_571_cast_fp16, y = var_17084_to_fp16)[name = string("op_17085_cast_fp16")]; + fp32 var_17086_epsilon_0 = const()[name = string("op_17086_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17086_cast_fp16 = rsqrt(epsilon = var_17086_epsilon_0, x = var_17085_cast_fp16)[name = string("op_17086_cast_fp16")]; + tensor var_17087_cast_fp16 = mul(x = x_519_cast_fp16, y = var_17086_cast_fp16)[name = string("op_17087_cast_fp16")]; + tensor q_411_cast_fp16 = mul(x = var_17087_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_411_cast_fp16")]; + tensor var_17089 = const()[name = string("op_17089"), val = tensor([8, 128, 1, 1])]; + tensor x_521_cast_fp16 = reshape(shape = var_17089, x = k_409_cast_fp16)[name = string("x_521_cast_fp16")]; + tensor var_17092_cast_fp16 = mul(x = x_521_cast_fp16, y = x_521_cast_fp16)[name = string("op_17092_cast_fp16")]; + tensor variance_573_axes_0 = const()[name = string("variance_573_axes_0"), val = tensor([1])]; + bool variance_573_keep_dims_0 = const()[name = string("variance_573_keep_dims_0"), val = bool(true)]; + tensor variance_573_cast_fp16 = reduce_mean(axes = variance_573_axes_0, keep_dims = variance_573_keep_dims_0, x = var_17092_cast_fp16)[name = string("variance_573_cast_fp16")]; + fp16 var_17095_to_fp16 = const()[name = string("op_17095_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17096_cast_fp16 = add(x = variance_573_cast_fp16, y = var_17095_to_fp16)[name = string("op_17096_cast_fp16")]; + fp32 var_17097_epsilon_0 = const()[name = string("op_17097_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17097_cast_fp16 = rsqrt(epsilon = var_17097_epsilon_0, x = var_17096_cast_fp16)[name = string("op_17097_cast_fp16")]; + tensor var_17098_cast_fp16 = mul(x = x_521_cast_fp16, y = var_17097_cast_fp16)[name = string("op_17098_cast_fp16")]; + tensor k_411_cast_fp16 = mul(x = var_17098_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_411_cast_fp16")]; + tensor var_17100 = const()[name = string("op_17100"), val = tensor([1, 16, 128, 1])]; + tensor z_273_cast_fp16 = reshape(shape = var_17100, x = q_411_cast_fp16)[name = string("z_273_cast_fp16")]; + tensor var_17102 = const()[name = string("op_17102"), val = tensor([1, 8, 128, 1])]; + tensor z_275_cast_fp16 = reshape(shape = var_17102, x = k_411_cast_fp16)[name = string("z_275_cast_fp16")]; + tensor z1_273_begin_0 = const()[name = string("z1_273_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_273_end_0 = const()[name = string("z1_273_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_273_end_mask_0 = const()[name = string("z1_273_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_273_cast_fp16 = slice_by_index(begin = z1_273_begin_0, end = z1_273_end_0, end_mask = z1_273_end_mask_0, x = z_273_cast_fp16)[name = string("z1_273_cast_fp16")]; + tensor z2_273_begin_0 = const()[name = string("z2_273_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_273_end_0 = const()[name = string("z2_273_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_273_end_mask_0 = const()[name = string("z2_273_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_273_cast_fp16 = slice_by_index(begin = z2_273_begin_0, end = z2_273_end_0, end_mask = z2_273_end_mask_0, x = z_273_cast_fp16)[name = string("z2_273_cast_fp16")]; + tensor var_17110_cast_fp16 = mul(x = z_273_cast_fp16, y = cos_131_to_fp16)[name = string("op_17110_cast_fp16")]; + fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17111_cast_fp16 = mul(x = z2_273_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_17111_cast_fp16")]; + bool var_17113_interleave_0 = const()[name = string("op_17113_interleave_0"), val = bool(false)]; + tensor var_17113_cast_fp16 = concat(axis = var_17019, interleave = var_17113_interleave_0, values = (var_17111_cast_fp16, z1_273_cast_fp16))[name = string("op_17113_cast_fp16")]; + tensor var_17114_cast_fp16 = mul(x = var_17113_cast_fp16, y = sin_131_to_fp16)[name = string("op_17114_cast_fp16")]; + tensor q_413_cast_fp16 = add(x = var_17110_cast_fp16, y = var_17114_cast_fp16)[name = string("q_413_cast_fp16")]; + tensor z1_275_begin_0 = const()[name = string("z1_275_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_275_end_0 = const()[name = string("z1_275_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_275_end_mask_0 = const()[name = string("z1_275_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_275_cast_fp16 = slice_by_index(begin = z1_275_begin_0, end = z1_275_end_0, end_mask = z1_275_end_mask_0, x = z_275_cast_fp16)[name = string("z1_275_cast_fp16")]; + tensor z2_275_begin_0 = const()[name = string("z2_275_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_275_end_0 = const()[name = string("z2_275_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_275_end_mask_0 = const()[name = string("z2_275_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_275_cast_fp16 = slice_by_index(begin = z2_275_begin_0, end = z2_275_end_0, end_mask = z2_275_end_mask_0, x = z_275_cast_fp16)[name = string("z2_275_cast_fp16")]; + tensor var_17122_cast_fp16 = mul(x = z_275_cast_fp16, y = cos_131_to_fp16)[name = string("op_17122_cast_fp16")]; + fp16 const_151_promoted_to_fp16 = const()[name = string("const_151_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17123_cast_fp16 = mul(x = z2_275_cast_fp16, y = const_151_promoted_to_fp16)[name = string("op_17123_cast_fp16")]; + bool var_17125_interleave_0 = const()[name = string("op_17125_interleave_0"), val = bool(false)]; + tensor var_17125_cast_fp16 = concat(axis = var_17019, interleave = var_17125_interleave_0, values = (var_17123_cast_fp16, z1_275_cast_fp16))[name = string("op_17125_cast_fp16")]; + tensor var_17126_cast_fp16 = mul(x = var_17125_cast_fp16, y = sin_131_to_fp16)[name = string("op_17126_cast_fp16")]; + tensor k_413_cast_fp16 = add(x = var_17122_cast_fp16, y = var_17126_cast_fp16)[name = string("k_413_cast_fp16")]; + tensor var_17128 = const()[name = string("op_17128"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_137_cast_fp16 = reshape(shape = var_17128, x = k_413_cast_fp16)[name = string("cur_key_137_cast_fp16")]; + tensor var_17130_to_fp16 = const()[name = string("op_17130_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645312)))]; + tensor var_17131_cast_fp16 = mul(x = key_cache_137_cast_fp16, y = var_17130_to_fp16)[name = string("op_17131_cast_fp16")]; + tensor upd_137_to_fp16 = const()[name = string("upd_137_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645440)))]; + tensor var_17132_cast_fp16 = mul(x = cur_key_137_cast_fp16, y = upd_137_to_fp16)[name = string("op_17132_cast_fp16")]; + tensor key_137_cast_fp16 = add(x = var_17131_cast_fp16, y = var_17132_cast_fp16)[name = string("key_137_cast_fp16")]; + tensor var_17134_to_fp16 = const()[name = string("op_17134_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645312)))]; + tensor var_17135_cast_fp16 = mul(x = value_cache_137_cast_fp16, y = var_17134_to_fp16)[name = string("op_17135_cast_fp16")]; + tensor var_17136_cast_fp16 = mul(x = v_137_cast_fp16, y = upd_137_to_fp16)[name = string("op_17136_cast_fp16")]; + tensor value_137_cast_fp16 = add(x = var_17135_cast_fp16, y = var_17136_cast_fp16)[name = string("value_137_cast_fp16")]; + tensor var_17138 = const()[name = string("op_17138"), val = tensor([1, 8, 128, 16])]; + tensor kh_273_cast_fp16 = reshape(shape = var_17138, x = key_137_cast_fp16)[name = string("kh_273_cast_fp16")]; + tensor var_17140 = const()[name = string("op_17140"), val = tensor([1, 8, 128, 16])]; + tensor vh_273_cast_fp16 = reshape(shape = var_17140, x = value_137_cast_fp16)[name = string("vh_273_cast_fp16")]; + tensor transpose_272_perm_0 = const()[name = string("transpose_272_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_136_reps_0 = const()[name = string("tile_136_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_272_cast_fp16 = transpose(perm = transpose_272_perm_0, x = kh_273_cast_fp16)[name = string("transpose_71")]; + tensor tile_136_cast_fp16 = tile(reps = tile_136_reps_0, x = transpose_272_cast_fp16)[name = string("tile_136_cast_fp16")]; + tensor concat_340 = const()[name = string("concat_340"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_272_cast_fp16 = reshape(shape = concat_340, x = tile_136_cast_fp16)[name = string("reshape_272_cast_fp16")]; + tensor transpose_273_perm_0 = const()[name = string("transpose_273_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_341 = const()[name = string("concat_341"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_273_cast_fp16 = transpose(perm = transpose_273_perm_0, x = reshape_272_cast_fp16)[name = string("transpose_70")]; + tensor reshape_273_cast_fp16 = reshape(shape = concat_341, x = transpose_273_cast_fp16)[name = string("reshape_273_cast_fp16")]; + tensor transpose_274_perm_0 = const()[name = string("transpose_274_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_137_reps_0 = const()[name = string("tile_137_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_274_cast_fp16 = transpose(perm = transpose_274_perm_0, x = vh_273_cast_fp16)[name = string("transpose_69")]; + tensor tile_137_cast_fp16 = tile(reps = tile_137_reps_0, x = transpose_274_cast_fp16)[name = string("tile_137_cast_fp16")]; + tensor concat_342 = const()[name = string("concat_342"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_274_cast_fp16 = reshape(shape = concat_342, x = tile_137_cast_fp16)[name = string("reshape_274_cast_fp16")]; + tensor transpose_275_perm_0 = const()[name = string("transpose_275_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_343 = const()[name = string("concat_343"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_275_cast_fp16 = transpose(perm = transpose_275_perm_0, x = reshape_274_cast_fp16)[name = string("transpose_68")]; + tensor reshape_275_cast_fp16 = reshape(shape = concat_343, x = transpose_275_cast_fp16)[name = string("reshape_275_cast_fp16")]; + fp16 var_17144_to_fp16 = const()[name = string("op_17144_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_17145_cast_fp16 = mul(x = q_413_cast_fp16, y = var_17144_to_fp16)[name = string("op_17145_cast_fp16")]; + tensor transpose_589_perm_0 = const()[name = string("transpose_589_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_297_transpose_x_1 = const()[name = string("w_297_transpose_x_1"), val = bool(true)]; + bool w_297_transpose_y_1 = const()[name = string("w_297_transpose_y_1"), val = bool(false)]; + tensor transpose_589_cast_fp16 = transpose(perm = transpose_589_perm_0, x = reshape_273_cast_fp16)[name = string("transpose_67")]; + tensor w_297_cast_fp16 = matmul(transpose_x = w_297_transpose_x_1, transpose_y = w_297_transpose_y_1, x = var_17145_cast_fp16, y = transpose_589_cast_fp16)[name = string("w_297_cast_fp16")]; + tensor pad_137_to_fp16 = const()[name = string("pad_137_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645568)))]; + tensor var_17148_cast_fp16 = add(x = w_297_cast_fp16, y = pad_137_to_fp16)[name = string("op_17148_cast_fp16")]; + tensor w_299_cast_fp16 = softmax(axis = var_17023, x = var_17148_cast_fp16)[name = string("w_299_cast_fp16")]; + tensor transpose_590_perm_0 = const()[name = string("transpose_590_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_137_transpose_x_1 = const()[name = string("attn_137_transpose_x_1"), val = bool(false)]; + bool attn_137_transpose_y_1 = const()[name = string("attn_137_transpose_y_1"), val = bool(true)]; + tensor transpose_590_cast_fp16 = transpose(perm = transpose_590_perm_0, x = reshape_275_cast_fp16)[name = string("transpose_66")]; + tensor attn_137_cast_fp16 = matmul(transpose_x = attn_137_transpose_x_1, transpose_y = attn_137_transpose_y_1, x = transpose_590_cast_fp16, y = w_299_cast_fp16)[name = string("attn_137_cast_fp16")]; + tensor var_17152 = const()[name = string("op_17152"), val = tensor([1, 2048, 1, 1])]; + tensor input_731_cast_fp16 = reshape(shape = var_17152, x = attn_137_cast_fp16)[name = string("input_731_cast_fp16")]; + string attn_output_137_pad_type_0 = const()[name = string("attn_output_137_pad_type_0"), val = string("valid")]; + tensor attn_output_137_strides_0 = const()[name = string("attn_output_137_strides_0"), val = tensor([1, 1])]; + tensor attn_output_137_pad_0 = const()[name = string("attn_output_137_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_137_dilations_0 = const()[name = string("attn_output_137_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_137_groups_0 = const()[name = string("attn_output_137_groups_0"), val = int32(1)]; + tensor attn_output_137_cast_fp16 = conv(dilations = attn_output_137_dilations_0, groups = attn_output_137_groups_0, pad = attn_output_137_pad_0, pad_type = attn_output_137_pad_type_0, strides = attn_output_137_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_731_cast_fp16)[name = string("attn_output_137_cast_fp16")]; + tensor x_523_cast_fp16 = add(x = x_517_cast_fp16, y = attn_output_137_cast_fp16)[name = string("x_523_cast_fp16")]; + tensor var_17166_cast_fp16 = mul(x = x_523_cast_fp16, y = x_523_cast_fp16)[name = string("op_17166_cast_fp16")]; + tensor variance_575_axes_0 = const()[name = string("variance_575_axes_0"), val = tensor([1])]; + bool variance_575_keep_dims_0 = const()[name = string("variance_575_keep_dims_0"), val = bool(true)]; + tensor variance_575_cast_fp16 = reduce_mean(axes = variance_575_axes_0, keep_dims = variance_575_keep_dims_0, x = var_17166_cast_fp16)[name = string("variance_575_cast_fp16")]; + fp16 var_17169_to_fp16 = const()[name = string("op_17169_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17170_cast_fp16 = add(x = variance_575_cast_fp16, y = var_17169_to_fp16)[name = string("op_17170_cast_fp16")]; + fp32 var_17171_epsilon_0 = const()[name = string("op_17171_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17171_cast_fp16 = rsqrt(epsilon = var_17171_epsilon_0, x = var_17170_cast_fp16)[name = string("op_17171_cast_fp16")]; + tensor var_17172_cast_fp16 = mul(x = x_523_cast_fp16, y = var_17171_cast_fp16)[name = string("op_17172_cast_fp16")]; + tensor input_733_cast_fp16 = mul(x = var_17172_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_733_cast_fp16")]; + string input_735_pad_type_0 = const()[name = string("input_735_pad_type_0"), val = string("valid")]; + tensor input_735_strides_0 = const()[name = string("input_735_strides_0"), val = tensor([1, 1])]; + tensor input_735_pad_0 = const()[name = string("input_735_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_735_dilations_0 = const()[name = string("input_735_dilations_0"), val = tensor([1, 1])]; + int32 input_735_groups_0 = const()[name = string("input_735_groups_0"), val = int32(1)]; + tensor input_735_cast_fp16 = conv(dilations = input_735_dilations_0, groups = input_735_groups_0, pad = input_735_pad_0, pad_type = input_735_pad_type_0, strides = input_735_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_733_cast_fp16)[name = string("input_735_cast_fp16")]; + tensor var_17180_cast_fp16 = silu(x = input_735_cast_fp16)[name = string("op_17180_cast_fp16")]; + string var_17186_pad_type_0 = const()[name = string("op_17186_pad_type_0"), val = string("valid")]; + tensor var_17186_strides_0 = const()[name = string("op_17186_strides_0"), val = tensor([1, 1])]; + tensor var_17186_pad_0 = const()[name = string("op_17186_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17186_dilations_0 = const()[name = string("op_17186_dilations_0"), val = tensor([1, 1])]; + int32 var_17186_groups_0 = const()[name = string("op_17186_groups_0"), val = int32(1)]; + tensor var_17186_cast_fp16 = conv(dilations = var_17186_dilations_0, groups = var_17186_groups_0, pad = var_17186_pad_0, pad_type = var_17186_pad_type_0, strides = var_17186_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_733_cast_fp16)[name = string("op_17186_cast_fp16")]; + tensor input_737_cast_fp16 = mul(x = var_17180_cast_fp16, y = var_17186_cast_fp16)[name = string("input_737_cast_fp16")]; + string h_137_pad_type_0 = const()[name = string("h_137_pad_type_0"), val = string("valid")]; + tensor h_137_strides_0 = const()[name = string("h_137_strides_0"), val = tensor([1, 1])]; + tensor h_137_pad_0 = const()[name = string("h_137_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_137_dilations_0 = const()[name = string("h_137_dilations_0"), val = tensor([1, 1])]; + int32 h_137_groups_0 = const()[name = string("h_137_groups_0"), val = int32(1)]; + tensor h_137_cast_fp16 = conv(dilations = h_137_dilations_0, groups = h_137_groups_0, pad = h_137_pad_0, pad_type = h_137_pad_type_0, strides = h_137_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_737_cast_fp16)[name = string("h_137_cast_fp16")]; + tensor x_525_cast_fp16 = add(x = x_523_cast_fp16, y = h_137_cast_fp16)[name = string("x_525_cast_fp16")]; + tensor key_cache_139_begin_0 = const()[name = string("key_cache_139_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_139_end_0 = const()[name = string("key_cache_139_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_139_end_mask_0 = const()[name = string("key_cache_139_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_139_cast_fp16 = slice_by_index(begin = key_cache_139_begin_0, end = key_cache_139_end_0, end_mask = key_cache_139_end_mask_0, x = layer_key_caches_27_cast_fp16)[name = string("key_cache_139_cast_fp16")]; + tensor value_cache_139_begin_0 = const()[name = string("value_cache_139_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_139_end_0 = const()[name = string("value_cache_139_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_139_end_mask_0 = const()[name = string("value_cache_139_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_139_cast_fp16 = slice_by_index(begin = value_cache_139_begin_0, end = value_cache_139_end_0, end_mask = value_cache_139_end_mask_0, x = layer_value_caches_27_cast_fp16)[name = string("value_cache_139_cast_fp16")]; + int32 var_17239 = const()[name = string("op_17239"), val = int32(2)]; + int32 var_17243 = const()[name = string("op_17243"), val = int32(3)]; + tensor var_17258_cast_fp16 = mul(x = x_525_cast_fp16, y = x_525_cast_fp16)[name = string("op_17258_cast_fp16")]; + tensor variance_577_axes_0 = const()[name = string("variance_577_axes_0"), val = tensor([1])]; + bool variance_577_keep_dims_0 = const()[name = string("variance_577_keep_dims_0"), val = bool(true)]; + tensor variance_577_cast_fp16 = reduce_mean(axes = variance_577_axes_0, keep_dims = variance_577_keep_dims_0, x = var_17258_cast_fp16)[name = string("variance_577_cast_fp16")]; + fp16 var_17261_to_fp16 = const()[name = string("op_17261_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17262_cast_fp16 = add(x = variance_577_cast_fp16, y = var_17261_to_fp16)[name = string("op_17262_cast_fp16")]; + fp32 var_17263_epsilon_0 = const()[name = string("op_17263_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17263_cast_fp16 = rsqrt(epsilon = var_17263_epsilon_0, x = var_17262_cast_fp16)[name = string("op_17263_cast_fp16")]; + tensor var_17264_cast_fp16 = mul(x = x_525_cast_fp16, y = var_17263_cast_fp16)[name = string("op_17264_cast_fp16")]; + tensor input_739_cast_fp16 = mul(x = var_17264_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_739_cast_fp16")]; + string q_415_pad_type_0 = const()[name = string("q_415_pad_type_0"), val = string("valid")]; + tensor q_415_strides_0 = const()[name = string("q_415_strides_0"), val = tensor([1, 1])]; + tensor q_415_pad_0 = const()[name = string("q_415_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_415_dilations_0 = const()[name = string("q_415_dilations_0"), val = tensor([1, 1])]; + int32 q_415_groups_0 = const()[name = string("q_415_groups_0"), val = int32(1)]; + tensor q_415_cast_fp16 = conv(dilations = q_415_dilations_0, groups = q_415_groups_0, pad = q_415_pad_0, pad_type = q_415_pad_type_0, strides = q_415_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_739_cast_fp16)[name = string("q_415_cast_fp16")]; + string k_415_pad_type_0 = const()[name = string("k_415_pad_type_0"), val = string("valid")]; + tensor k_415_strides_0 = const()[name = string("k_415_strides_0"), val = tensor([1, 1])]; + tensor k_415_pad_0 = const()[name = string("k_415_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_415_dilations_0 = const()[name = string("k_415_dilations_0"), val = tensor([1, 1])]; + int32 k_415_groups_0 = const()[name = string("k_415_groups_0"), val = int32(1)]; + tensor k_415_cast_fp16 = conv(dilations = k_415_dilations_0, groups = k_415_groups_0, pad = k_415_pad_0, pad_type = k_415_pad_type_0, strides = k_415_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_739_cast_fp16)[name = string("k_415_cast_fp16")]; + string v_139_pad_type_0 = const()[name = string("v_139_pad_type_0"), val = string("valid")]; + tensor v_139_strides_0 = const()[name = string("v_139_strides_0"), val = tensor([1, 1])]; + tensor v_139_pad_0 = const()[name = string("v_139_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_139_dilations_0 = const()[name = string("v_139_dilations_0"), val = tensor([1, 1])]; + int32 v_139_groups_0 = const()[name = string("v_139_groups_0"), val = int32(1)]; + tensor v_139_cast_fp16 = conv(dilations = v_139_dilations_0, groups = v_139_groups_0, pad = v_139_pad_0, pad_type = v_139_pad_type_0, strides = v_139_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_739_cast_fp16)[name = string("v_139_cast_fp16")]; + tensor var_17298 = const()[name = string("op_17298"), val = tensor([16, 128, 1, 1])]; + tensor x_527_cast_fp16 = reshape(shape = var_17298, x = q_415_cast_fp16)[name = string("x_527_cast_fp16")]; + tensor var_17301_cast_fp16 = mul(x = x_527_cast_fp16, y = x_527_cast_fp16)[name = string("op_17301_cast_fp16")]; + tensor variance_579_axes_0 = const()[name = string("variance_579_axes_0"), val = tensor([1])]; + bool variance_579_keep_dims_0 = const()[name = string("variance_579_keep_dims_0"), val = bool(true)]; + tensor variance_579_cast_fp16 = reduce_mean(axes = variance_579_axes_0, keep_dims = variance_579_keep_dims_0, x = var_17301_cast_fp16)[name = string("variance_579_cast_fp16")]; + fp16 var_17304_to_fp16 = const()[name = string("op_17304_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17305_cast_fp16 = add(x = variance_579_cast_fp16, y = var_17304_to_fp16)[name = string("op_17305_cast_fp16")]; + fp32 var_17306_epsilon_0 = const()[name = string("op_17306_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17306_cast_fp16 = rsqrt(epsilon = var_17306_epsilon_0, x = var_17305_cast_fp16)[name = string("op_17306_cast_fp16")]; + tensor var_17307_cast_fp16 = mul(x = x_527_cast_fp16, y = var_17306_cast_fp16)[name = string("op_17307_cast_fp16")]; + tensor q_417_cast_fp16 = mul(x = var_17307_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_417_cast_fp16")]; + tensor var_17309 = const()[name = string("op_17309"), val = tensor([8, 128, 1, 1])]; + tensor x_529_cast_fp16 = reshape(shape = var_17309, x = k_415_cast_fp16)[name = string("x_529_cast_fp16")]; + tensor var_17312_cast_fp16 = mul(x = x_529_cast_fp16, y = x_529_cast_fp16)[name = string("op_17312_cast_fp16")]; + tensor variance_581_axes_0 = const()[name = string("variance_581_axes_0"), val = tensor([1])]; + bool variance_581_keep_dims_0 = const()[name = string("variance_581_keep_dims_0"), val = bool(true)]; + tensor variance_581_cast_fp16 = reduce_mean(axes = variance_581_axes_0, keep_dims = variance_581_keep_dims_0, x = var_17312_cast_fp16)[name = string("variance_581_cast_fp16")]; + fp16 var_17315_to_fp16 = const()[name = string("op_17315_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17316_cast_fp16 = add(x = variance_581_cast_fp16, y = var_17315_to_fp16)[name = string("op_17316_cast_fp16")]; + fp32 var_17317_epsilon_0 = const()[name = string("op_17317_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17317_cast_fp16 = rsqrt(epsilon = var_17317_epsilon_0, x = var_17316_cast_fp16)[name = string("op_17317_cast_fp16")]; + tensor var_17318_cast_fp16 = mul(x = x_529_cast_fp16, y = var_17317_cast_fp16)[name = string("op_17318_cast_fp16")]; + tensor k_417_cast_fp16 = mul(x = var_17318_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_417_cast_fp16")]; + tensor var_17320 = const()[name = string("op_17320"), val = tensor([1, 16, 128, 1])]; + tensor z_277_cast_fp16 = reshape(shape = var_17320, x = q_417_cast_fp16)[name = string("z_277_cast_fp16")]; + tensor var_17322 = const()[name = string("op_17322"), val = tensor([1, 8, 128, 1])]; + tensor z_279_cast_fp16 = reshape(shape = var_17322, x = k_417_cast_fp16)[name = string("z_279_cast_fp16")]; + tensor z1_277_begin_0 = const()[name = string("z1_277_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_277_end_0 = const()[name = string("z1_277_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_277_end_mask_0 = const()[name = string("z1_277_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_277_cast_fp16 = slice_by_index(begin = z1_277_begin_0, end = z1_277_end_0, end_mask = z1_277_end_mask_0, x = z_277_cast_fp16)[name = string("z1_277_cast_fp16")]; + tensor z2_277_begin_0 = const()[name = string("z2_277_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_277_end_0 = const()[name = string("z2_277_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_277_end_mask_0 = const()[name = string("z2_277_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_277_cast_fp16 = slice_by_index(begin = z2_277_begin_0, end = z2_277_end_0, end_mask = z2_277_end_mask_0, x = z_277_cast_fp16)[name = string("z2_277_cast_fp16")]; + tensor var_17330_cast_fp16 = mul(x = z_277_cast_fp16, y = cos_131_to_fp16)[name = string("op_17330_cast_fp16")]; + fp16 const_152_promoted_to_fp16 = const()[name = string("const_152_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17331_cast_fp16 = mul(x = z2_277_cast_fp16, y = const_152_promoted_to_fp16)[name = string("op_17331_cast_fp16")]; + bool var_17333_interleave_0 = const()[name = string("op_17333_interleave_0"), val = bool(false)]; + tensor var_17333_cast_fp16 = concat(axis = var_17239, interleave = var_17333_interleave_0, values = (var_17331_cast_fp16, z1_277_cast_fp16))[name = string("op_17333_cast_fp16")]; + tensor var_17334_cast_fp16 = mul(x = var_17333_cast_fp16, y = sin_131_to_fp16)[name = string("op_17334_cast_fp16")]; + tensor q_419_cast_fp16 = add(x = var_17330_cast_fp16, y = var_17334_cast_fp16)[name = string("q_419_cast_fp16")]; + tensor z1_279_begin_0 = const()[name = string("z1_279_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_279_end_0 = const()[name = string("z1_279_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_279_end_mask_0 = const()[name = string("z1_279_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_279_cast_fp16 = slice_by_index(begin = z1_279_begin_0, end = z1_279_end_0, end_mask = z1_279_end_mask_0, x = z_279_cast_fp16)[name = string("z1_279_cast_fp16")]; + tensor z2_279_begin_0 = const()[name = string("z2_279_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_279_end_0 = const()[name = string("z2_279_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_279_end_mask_0 = const()[name = string("z2_279_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_279_cast_fp16 = slice_by_index(begin = z2_279_begin_0, end = z2_279_end_0, end_mask = z2_279_end_mask_0, x = z_279_cast_fp16)[name = string("z2_279_cast_fp16")]; + tensor var_17342_cast_fp16 = mul(x = z_279_cast_fp16, y = cos_131_to_fp16)[name = string("op_17342_cast_fp16")]; + fp16 const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17343_cast_fp16 = mul(x = z2_279_cast_fp16, y = const_153_promoted_to_fp16)[name = string("op_17343_cast_fp16")]; + bool var_17345_interleave_0 = const()[name = string("op_17345_interleave_0"), val = bool(false)]; + tensor var_17345_cast_fp16 = concat(axis = var_17239, interleave = var_17345_interleave_0, values = (var_17343_cast_fp16, z1_279_cast_fp16))[name = string("op_17345_cast_fp16")]; + tensor var_17346_cast_fp16 = mul(x = var_17345_cast_fp16, y = sin_131_to_fp16)[name = string("op_17346_cast_fp16")]; + tensor k_419_cast_fp16 = add(x = var_17342_cast_fp16, y = var_17346_cast_fp16)[name = string("k_419_cast_fp16")]; + tensor var_17348 = const()[name = string("op_17348"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_139_cast_fp16 = reshape(shape = var_17348, x = k_419_cast_fp16)[name = string("cur_key_139_cast_fp16")]; + tensor var_17350_to_fp16 = const()[name = string("op_17350_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645312)))]; + tensor var_17351_cast_fp16 = mul(x = key_cache_139_cast_fp16, y = var_17350_to_fp16)[name = string("op_17351_cast_fp16")]; + tensor upd_139_to_fp16 = const()[name = string("upd_139_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645440)))]; + tensor var_17352_cast_fp16 = mul(x = cur_key_139_cast_fp16, y = upd_139_to_fp16)[name = string("op_17352_cast_fp16")]; + tensor key_139_cast_fp16 = add(x = var_17351_cast_fp16, y = var_17352_cast_fp16)[name = string("key_139_cast_fp16")]; + tensor var_17354_to_fp16 = const()[name = string("op_17354_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645312)))]; + tensor var_17355_cast_fp16 = mul(x = value_cache_139_cast_fp16, y = var_17354_to_fp16)[name = string("op_17355_cast_fp16")]; + tensor var_17356_cast_fp16 = mul(x = v_139_cast_fp16, y = upd_139_to_fp16)[name = string("op_17356_cast_fp16")]; + tensor value_139_cast_fp16 = add(x = var_17355_cast_fp16, y = var_17356_cast_fp16)[name = string("value_139_cast_fp16")]; + tensor var_17358 = const()[name = string("op_17358"), val = tensor([1, 8, 128, 16])]; + tensor kh_277_cast_fp16 = reshape(shape = var_17358, x = key_139_cast_fp16)[name = string("kh_277_cast_fp16")]; + tensor var_17360 = const()[name = string("op_17360"), val = tensor([1, 8, 128, 16])]; + tensor vh_277_cast_fp16 = reshape(shape = var_17360, x = value_139_cast_fp16)[name = string("vh_277_cast_fp16")]; + tensor transpose_276_perm_0 = const()[name = string("transpose_276_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_138_reps_0 = const()[name = string("tile_138_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_276_cast_fp16 = transpose(perm = transpose_276_perm_0, x = kh_277_cast_fp16)[name = string("transpose_65")]; + tensor tile_138_cast_fp16 = tile(reps = tile_138_reps_0, x = transpose_276_cast_fp16)[name = string("tile_138_cast_fp16")]; + tensor concat_344 = const()[name = string("concat_344"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_276_cast_fp16 = reshape(shape = concat_344, x = tile_138_cast_fp16)[name = string("reshape_276_cast_fp16")]; + tensor transpose_277_perm_0 = const()[name = string("transpose_277_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_345 = const()[name = string("concat_345"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_277_cast_fp16 = transpose(perm = transpose_277_perm_0, x = reshape_276_cast_fp16)[name = string("transpose_64")]; + tensor reshape_277_cast_fp16 = reshape(shape = concat_345, x = transpose_277_cast_fp16)[name = string("reshape_277_cast_fp16")]; + tensor transpose_278_perm_0 = const()[name = string("transpose_278_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_139_reps_0 = const()[name = string("tile_139_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_278_cast_fp16 = transpose(perm = transpose_278_perm_0, x = vh_277_cast_fp16)[name = string("transpose_63")]; + tensor tile_139_cast_fp16 = tile(reps = tile_139_reps_0, x = transpose_278_cast_fp16)[name = string("tile_139_cast_fp16")]; + tensor concat_346 = const()[name = string("concat_346"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_278_cast_fp16 = reshape(shape = concat_346, x = tile_139_cast_fp16)[name = string("reshape_278_cast_fp16")]; + tensor transpose_279_perm_0 = const()[name = string("transpose_279_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_347 = const()[name = string("concat_347"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_279_cast_fp16 = transpose(perm = transpose_279_perm_0, x = reshape_278_cast_fp16)[name = string("transpose_62")]; + tensor reshape_279_cast_fp16 = reshape(shape = concat_347, x = transpose_279_cast_fp16)[name = string("reshape_279_cast_fp16")]; + fp16 var_17364_to_fp16 = const()[name = string("op_17364_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_17365_cast_fp16 = mul(x = q_419_cast_fp16, y = var_17364_to_fp16)[name = string("op_17365_cast_fp16")]; + tensor transpose_593_perm_0 = const()[name = string("transpose_593_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_301_transpose_x_1 = const()[name = string("w_301_transpose_x_1"), val = bool(true)]; + bool w_301_transpose_y_1 = const()[name = string("w_301_transpose_y_1"), val = bool(false)]; + tensor transpose_593_cast_fp16 = transpose(perm = transpose_593_perm_0, x = reshape_277_cast_fp16)[name = string("transpose_61")]; + tensor w_301_cast_fp16 = matmul(transpose_x = w_301_transpose_x_1, transpose_y = w_301_transpose_y_1, x = var_17365_cast_fp16, y = transpose_593_cast_fp16)[name = string("w_301_cast_fp16")]; + tensor pad_139_to_fp16 = const()[name = string("pad_139_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645568)))]; + tensor var_17368_cast_fp16 = add(x = w_301_cast_fp16, y = pad_139_to_fp16)[name = string("op_17368_cast_fp16")]; + tensor w_303_cast_fp16 = softmax(axis = var_17243, x = var_17368_cast_fp16)[name = string("w_303_cast_fp16")]; + tensor transpose_594_perm_0 = const()[name = string("transpose_594_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_139_transpose_x_1 = const()[name = string("attn_139_transpose_x_1"), val = bool(false)]; + bool attn_139_transpose_y_1 = const()[name = string("attn_139_transpose_y_1"), val = bool(true)]; + tensor transpose_594_cast_fp16 = transpose(perm = transpose_594_perm_0, x = reshape_279_cast_fp16)[name = string("transpose_60")]; + tensor attn_139_cast_fp16 = matmul(transpose_x = attn_139_transpose_x_1, transpose_y = attn_139_transpose_y_1, x = transpose_594_cast_fp16, y = w_303_cast_fp16)[name = string("attn_139_cast_fp16")]; + tensor var_17372 = const()[name = string("op_17372"), val = tensor([1, 2048, 1, 1])]; + tensor input_741_cast_fp16 = reshape(shape = var_17372, x = attn_139_cast_fp16)[name = string("input_741_cast_fp16")]; + string attn_output_139_pad_type_0 = const()[name = string("attn_output_139_pad_type_0"), val = string("valid")]; + tensor attn_output_139_strides_0 = const()[name = string("attn_output_139_strides_0"), val = tensor([1, 1])]; + tensor attn_output_139_pad_0 = const()[name = string("attn_output_139_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_139_dilations_0 = const()[name = string("attn_output_139_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_139_groups_0 = const()[name = string("attn_output_139_groups_0"), val = int32(1)]; + tensor attn_output_139_cast_fp16 = conv(dilations = attn_output_139_dilations_0, groups = attn_output_139_groups_0, pad = attn_output_139_pad_0, pad_type = attn_output_139_pad_type_0, strides = attn_output_139_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_741_cast_fp16)[name = string("attn_output_139_cast_fp16")]; + tensor x_531_cast_fp16 = add(x = x_525_cast_fp16, y = attn_output_139_cast_fp16)[name = string("x_531_cast_fp16")]; + tensor var_17386_cast_fp16 = mul(x = x_531_cast_fp16, y = x_531_cast_fp16)[name = string("op_17386_cast_fp16")]; + tensor variance_583_axes_0 = const()[name = string("variance_583_axes_0"), val = tensor([1])]; + bool variance_583_keep_dims_0 = const()[name = string("variance_583_keep_dims_0"), val = bool(true)]; + tensor variance_583_cast_fp16 = reduce_mean(axes = variance_583_axes_0, keep_dims = variance_583_keep_dims_0, x = var_17386_cast_fp16)[name = string("variance_583_cast_fp16")]; + fp16 var_17389_to_fp16 = const()[name = string("op_17389_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17390_cast_fp16 = add(x = variance_583_cast_fp16, y = var_17389_to_fp16)[name = string("op_17390_cast_fp16")]; + fp32 var_17391_epsilon_0 = const()[name = string("op_17391_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17391_cast_fp16 = rsqrt(epsilon = var_17391_epsilon_0, x = var_17390_cast_fp16)[name = string("op_17391_cast_fp16")]; + tensor var_17392_cast_fp16 = mul(x = x_531_cast_fp16, y = var_17391_cast_fp16)[name = string("op_17392_cast_fp16")]; + tensor input_743_cast_fp16 = mul(x = var_17392_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_743_cast_fp16")]; + string input_745_pad_type_0 = const()[name = string("input_745_pad_type_0"), val = string("valid")]; + tensor input_745_strides_0 = const()[name = string("input_745_strides_0"), val = tensor([1, 1])]; + tensor input_745_pad_0 = const()[name = string("input_745_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_745_dilations_0 = const()[name = string("input_745_dilations_0"), val = tensor([1, 1])]; + int32 input_745_groups_0 = const()[name = string("input_745_groups_0"), val = int32(1)]; + tensor input_745_cast_fp16 = conv(dilations = input_745_dilations_0, groups = input_745_groups_0, pad = input_745_pad_0, pad_type = input_745_pad_type_0, strides = input_745_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_743_cast_fp16)[name = string("input_745_cast_fp16")]; + tensor var_17400_cast_fp16 = silu(x = input_745_cast_fp16)[name = string("op_17400_cast_fp16")]; + string var_17406_pad_type_0 = const()[name = string("op_17406_pad_type_0"), val = string("valid")]; + tensor var_17406_strides_0 = const()[name = string("op_17406_strides_0"), val = tensor([1, 1])]; + tensor var_17406_pad_0 = const()[name = string("op_17406_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17406_dilations_0 = const()[name = string("op_17406_dilations_0"), val = tensor([1, 1])]; + int32 var_17406_groups_0 = const()[name = string("op_17406_groups_0"), val = int32(1)]; + tensor var_17406_cast_fp16 = conv(dilations = var_17406_dilations_0, groups = var_17406_groups_0, pad = var_17406_pad_0, pad_type = var_17406_pad_type_0, strides = var_17406_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_743_cast_fp16)[name = string("op_17406_cast_fp16")]; + tensor input_747_cast_fp16 = mul(x = var_17400_cast_fp16, y = var_17406_cast_fp16)[name = string("input_747_cast_fp16")]; + string h_139_pad_type_0 = const()[name = string("h_139_pad_type_0"), val = string("valid")]; + tensor h_139_strides_0 = const()[name = string("h_139_strides_0"), val = tensor([1, 1])]; + tensor h_139_pad_0 = const()[name = string("h_139_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_139_dilations_0 = const()[name = string("h_139_dilations_0"), val = tensor([1, 1])]; + int32 h_139_groups_0 = const()[name = string("h_139_groups_0"), val = int32(1)]; + tensor h_139_cast_fp16 = conv(dilations = h_139_dilations_0, groups = h_139_groups_0, pad = h_139_pad_0, pad_type = h_139_pad_type_0, strides = h_139_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_747_cast_fp16)[name = string("h_139_cast_fp16")]; + tensor inputs_25_cast_fp16 = add(x = x_531_cast_fp16, y = h_139_cast_fp16)[name = string("inputs_25_cast_fp16")]; + int32 var_17434 = const()[name = string("op_17434"), val = int32(1)]; + bool layer_key_caches_29_interleave_0 = const()[name = string("layer_key_caches_29_interleave_0"), val = bool(false)]; + tensor layer_key_caches_29_cast_fp16 = concat(axis = var_17434, interleave = layer_key_caches_29_interleave_0, values = (key_131_cast_fp16, key_133_cast_fp16, key_135_cast_fp16, key_137_cast_fp16, key_139_cast_fp16))[name = string("layer_key_caches_29_cast_fp16")]; + int32 var_17437 = const()[name = string("op_17437"), val = int32(1)]; + bool layer_value_caches_29_interleave_0 = const()[name = string("layer_value_caches_29_interleave_0"), val = bool(false)]; + tensor layer_value_caches_29_cast_fp16 = concat(axis = var_17437, interleave = layer_value_caches_29_interleave_0, values = (value_131_cast_fp16, value_133_cast_fp16, value_135_cast_fp16, value_137_cast_fp16, value_139_cast_fp16))[name = string("layer_value_caches_29_cast_fp16")]; + tensor inputs_sq_25_cast_fp16 = mul(x = inputs_25_cast_fp16, y = inputs_25_cast_fp16)[name = string("inputs_sq_25_cast_fp16")]; + tensor variance_585_axes_0 = const()[name = string("variance_585_axes_0"), val = tensor([1])]; + bool variance_585_keep_dims_0 = const()[name = string("variance_585_keep_dims_0"), val = bool(true)]; + tensor variance_585_cast_fp16 = reduce_mean(axes = variance_585_axes_0, keep_dims = variance_585_keep_dims_0, x = inputs_sq_25_cast_fp16)[name = string("variance_585_cast_fp16")]; + fp16 var_17447_to_fp16 = const()[name = string("op_17447_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17448_cast_fp16 = add(x = variance_585_cast_fp16, y = var_17447_to_fp16)[name = string("op_17448_cast_fp16")]; + fp32 var_17449_epsilon_0 = const()[name = string("op_17449_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17449_cast_fp16 = rsqrt(epsilon = var_17449_epsilon_0, x = var_17448_cast_fp16)[name = string("op_17449_cast_fp16")]; + tensor hidden_states_25_cast_fp16 = mul(x = inputs_25_cast_fp16, y = var_17449_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; + tensor input_749_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_25_cast_fp16)[name = string("input_749_cast_fp16")]; + string logits_49_pad_type_0 = const()[name = string("logits_49_pad_type_0"), val = string("valid")]; + tensor logits_49_strides_0 = const()[name = string("logits_49_strides_0"), val = tensor([1, 1])]; + tensor logits_49_pad_0 = const()[name = string("logits_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_49_dilations_0 = const()[name = string("logits_49_dilations_0"), val = tensor([1, 1])]; + int32 logits_49_groups_0 = const()[name = string("logits_49_groups_0"), val = int32(1)]; + tensor lm_heads_12_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103880192))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105977408))))[name = string("lm_heads_12_weight_to_fp16_palettized")]; + tensor logits_49_cast_fp16 = conv(dilations = logits_49_dilations_0, groups = logits_49_groups_0, pad = logits_49_pad_0, pad_type = logits_49_pad_type_0, strides = logits_49_strides_0, weight = lm_heads_12_weight_to_fp16_palettized, x = input_749_cast_fp16)[name = string("logits_49_cast_fp16")]; + tensor var_17467 = const()[name = string("op_17467"), val = tensor([1, 2048])]; + tensor logits_51_cast_fp16 = reshape(shape = var_17467, x = logits_49_cast_fp16)[name = string("logits_51_cast_fp16")]; + tensor scaled_logits_25_cast_fp16 = real_div(x = logits_51_cast_fp16, y = temperature)[name = string("scaled_logits_25_cast_fp16")]; + int32 var_17477 = const()[name = string("op_17477"), val = int32(100)]; + int32 top_values_25_axis_0 = const()[name = string("top_values_25_axis_0"), val = int32(1)]; + bool top_values_25_ascending_0 = const()[name = string("top_values_25_ascending_0"), val = bool(false)]; + bool top_values_25_sort_0 = const()[name = string("top_values_25_sort_0"), val = bool(true)]; + bool top_values_25_return_indices_0 = const()[name = string("top_values_25_return_indices_0"), val = bool(true)]; + string top_values_25_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_25_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_25_cast_fp16_cast_uint16_0, tensor top_values_25_cast_fp16_cast_uint16_1 = topk(ascending = top_values_25_ascending_0, axis = top_values_25_axis_0, k = var_17477, output_indices_dtype = top_values_25_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_25_return_indices_0, sort = top_values_25_sort_0, x = scaled_logits_25_cast_fp16)[name = string("top_values_25_cast_fp16_cast_uint16")]; + tensor var_17483_cast_fp16 = mul(x = top_values_25_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_17483_cast_fp16")]; + tensor var_17487_cast_fp16 = add(x = var_17483_cast_fp16, y = var_2438_cast_fp16)[name = string("op_17487_cast_fp16")]; + tensor reduce_min_12_axes_0 = const()[name = string("reduce_min_12_axes_0"), val = tensor([1])]; + bool reduce_min_12_keep_dims_0 = const()[name = string("reduce_min_12_keep_dims_0"), val = bool(true)]; + tensor reduce_min_12_cast_fp16 = reduce_min(axes = reduce_min_12_axes_0, keep_dims = reduce_min_12_keep_dims_0, x = var_17487_cast_fp16)[name = string("reduce_min_12_cast_fp16")]; + tensor var_17490_cast_fp16 = greater_equal(x = scaled_logits_25_cast_fp16, y = reduce_min_12_cast_fp16)[name = string("op_17490_cast_fp16")]; + fp16 var_17491_value_0_to_fp16 = const()[name = string("op_17491_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_17491_cast_fp16 = fill_like(ref_tensor = scaled_logits_25_cast_fp16, value = var_17491_value_0_to_fp16)[name = string("op_17491_cast_fp16")]; + tensor masked_logits_25_cast_fp16 = select(a = scaled_logits_25_cast_fp16, b = var_17491_cast_fp16, cond = var_17490_cast_fp16)[name = string("masked_logits_25_cast_fp16")]; + tensor var_17495_begin_0 = const()[name = string("op_17495_begin_0"), val = tensor([12, 0])]; + tensor var_17495_end_0 = const()[name = string("op_17495_end_0"), val = tensor([13, 2048])]; + tensor var_17495_end_mask_0 = const()[name = string("op_17495_end_mask_0"), val = tensor([false, true])]; + tensor var_17495_squeeze_mask_0 = const()[name = string("op_17495_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_17495_cast_fp16 = slice_by_index(begin = var_17495_begin_0, end = var_17495_end_0, end_mask = var_17495_end_mask_0, squeeze_mask = var_17495_squeeze_mask_0, x = gumbel)[name = string("op_17495_cast_fp16")]; + tensor var_17498 = const()[name = string("op_17498"), val = tensor([1, 2048])]; + tensor var_17499_cast_fp16 = reshape(shape = var_17498, x = var_17495_cast_fp16)[name = string("op_17499_cast_fp16")]; + tensor noisy_logits_25_cast_fp16 = add(x = masked_logits_25_cast_fp16, y = var_17499_cast_fp16)[name = string("noisy_logits_25_cast_fp16")]; + int32 code_25_axis_0 = const()[name = string("code_25_axis_0"), val = int32(1)]; + bool code_25_keep_dims_0 = const()[name = string("code_25_keep_dims_0"), val = bool(false)]; + string code_25_output_dtype_0 = const()[name = string("code_25_output_dtype_0"), val = string("int32")]; + tensor code_25_cast_fp16 = reduce_argmax(axis = code_25_axis_0, keep_dims = code_25_keep_dims_0, output_dtype = code_25_output_dtype_0, x = noisy_logits_25_cast_fp16)[name = string("code_25_cast_fp16")]; + int32 var_17510 = const()[name = string("op_17510"), val = int32(24576)]; + tensor input_751 = add(x = code_25_cast_fp16, y = var_17510)[name = string("input_751")]; + int32 code_embed_49_axis_0 = const()[name = string("code_embed_49_axis_0"), val = int32(0)]; + int32 code_embed_49_batch_dims_0 = const()[name = string("code_embed_49_batch_dims_0"), val = int32(0)]; + bool code_embed_49_validate_indices_0 = const()[name = string("code_embed_49_validate_indices_0"), val = bool(false)]; + string input_751_to_uint16_dtype_0 = const()[name = string("input_751_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_751_to_uint16 = cast(dtype = input_751_to_uint16_dtype_0, x = input_751)[name = string("cast_2")]; + tensor code_embed_49_cast_fp16_cast_uint16 = gather(axis = code_embed_49_axis_0, batch_dims = code_embed_49_batch_dims_0, indices = input_751_to_uint16, validate_indices = code_embed_49_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_49_cast_fp16_cast_uint16")]; + tensor var_17514 = const()[name = string("op_17514"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_51_cast_fp16 = reshape(shape = var_17514, x = code_embed_49_cast_fp16_cast_uint16)[name = string("code_embed_51_cast_fp16")]; + tensor embed_sum_27_cast_fp16 = add(x = embed_sum_25_cast_fp16, y = code_embed_51_cast_fp16)[name = string("embed_sum_27_cast_fp16")]; + tensor key_cache_141_begin_0 = const()[name = string("key_cache_141_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_141_end_0 = const()[name = string("key_cache_141_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_141_end_mask_0 = const()[name = string("key_cache_141_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_141_cast_fp16 = slice_by_index(begin = key_cache_141_begin_0, end = key_cache_141_end_0, end_mask = key_cache_141_end_mask_0, x = layer_key_caches_29_cast_fp16)[name = string("key_cache_141_cast_fp16")]; + tensor value_cache_141_begin_0 = const()[name = string("value_cache_141_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_141_end_0 = const()[name = string("value_cache_141_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_141_end_mask_0 = const()[name = string("value_cache_141_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_141_cast_fp16 = slice_by_index(begin = value_cache_141_begin_0, end = value_cache_141_end_0, end_mask = value_cache_141_end_mask_0, x = layer_value_caches_29_cast_fp16)[name = string("value_cache_141_cast_fp16")]; + int32 var_17613 = const()[name = string("op_17613"), val = int32(2)]; + int32 var_17617 = const()[name = string("op_17617"), val = int32(3)]; + tensor var_17632_cast_fp16 = mul(x = code_embed_51_cast_fp16, y = code_embed_51_cast_fp16)[name = string("op_17632_cast_fp16")]; + tensor variance_587_axes_0 = const()[name = string("variance_587_axes_0"), val = tensor([1])]; + bool variance_587_keep_dims_0 = const()[name = string("variance_587_keep_dims_0"), val = bool(true)]; + tensor variance_587_cast_fp16 = reduce_mean(axes = variance_587_axes_0, keep_dims = variance_587_keep_dims_0, x = var_17632_cast_fp16)[name = string("variance_587_cast_fp16")]; + fp16 var_17635_to_fp16 = const()[name = string("op_17635_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17636_cast_fp16 = add(x = variance_587_cast_fp16, y = var_17635_to_fp16)[name = string("op_17636_cast_fp16")]; + fp32 var_17637_epsilon_0 = const()[name = string("op_17637_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17637_cast_fp16 = rsqrt(epsilon = var_17637_epsilon_0, x = var_17636_cast_fp16)[name = string("op_17637_cast_fp16")]; + tensor var_17638_cast_fp16 = mul(x = code_embed_51_cast_fp16, y = var_17637_cast_fp16)[name = string("op_17638_cast_fp16")]; + tensor input_753_cast_fp16 = mul(x = var_17638_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_753_cast_fp16")]; + string q_421_pad_type_0 = const()[name = string("q_421_pad_type_0"), val = string("valid")]; + tensor q_421_strides_0 = const()[name = string("q_421_strides_0"), val = tensor([1, 1])]; + tensor q_421_pad_0 = const()[name = string("q_421_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_421_dilations_0 = const()[name = string("q_421_dilations_0"), val = tensor([1, 1])]; + int32 q_421_groups_0 = const()[name = string("q_421_groups_0"), val = int32(1)]; + tensor q_421_cast_fp16 = conv(dilations = q_421_dilations_0, groups = q_421_groups_0, pad = q_421_pad_0, pad_type = q_421_pad_type_0, strides = q_421_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_753_cast_fp16)[name = string("q_421_cast_fp16")]; + string k_421_pad_type_0 = const()[name = string("k_421_pad_type_0"), val = string("valid")]; + tensor k_421_strides_0 = const()[name = string("k_421_strides_0"), val = tensor([1, 1])]; + tensor k_421_pad_0 = const()[name = string("k_421_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_421_dilations_0 = const()[name = string("k_421_dilations_0"), val = tensor([1, 1])]; + int32 k_421_groups_0 = const()[name = string("k_421_groups_0"), val = int32(1)]; + tensor k_421_cast_fp16 = conv(dilations = k_421_dilations_0, groups = k_421_groups_0, pad = k_421_pad_0, pad_type = k_421_pad_type_0, strides = k_421_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_753_cast_fp16)[name = string("k_421_cast_fp16")]; + string v_141_pad_type_0 = const()[name = string("v_141_pad_type_0"), val = string("valid")]; + tensor v_141_strides_0 = const()[name = string("v_141_strides_0"), val = tensor([1, 1])]; + tensor v_141_pad_0 = const()[name = string("v_141_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_141_dilations_0 = const()[name = string("v_141_dilations_0"), val = tensor([1, 1])]; + int32 v_141_groups_0 = const()[name = string("v_141_groups_0"), val = int32(1)]; + tensor v_141_cast_fp16 = conv(dilations = v_141_dilations_0, groups = v_141_groups_0, pad = v_141_pad_0, pad_type = v_141_pad_type_0, strides = v_141_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_753_cast_fp16)[name = string("v_141_cast_fp16")]; + tensor var_17672 = const()[name = string("op_17672"), val = tensor([16, 128, 1, 1])]; + tensor x_533_cast_fp16 = reshape(shape = var_17672, x = q_421_cast_fp16)[name = string("x_533_cast_fp16")]; + tensor var_17675_cast_fp16 = mul(x = x_533_cast_fp16, y = x_533_cast_fp16)[name = string("op_17675_cast_fp16")]; + tensor variance_589_axes_0 = const()[name = string("variance_589_axes_0"), val = tensor([1])]; + bool variance_589_keep_dims_0 = const()[name = string("variance_589_keep_dims_0"), val = bool(true)]; + tensor variance_589_cast_fp16 = reduce_mean(axes = variance_589_axes_0, keep_dims = variance_589_keep_dims_0, x = var_17675_cast_fp16)[name = string("variance_589_cast_fp16")]; + fp16 var_17678_to_fp16 = const()[name = string("op_17678_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17679_cast_fp16 = add(x = variance_589_cast_fp16, y = var_17678_to_fp16)[name = string("op_17679_cast_fp16")]; + fp32 var_17680_epsilon_0 = const()[name = string("op_17680_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17680_cast_fp16 = rsqrt(epsilon = var_17680_epsilon_0, x = var_17679_cast_fp16)[name = string("op_17680_cast_fp16")]; + tensor var_17681_cast_fp16 = mul(x = x_533_cast_fp16, y = var_17680_cast_fp16)[name = string("op_17681_cast_fp16")]; + tensor q_423_cast_fp16 = mul(x = var_17681_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_423_cast_fp16")]; + tensor var_17683 = const()[name = string("op_17683"), val = tensor([8, 128, 1, 1])]; + tensor x_535_cast_fp16 = reshape(shape = var_17683, x = k_421_cast_fp16)[name = string("x_535_cast_fp16")]; + tensor var_17686_cast_fp16 = mul(x = x_535_cast_fp16, y = x_535_cast_fp16)[name = string("op_17686_cast_fp16")]; + tensor variance_591_axes_0 = const()[name = string("variance_591_axes_0"), val = tensor([1])]; + bool variance_591_keep_dims_0 = const()[name = string("variance_591_keep_dims_0"), val = bool(true)]; + tensor variance_591_cast_fp16 = reduce_mean(axes = variance_591_axes_0, keep_dims = variance_591_keep_dims_0, x = var_17686_cast_fp16)[name = string("variance_591_cast_fp16")]; + fp16 var_17689_to_fp16 = const()[name = string("op_17689_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17690_cast_fp16 = add(x = variance_591_cast_fp16, y = var_17689_to_fp16)[name = string("op_17690_cast_fp16")]; + fp32 var_17691_epsilon_0 = const()[name = string("op_17691_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17691_cast_fp16 = rsqrt(epsilon = var_17691_epsilon_0, x = var_17690_cast_fp16)[name = string("op_17691_cast_fp16")]; + tensor var_17692_cast_fp16 = mul(x = x_535_cast_fp16, y = var_17691_cast_fp16)[name = string("op_17692_cast_fp16")]; + tensor k_423_cast_fp16 = mul(x = var_17692_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_423_cast_fp16")]; + tensor var_17694 = const()[name = string("op_17694"), val = tensor([1, 16, 128, 1])]; + tensor z_281_cast_fp16 = reshape(shape = var_17694, x = q_423_cast_fp16)[name = string("z_281_cast_fp16")]; + tensor var_17696 = const()[name = string("op_17696"), val = tensor([1, 8, 128, 1])]; + tensor z_283_cast_fp16 = reshape(shape = var_17696, x = k_423_cast_fp16)[name = string("z_283_cast_fp16")]; + tensor z1_281_begin_0 = const()[name = string("z1_281_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_281_end_0 = const()[name = string("z1_281_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_281_end_mask_0 = const()[name = string("z1_281_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_281_cast_fp16 = slice_by_index(begin = z1_281_begin_0, end = z1_281_end_0, end_mask = z1_281_end_mask_0, x = z_281_cast_fp16)[name = string("z1_281_cast_fp16")]; + tensor z2_281_begin_0 = const()[name = string("z2_281_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_281_end_0 = const()[name = string("z2_281_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_281_end_mask_0 = const()[name = string("z2_281_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_281_cast_fp16 = slice_by_index(begin = z2_281_begin_0, end = z2_281_end_0, end_mask = z2_281_end_mask_0, x = z_281_cast_fp16)[name = string("z2_281_cast_fp16")]; + tensor cos_141_to_fp16 = const()[name = string("cos_141_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141645696)))]; + tensor var_17704_cast_fp16 = mul(x = z_281_cast_fp16, y = cos_141_to_fp16)[name = string("op_17704_cast_fp16")]; + fp16 const_155_promoted_to_fp16 = const()[name = string("const_155_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17705_cast_fp16 = mul(x = z2_281_cast_fp16, y = const_155_promoted_to_fp16)[name = string("op_17705_cast_fp16")]; + bool var_17707_interleave_0 = const()[name = string("op_17707_interleave_0"), val = bool(false)]; + tensor var_17707_cast_fp16 = concat(axis = var_17613, interleave = var_17707_interleave_0, values = (var_17705_cast_fp16, z1_281_cast_fp16))[name = string("op_17707_cast_fp16")]; + tensor sin_141_to_fp16 = const()[name = string("sin_141_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646016)))]; + tensor var_17708_cast_fp16 = mul(x = var_17707_cast_fp16, y = sin_141_to_fp16)[name = string("op_17708_cast_fp16")]; + tensor q_425_cast_fp16 = add(x = var_17704_cast_fp16, y = var_17708_cast_fp16)[name = string("q_425_cast_fp16")]; + tensor z1_283_begin_0 = const()[name = string("z1_283_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_283_end_0 = const()[name = string("z1_283_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_283_end_mask_0 = const()[name = string("z1_283_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_283_cast_fp16 = slice_by_index(begin = z1_283_begin_0, end = z1_283_end_0, end_mask = z1_283_end_mask_0, x = z_283_cast_fp16)[name = string("z1_283_cast_fp16")]; + tensor z2_283_begin_0 = const()[name = string("z2_283_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_283_end_0 = const()[name = string("z2_283_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_283_end_mask_0 = const()[name = string("z2_283_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_283_cast_fp16 = slice_by_index(begin = z2_283_begin_0, end = z2_283_end_0, end_mask = z2_283_end_mask_0, x = z_283_cast_fp16)[name = string("z2_283_cast_fp16")]; + tensor var_17716_cast_fp16 = mul(x = z_283_cast_fp16, y = cos_141_to_fp16)[name = string("op_17716_cast_fp16")]; + fp16 const_156_promoted_to_fp16 = const()[name = string("const_156_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17717_cast_fp16 = mul(x = z2_283_cast_fp16, y = const_156_promoted_to_fp16)[name = string("op_17717_cast_fp16")]; + bool var_17719_interleave_0 = const()[name = string("op_17719_interleave_0"), val = bool(false)]; + tensor var_17719_cast_fp16 = concat(axis = var_17613, interleave = var_17719_interleave_0, values = (var_17717_cast_fp16, z1_283_cast_fp16))[name = string("op_17719_cast_fp16")]; + tensor var_17720_cast_fp16 = mul(x = var_17719_cast_fp16, y = sin_141_to_fp16)[name = string("op_17720_cast_fp16")]; + tensor k_425_cast_fp16 = add(x = var_17716_cast_fp16, y = var_17720_cast_fp16)[name = string("k_425_cast_fp16")]; + tensor var_17722 = const()[name = string("op_17722"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_141_cast_fp16 = reshape(shape = var_17722, x = k_425_cast_fp16)[name = string("cur_key_141_cast_fp16")]; + tensor var_17724_to_fp16 = const()[name = string("op_17724_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646336)))]; + tensor var_17725_cast_fp16 = mul(x = key_cache_141_cast_fp16, y = var_17724_to_fp16)[name = string("op_17725_cast_fp16")]; + tensor upd_141_to_fp16 = const()[name = string("upd_141_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646464)))]; + tensor var_17726_cast_fp16 = mul(x = cur_key_141_cast_fp16, y = upd_141_to_fp16)[name = string("op_17726_cast_fp16")]; + tensor key_141_cast_fp16 = add(x = var_17725_cast_fp16, y = var_17726_cast_fp16)[name = string("key_141_cast_fp16")]; + tensor var_17728_to_fp16 = const()[name = string("op_17728_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646336)))]; + tensor var_17729_cast_fp16 = mul(x = value_cache_141_cast_fp16, y = var_17728_to_fp16)[name = string("op_17729_cast_fp16")]; + tensor var_17730_cast_fp16 = mul(x = v_141_cast_fp16, y = upd_141_to_fp16)[name = string("op_17730_cast_fp16")]; + tensor value_141_cast_fp16 = add(x = var_17729_cast_fp16, y = var_17730_cast_fp16)[name = string("value_141_cast_fp16")]; + tensor var_17732 = const()[name = string("op_17732"), val = tensor([1, 8, 128, 16])]; + tensor kh_281_cast_fp16 = reshape(shape = var_17732, x = key_141_cast_fp16)[name = string("kh_281_cast_fp16")]; + tensor var_17734 = const()[name = string("op_17734"), val = tensor([1, 8, 128, 16])]; + tensor vh_281_cast_fp16 = reshape(shape = var_17734, x = value_141_cast_fp16)[name = string("vh_281_cast_fp16")]; + tensor transpose_280_perm_0 = const()[name = string("transpose_280_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_140_reps_0 = const()[name = string("tile_140_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_280_cast_fp16 = transpose(perm = transpose_280_perm_0, x = kh_281_cast_fp16)[name = string("transpose_59")]; + tensor tile_140_cast_fp16 = tile(reps = tile_140_reps_0, x = transpose_280_cast_fp16)[name = string("tile_140_cast_fp16")]; + tensor concat_353 = const()[name = string("concat_353"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_280_cast_fp16 = reshape(shape = concat_353, x = tile_140_cast_fp16)[name = string("reshape_280_cast_fp16")]; + tensor transpose_281_perm_0 = const()[name = string("transpose_281_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_354 = const()[name = string("concat_354"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_281_cast_fp16 = transpose(perm = transpose_281_perm_0, x = reshape_280_cast_fp16)[name = string("transpose_58")]; + tensor reshape_281_cast_fp16 = reshape(shape = concat_354, x = transpose_281_cast_fp16)[name = string("reshape_281_cast_fp16")]; + tensor transpose_282_perm_0 = const()[name = string("transpose_282_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_141_reps_0 = const()[name = string("tile_141_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_282_cast_fp16 = transpose(perm = transpose_282_perm_0, x = vh_281_cast_fp16)[name = string("transpose_57")]; + tensor tile_141_cast_fp16 = tile(reps = tile_141_reps_0, x = transpose_282_cast_fp16)[name = string("tile_141_cast_fp16")]; + tensor concat_355 = const()[name = string("concat_355"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_282_cast_fp16 = reshape(shape = concat_355, x = tile_141_cast_fp16)[name = string("reshape_282_cast_fp16")]; + tensor transpose_283_perm_0 = const()[name = string("transpose_283_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_356 = const()[name = string("concat_356"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_283_cast_fp16 = transpose(perm = transpose_283_perm_0, x = reshape_282_cast_fp16)[name = string("transpose_56")]; + tensor reshape_283_cast_fp16 = reshape(shape = concat_356, x = transpose_283_cast_fp16)[name = string("reshape_283_cast_fp16")]; + fp16 var_17738_to_fp16 = const()[name = string("op_17738_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_17739_cast_fp16 = mul(x = q_425_cast_fp16, y = var_17738_to_fp16)[name = string("op_17739_cast_fp16")]; + tensor transpose_597_perm_0 = const()[name = string("transpose_597_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_307_transpose_x_1 = const()[name = string("w_307_transpose_x_1"), val = bool(true)]; + bool w_307_transpose_y_1 = const()[name = string("w_307_transpose_y_1"), val = bool(false)]; + tensor transpose_597_cast_fp16 = transpose(perm = transpose_597_perm_0, x = reshape_281_cast_fp16)[name = string("transpose_55")]; + tensor w_307_cast_fp16 = matmul(transpose_x = w_307_transpose_x_1, transpose_y = w_307_transpose_y_1, x = var_17739_cast_fp16, y = transpose_597_cast_fp16)[name = string("w_307_cast_fp16")]; + tensor pad_141_to_fp16 = const()[name = string("pad_141_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646592)))]; + tensor var_17742_cast_fp16 = add(x = w_307_cast_fp16, y = pad_141_to_fp16)[name = string("op_17742_cast_fp16")]; + tensor w_309_cast_fp16 = softmax(axis = var_17617, x = var_17742_cast_fp16)[name = string("w_309_cast_fp16")]; + tensor transpose_598_perm_0 = const()[name = string("transpose_598_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_141_transpose_x_1 = const()[name = string("attn_141_transpose_x_1"), val = bool(false)]; + bool attn_141_transpose_y_1 = const()[name = string("attn_141_transpose_y_1"), val = bool(true)]; + tensor transpose_598_cast_fp16 = transpose(perm = transpose_598_perm_0, x = reshape_283_cast_fp16)[name = string("transpose_54")]; + tensor attn_141_cast_fp16 = matmul(transpose_x = attn_141_transpose_x_1, transpose_y = attn_141_transpose_y_1, x = transpose_598_cast_fp16, y = w_309_cast_fp16)[name = string("attn_141_cast_fp16")]; + tensor var_17746 = const()[name = string("op_17746"), val = tensor([1, 2048, 1, 1])]; + tensor input_755_cast_fp16 = reshape(shape = var_17746, x = attn_141_cast_fp16)[name = string("input_755_cast_fp16")]; + string attn_output_141_pad_type_0 = const()[name = string("attn_output_141_pad_type_0"), val = string("valid")]; + tensor attn_output_141_strides_0 = const()[name = string("attn_output_141_strides_0"), val = tensor([1, 1])]; + tensor attn_output_141_pad_0 = const()[name = string("attn_output_141_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_141_dilations_0 = const()[name = string("attn_output_141_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_141_groups_0 = const()[name = string("attn_output_141_groups_0"), val = int32(1)]; + tensor attn_output_141_cast_fp16 = conv(dilations = attn_output_141_dilations_0, groups = attn_output_141_groups_0, pad = attn_output_141_pad_0, pad_type = attn_output_141_pad_type_0, strides = attn_output_141_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_755_cast_fp16)[name = string("attn_output_141_cast_fp16")]; + tensor x_537_cast_fp16 = add(x = code_embed_51_cast_fp16, y = attn_output_141_cast_fp16)[name = string("x_537_cast_fp16")]; + tensor var_17760_cast_fp16 = mul(x = x_537_cast_fp16, y = x_537_cast_fp16)[name = string("op_17760_cast_fp16")]; + tensor variance_593_axes_0 = const()[name = string("variance_593_axes_0"), val = tensor([1])]; + bool variance_593_keep_dims_0 = const()[name = string("variance_593_keep_dims_0"), val = bool(true)]; + tensor variance_593_cast_fp16 = reduce_mean(axes = variance_593_axes_0, keep_dims = variance_593_keep_dims_0, x = var_17760_cast_fp16)[name = string("variance_593_cast_fp16")]; + fp16 var_17763_to_fp16 = const()[name = string("op_17763_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17764_cast_fp16 = add(x = variance_593_cast_fp16, y = var_17763_to_fp16)[name = string("op_17764_cast_fp16")]; + fp32 var_17765_epsilon_0 = const()[name = string("op_17765_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17765_cast_fp16 = rsqrt(epsilon = var_17765_epsilon_0, x = var_17764_cast_fp16)[name = string("op_17765_cast_fp16")]; + tensor var_17766_cast_fp16 = mul(x = x_537_cast_fp16, y = var_17765_cast_fp16)[name = string("op_17766_cast_fp16")]; + tensor input_757_cast_fp16 = mul(x = var_17766_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_757_cast_fp16")]; + string input_759_pad_type_0 = const()[name = string("input_759_pad_type_0"), val = string("valid")]; + tensor input_759_strides_0 = const()[name = string("input_759_strides_0"), val = tensor([1, 1])]; + tensor input_759_pad_0 = const()[name = string("input_759_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_759_dilations_0 = const()[name = string("input_759_dilations_0"), val = tensor([1, 1])]; + int32 input_759_groups_0 = const()[name = string("input_759_groups_0"), val = int32(1)]; + tensor input_759_cast_fp16 = conv(dilations = input_759_dilations_0, groups = input_759_groups_0, pad = input_759_pad_0, pad_type = input_759_pad_type_0, strides = input_759_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_757_cast_fp16)[name = string("input_759_cast_fp16")]; + tensor var_17774_cast_fp16 = silu(x = input_759_cast_fp16)[name = string("op_17774_cast_fp16")]; + string var_17780_pad_type_0 = const()[name = string("op_17780_pad_type_0"), val = string("valid")]; + tensor var_17780_strides_0 = const()[name = string("op_17780_strides_0"), val = tensor([1, 1])]; + tensor var_17780_pad_0 = const()[name = string("op_17780_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17780_dilations_0 = const()[name = string("op_17780_dilations_0"), val = tensor([1, 1])]; + int32 var_17780_groups_0 = const()[name = string("op_17780_groups_0"), val = int32(1)]; + tensor var_17780_cast_fp16 = conv(dilations = var_17780_dilations_0, groups = var_17780_groups_0, pad = var_17780_pad_0, pad_type = var_17780_pad_type_0, strides = var_17780_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_757_cast_fp16)[name = string("op_17780_cast_fp16")]; + tensor input_761_cast_fp16 = mul(x = var_17774_cast_fp16, y = var_17780_cast_fp16)[name = string("input_761_cast_fp16")]; + string h_141_pad_type_0 = const()[name = string("h_141_pad_type_0"), val = string("valid")]; + tensor h_141_strides_0 = const()[name = string("h_141_strides_0"), val = tensor([1, 1])]; + tensor h_141_pad_0 = const()[name = string("h_141_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_141_dilations_0 = const()[name = string("h_141_dilations_0"), val = tensor([1, 1])]; + int32 h_141_groups_0 = const()[name = string("h_141_groups_0"), val = int32(1)]; + tensor h_141_cast_fp16 = conv(dilations = h_141_dilations_0, groups = h_141_groups_0, pad = h_141_pad_0, pad_type = h_141_pad_type_0, strides = h_141_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_761_cast_fp16)[name = string("h_141_cast_fp16")]; + tensor x_539_cast_fp16 = add(x = x_537_cast_fp16, y = h_141_cast_fp16)[name = string("x_539_cast_fp16")]; + tensor key_cache_143_begin_0 = const()[name = string("key_cache_143_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_143_end_0 = const()[name = string("key_cache_143_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_143_end_mask_0 = const()[name = string("key_cache_143_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_143_cast_fp16 = slice_by_index(begin = key_cache_143_begin_0, end = key_cache_143_end_0, end_mask = key_cache_143_end_mask_0, x = layer_key_caches_29_cast_fp16)[name = string("key_cache_143_cast_fp16")]; + tensor value_cache_143_begin_0 = const()[name = string("value_cache_143_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_143_end_0 = const()[name = string("value_cache_143_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_143_end_mask_0 = const()[name = string("value_cache_143_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_143_cast_fp16 = slice_by_index(begin = value_cache_143_begin_0, end = value_cache_143_end_0, end_mask = value_cache_143_end_mask_0, x = layer_value_caches_29_cast_fp16)[name = string("value_cache_143_cast_fp16")]; + int32 var_17833 = const()[name = string("op_17833"), val = int32(2)]; + int32 var_17837 = const()[name = string("op_17837"), val = int32(3)]; + tensor var_17852_cast_fp16 = mul(x = x_539_cast_fp16, y = x_539_cast_fp16)[name = string("op_17852_cast_fp16")]; + tensor variance_595_axes_0 = const()[name = string("variance_595_axes_0"), val = tensor([1])]; + bool variance_595_keep_dims_0 = const()[name = string("variance_595_keep_dims_0"), val = bool(true)]; + tensor variance_595_cast_fp16 = reduce_mean(axes = variance_595_axes_0, keep_dims = variance_595_keep_dims_0, x = var_17852_cast_fp16)[name = string("variance_595_cast_fp16")]; + fp16 var_17855_to_fp16 = const()[name = string("op_17855_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17856_cast_fp16 = add(x = variance_595_cast_fp16, y = var_17855_to_fp16)[name = string("op_17856_cast_fp16")]; + fp32 var_17857_epsilon_0 = const()[name = string("op_17857_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17857_cast_fp16 = rsqrt(epsilon = var_17857_epsilon_0, x = var_17856_cast_fp16)[name = string("op_17857_cast_fp16")]; + tensor var_17858_cast_fp16 = mul(x = x_539_cast_fp16, y = var_17857_cast_fp16)[name = string("op_17858_cast_fp16")]; + tensor input_763_cast_fp16 = mul(x = var_17858_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_763_cast_fp16")]; + string q_427_pad_type_0 = const()[name = string("q_427_pad_type_0"), val = string("valid")]; + tensor q_427_strides_0 = const()[name = string("q_427_strides_0"), val = tensor([1, 1])]; + tensor q_427_pad_0 = const()[name = string("q_427_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_427_dilations_0 = const()[name = string("q_427_dilations_0"), val = tensor([1, 1])]; + int32 q_427_groups_0 = const()[name = string("q_427_groups_0"), val = int32(1)]; + tensor q_427_cast_fp16 = conv(dilations = q_427_dilations_0, groups = q_427_groups_0, pad = q_427_pad_0, pad_type = q_427_pad_type_0, strides = q_427_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_763_cast_fp16)[name = string("q_427_cast_fp16")]; + string k_427_pad_type_0 = const()[name = string("k_427_pad_type_0"), val = string("valid")]; + tensor k_427_strides_0 = const()[name = string("k_427_strides_0"), val = tensor([1, 1])]; + tensor k_427_pad_0 = const()[name = string("k_427_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_427_dilations_0 = const()[name = string("k_427_dilations_0"), val = tensor([1, 1])]; + int32 k_427_groups_0 = const()[name = string("k_427_groups_0"), val = int32(1)]; + tensor k_427_cast_fp16 = conv(dilations = k_427_dilations_0, groups = k_427_groups_0, pad = k_427_pad_0, pad_type = k_427_pad_type_0, strides = k_427_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_763_cast_fp16)[name = string("k_427_cast_fp16")]; + string v_143_pad_type_0 = const()[name = string("v_143_pad_type_0"), val = string("valid")]; + tensor v_143_strides_0 = const()[name = string("v_143_strides_0"), val = tensor([1, 1])]; + tensor v_143_pad_0 = const()[name = string("v_143_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_143_dilations_0 = const()[name = string("v_143_dilations_0"), val = tensor([1, 1])]; + int32 v_143_groups_0 = const()[name = string("v_143_groups_0"), val = int32(1)]; + tensor v_143_cast_fp16 = conv(dilations = v_143_dilations_0, groups = v_143_groups_0, pad = v_143_pad_0, pad_type = v_143_pad_type_0, strides = v_143_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_763_cast_fp16)[name = string("v_143_cast_fp16")]; + tensor var_17892 = const()[name = string("op_17892"), val = tensor([16, 128, 1, 1])]; + tensor x_541_cast_fp16 = reshape(shape = var_17892, x = q_427_cast_fp16)[name = string("x_541_cast_fp16")]; + tensor var_17895_cast_fp16 = mul(x = x_541_cast_fp16, y = x_541_cast_fp16)[name = string("op_17895_cast_fp16")]; + tensor variance_597_axes_0 = const()[name = string("variance_597_axes_0"), val = tensor([1])]; + bool variance_597_keep_dims_0 = const()[name = string("variance_597_keep_dims_0"), val = bool(true)]; + tensor variance_597_cast_fp16 = reduce_mean(axes = variance_597_axes_0, keep_dims = variance_597_keep_dims_0, x = var_17895_cast_fp16)[name = string("variance_597_cast_fp16")]; + fp16 var_17898_to_fp16 = const()[name = string("op_17898_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17899_cast_fp16 = add(x = variance_597_cast_fp16, y = var_17898_to_fp16)[name = string("op_17899_cast_fp16")]; + fp32 var_17900_epsilon_0 = const()[name = string("op_17900_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17900_cast_fp16 = rsqrt(epsilon = var_17900_epsilon_0, x = var_17899_cast_fp16)[name = string("op_17900_cast_fp16")]; + tensor var_17901_cast_fp16 = mul(x = x_541_cast_fp16, y = var_17900_cast_fp16)[name = string("op_17901_cast_fp16")]; + tensor q_429_cast_fp16 = mul(x = var_17901_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_429_cast_fp16")]; + tensor var_17903 = const()[name = string("op_17903"), val = tensor([8, 128, 1, 1])]; + tensor x_543_cast_fp16 = reshape(shape = var_17903, x = k_427_cast_fp16)[name = string("x_543_cast_fp16")]; + tensor var_17906_cast_fp16 = mul(x = x_543_cast_fp16, y = x_543_cast_fp16)[name = string("op_17906_cast_fp16")]; + tensor variance_599_axes_0 = const()[name = string("variance_599_axes_0"), val = tensor([1])]; + bool variance_599_keep_dims_0 = const()[name = string("variance_599_keep_dims_0"), val = bool(true)]; + tensor variance_599_cast_fp16 = reduce_mean(axes = variance_599_axes_0, keep_dims = variance_599_keep_dims_0, x = var_17906_cast_fp16)[name = string("variance_599_cast_fp16")]; + fp16 var_17909_to_fp16 = const()[name = string("op_17909_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17910_cast_fp16 = add(x = variance_599_cast_fp16, y = var_17909_to_fp16)[name = string("op_17910_cast_fp16")]; + fp32 var_17911_epsilon_0 = const()[name = string("op_17911_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17911_cast_fp16 = rsqrt(epsilon = var_17911_epsilon_0, x = var_17910_cast_fp16)[name = string("op_17911_cast_fp16")]; + tensor var_17912_cast_fp16 = mul(x = x_543_cast_fp16, y = var_17911_cast_fp16)[name = string("op_17912_cast_fp16")]; + tensor k_429_cast_fp16 = mul(x = var_17912_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_429_cast_fp16")]; + tensor var_17914 = const()[name = string("op_17914"), val = tensor([1, 16, 128, 1])]; + tensor z_285_cast_fp16 = reshape(shape = var_17914, x = q_429_cast_fp16)[name = string("z_285_cast_fp16")]; + tensor var_17916 = const()[name = string("op_17916"), val = tensor([1, 8, 128, 1])]; + tensor z_287_cast_fp16 = reshape(shape = var_17916, x = k_429_cast_fp16)[name = string("z_287_cast_fp16")]; + tensor z1_285_begin_0 = const()[name = string("z1_285_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_285_end_0 = const()[name = string("z1_285_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_285_end_mask_0 = const()[name = string("z1_285_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_285_cast_fp16 = slice_by_index(begin = z1_285_begin_0, end = z1_285_end_0, end_mask = z1_285_end_mask_0, x = z_285_cast_fp16)[name = string("z1_285_cast_fp16")]; + tensor z2_285_begin_0 = const()[name = string("z2_285_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_285_end_0 = const()[name = string("z2_285_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_285_end_mask_0 = const()[name = string("z2_285_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_285_cast_fp16 = slice_by_index(begin = z2_285_begin_0, end = z2_285_end_0, end_mask = z2_285_end_mask_0, x = z_285_cast_fp16)[name = string("z2_285_cast_fp16")]; + tensor var_17924_cast_fp16 = mul(x = z_285_cast_fp16, y = cos_141_to_fp16)[name = string("op_17924_cast_fp16")]; + fp16 const_157_promoted_to_fp16 = const()[name = string("const_157_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17925_cast_fp16 = mul(x = z2_285_cast_fp16, y = const_157_promoted_to_fp16)[name = string("op_17925_cast_fp16")]; + bool var_17927_interleave_0 = const()[name = string("op_17927_interleave_0"), val = bool(false)]; + tensor var_17927_cast_fp16 = concat(axis = var_17833, interleave = var_17927_interleave_0, values = (var_17925_cast_fp16, z1_285_cast_fp16))[name = string("op_17927_cast_fp16")]; + tensor var_17928_cast_fp16 = mul(x = var_17927_cast_fp16, y = sin_141_to_fp16)[name = string("op_17928_cast_fp16")]; + tensor q_431_cast_fp16 = add(x = var_17924_cast_fp16, y = var_17928_cast_fp16)[name = string("q_431_cast_fp16")]; + tensor z1_287_begin_0 = const()[name = string("z1_287_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_287_end_0 = const()[name = string("z1_287_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_287_end_mask_0 = const()[name = string("z1_287_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_287_cast_fp16 = slice_by_index(begin = z1_287_begin_0, end = z1_287_end_0, end_mask = z1_287_end_mask_0, x = z_287_cast_fp16)[name = string("z1_287_cast_fp16")]; + tensor z2_287_begin_0 = const()[name = string("z2_287_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_287_end_0 = const()[name = string("z2_287_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_287_end_mask_0 = const()[name = string("z2_287_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_287_cast_fp16 = slice_by_index(begin = z2_287_begin_0, end = z2_287_end_0, end_mask = z2_287_end_mask_0, x = z_287_cast_fp16)[name = string("z2_287_cast_fp16")]; + tensor var_17936_cast_fp16 = mul(x = z_287_cast_fp16, y = cos_141_to_fp16)[name = string("op_17936_cast_fp16")]; + fp16 const_158_promoted_to_fp16 = const()[name = string("const_158_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17937_cast_fp16 = mul(x = z2_287_cast_fp16, y = const_158_promoted_to_fp16)[name = string("op_17937_cast_fp16")]; + bool var_17939_interleave_0 = const()[name = string("op_17939_interleave_0"), val = bool(false)]; + tensor var_17939_cast_fp16 = concat(axis = var_17833, interleave = var_17939_interleave_0, values = (var_17937_cast_fp16, z1_287_cast_fp16))[name = string("op_17939_cast_fp16")]; + tensor var_17940_cast_fp16 = mul(x = var_17939_cast_fp16, y = sin_141_to_fp16)[name = string("op_17940_cast_fp16")]; + tensor k_431_cast_fp16 = add(x = var_17936_cast_fp16, y = var_17940_cast_fp16)[name = string("k_431_cast_fp16")]; + tensor var_17942 = const()[name = string("op_17942"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_143_cast_fp16 = reshape(shape = var_17942, x = k_431_cast_fp16)[name = string("cur_key_143_cast_fp16")]; + tensor var_17944_to_fp16 = const()[name = string("op_17944_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646336)))]; + tensor var_17945_cast_fp16 = mul(x = key_cache_143_cast_fp16, y = var_17944_to_fp16)[name = string("op_17945_cast_fp16")]; + tensor upd_143_to_fp16 = const()[name = string("upd_143_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646464)))]; + tensor var_17946_cast_fp16 = mul(x = cur_key_143_cast_fp16, y = upd_143_to_fp16)[name = string("op_17946_cast_fp16")]; + tensor key_143_cast_fp16 = add(x = var_17945_cast_fp16, y = var_17946_cast_fp16)[name = string("key_143_cast_fp16")]; + tensor var_17948_to_fp16 = const()[name = string("op_17948_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646336)))]; + tensor var_17949_cast_fp16 = mul(x = value_cache_143_cast_fp16, y = var_17948_to_fp16)[name = string("op_17949_cast_fp16")]; + tensor var_17950_cast_fp16 = mul(x = v_143_cast_fp16, y = upd_143_to_fp16)[name = string("op_17950_cast_fp16")]; + tensor value_143_cast_fp16 = add(x = var_17949_cast_fp16, y = var_17950_cast_fp16)[name = string("value_143_cast_fp16")]; + tensor var_17952 = const()[name = string("op_17952"), val = tensor([1, 8, 128, 16])]; + tensor kh_285_cast_fp16 = reshape(shape = var_17952, x = key_143_cast_fp16)[name = string("kh_285_cast_fp16")]; + tensor var_17954 = const()[name = string("op_17954"), val = tensor([1, 8, 128, 16])]; + tensor vh_285_cast_fp16 = reshape(shape = var_17954, x = value_143_cast_fp16)[name = string("vh_285_cast_fp16")]; + tensor transpose_284_perm_0 = const()[name = string("transpose_284_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_142_reps_0 = const()[name = string("tile_142_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_284_cast_fp16 = transpose(perm = transpose_284_perm_0, x = kh_285_cast_fp16)[name = string("transpose_53")]; + tensor tile_142_cast_fp16 = tile(reps = tile_142_reps_0, x = transpose_284_cast_fp16)[name = string("tile_142_cast_fp16")]; + tensor concat_357 = const()[name = string("concat_357"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_284_cast_fp16 = reshape(shape = concat_357, x = tile_142_cast_fp16)[name = string("reshape_284_cast_fp16")]; + tensor transpose_285_perm_0 = const()[name = string("transpose_285_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_358 = const()[name = string("concat_358"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_285_cast_fp16 = transpose(perm = transpose_285_perm_0, x = reshape_284_cast_fp16)[name = string("transpose_52")]; + tensor reshape_285_cast_fp16 = reshape(shape = concat_358, x = transpose_285_cast_fp16)[name = string("reshape_285_cast_fp16")]; + tensor transpose_286_perm_0 = const()[name = string("transpose_286_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_143_reps_0 = const()[name = string("tile_143_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_286_cast_fp16 = transpose(perm = transpose_286_perm_0, x = vh_285_cast_fp16)[name = string("transpose_51")]; + tensor tile_143_cast_fp16 = tile(reps = tile_143_reps_0, x = transpose_286_cast_fp16)[name = string("tile_143_cast_fp16")]; + tensor concat_359 = const()[name = string("concat_359"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_286_cast_fp16 = reshape(shape = concat_359, x = tile_143_cast_fp16)[name = string("reshape_286_cast_fp16")]; + tensor transpose_287_perm_0 = const()[name = string("transpose_287_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_360 = const()[name = string("concat_360"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_287_cast_fp16 = transpose(perm = transpose_287_perm_0, x = reshape_286_cast_fp16)[name = string("transpose_50")]; + tensor reshape_287_cast_fp16 = reshape(shape = concat_360, x = transpose_287_cast_fp16)[name = string("reshape_287_cast_fp16")]; + fp16 var_17958_to_fp16 = const()[name = string("op_17958_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_17959_cast_fp16 = mul(x = q_431_cast_fp16, y = var_17958_to_fp16)[name = string("op_17959_cast_fp16")]; + tensor transpose_601_perm_0 = const()[name = string("transpose_601_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_311_transpose_x_1 = const()[name = string("w_311_transpose_x_1"), val = bool(true)]; + bool w_311_transpose_y_1 = const()[name = string("w_311_transpose_y_1"), val = bool(false)]; + tensor transpose_601_cast_fp16 = transpose(perm = transpose_601_perm_0, x = reshape_285_cast_fp16)[name = string("transpose_49")]; + tensor w_311_cast_fp16 = matmul(transpose_x = w_311_transpose_x_1, transpose_y = w_311_transpose_y_1, x = var_17959_cast_fp16, y = transpose_601_cast_fp16)[name = string("w_311_cast_fp16")]; + tensor pad_143_to_fp16 = const()[name = string("pad_143_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646592)))]; + tensor var_17962_cast_fp16 = add(x = w_311_cast_fp16, y = pad_143_to_fp16)[name = string("op_17962_cast_fp16")]; + tensor w_313_cast_fp16 = softmax(axis = var_17837, x = var_17962_cast_fp16)[name = string("w_313_cast_fp16")]; + tensor transpose_602_perm_0 = const()[name = string("transpose_602_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_143_transpose_x_1 = const()[name = string("attn_143_transpose_x_1"), val = bool(false)]; + bool attn_143_transpose_y_1 = const()[name = string("attn_143_transpose_y_1"), val = bool(true)]; + tensor transpose_602_cast_fp16 = transpose(perm = transpose_602_perm_0, x = reshape_287_cast_fp16)[name = string("transpose_48")]; + tensor attn_143_cast_fp16 = matmul(transpose_x = attn_143_transpose_x_1, transpose_y = attn_143_transpose_y_1, x = transpose_602_cast_fp16, y = w_313_cast_fp16)[name = string("attn_143_cast_fp16")]; + tensor var_17966 = const()[name = string("op_17966"), val = tensor([1, 2048, 1, 1])]; + tensor input_765_cast_fp16 = reshape(shape = var_17966, x = attn_143_cast_fp16)[name = string("input_765_cast_fp16")]; + string attn_output_143_pad_type_0 = const()[name = string("attn_output_143_pad_type_0"), val = string("valid")]; + tensor attn_output_143_strides_0 = const()[name = string("attn_output_143_strides_0"), val = tensor([1, 1])]; + tensor attn_output_143_pad_0 = const()[name = string("attn_output_143_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_143_dilations_0 = const()[name = string("attn_output_143_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_143_groups_0 = const()[name = string("attn_output_143_groups_0"), val = int32(1)]; + tensor attn_output_143_cast_fp16 = conv(dilations = attn_output_143_dilations_0, groups = attn_output_143_groups_0, pad = attn_output_143_pad_0, pad_type = attn_output_143_pad_type_0, strides = attn_output_143_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_765_cast_fp16)[name = string("attn_output_143_cast_fp16")]; + tensor x_545_cast_fp16 = add(x = x_539_cast_fp16, y = attn_output_143_cast_fp16)[name = string("x_545_cast_fp16")]; + tensor var_17980_cast_fp16 = mul(x = x_545_cast_fp16, y = x_545_cast_fp16)[name = string("op_17980_cast_fp16")]; + tensor variance_601_axes_0 = const()[name = string("variance_601_axes_0"), val = tensor([1])]; + bool variance_601_keep_dims_0 = const()[name = string("variance_601_keep_dims_0"), val = bool(true)]; + tensor variance_601_cast_fp16 = reduce_mean(axes = variance_601_axes_0, keep_dims = variance_601_keep_dims_0, x = var_17980_cast_fp16)[name = string("variance_601_cast_fp16")]; + fp16 var_17983_to_fp16 = const()[name = string("op_17983_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17984_cast_fp16 = add(x = variance_601_cast_fp16, y = var_17983_to_fp16)[name = string("op_17984_cast_fp16")]; + fp32 var_17985_epsilon_0 = const()[name = string("op_17985_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17985_cast_fp16 = rsqrt(epsilon = var_17985_epsilon_0, x = var_17984_cast_fp16)[name = string("op_17985_cast_fp16")]; + tensor var_17986_cast_fp16 = mul(x = x_545_cast_fp16, y = var_17985_cast_fp16)[name = string("op_17986_cast_fp16")]; + tensor input_767_cast_fp16 = mul(x = var_17986_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_767_cast_fp16")]; + string input_769_pad_type_0 = const()[name = string("input_769_pad_type_0"), val = string("valid")]; + tensor input_769_strides_0 = const()[name = string("input_769_strides_0"), val = tensor([1, 1])]; + tensor input_769_pad_0 = const()[name = string("input_769_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_769_dilations_0 = const()[name = string("input_769_dilations_0"), val = tensor([1, 1])]; + int32 input_769_groups_0 = const()[name = string("input_769_groups_0"), val = int32(1)]; + tensor input_769_cast_fp16 = conv(dilations = input_769_dilations_0, groups = input_769_groups_0, pad = input_769_pad_0, pad_type = input_769_pad_type_0, strides = input_769_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_767_cast_fp16)[name = string("input_769_cast_fp16")]; + tensor var_17994_cast_fp16 = silu(x = input_769_cast_fp16)[name = string("op_17994_cast_fp16")]; + string var_18000_pad_type_0 = const()[name = string("op_18000_pad_type_0"), val = string("valid")]; + tensor var_18000_strides_0 = const()[name = string("op_18000_strides_0"), val = tensor([1, 1])]; + tensor var_18000_pad_0 = const()[name = string("op_18000_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18000_dilations_0 = const()[name = string("op_18000_dilations_0"), val = tensor([1, 1])]; + int32 var_18000_groups_0 = const()[name = string("op_18000_groups_0"), val = int32(1)]; + tensor var_18000_cast_fp16 = conv(dilations = var_18000_dilations_0, groups = var_18000_groups_0, pad = var_18000_pad_0, pad_type = var_18000_pad_type_0, strides = var_18000_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_767_cast_fp16)[name = string("op_18000_cast_fp16")]; + tensor input_771_cast_fp16 = mul(x = var_17994_cast_fp16, y = var_18000_cast_fp16)[name = string("input_771_cast_fp16")]; + string h_143_pad_type_0 = const()[name = string("h_143_pad_type_0"), val = string("valid")]; + tensor h_143_strides_0 = const()[name = string("h_143_strides_0"), val = tensor([1, 1])]; + tensor h_143_pad_0 = const()[name = string("h_143_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_143_dilations_0 = const()[name = string("h_143_dilations_0"), val = tensor([1, 1])]; + int32 h_143_groups_0 = const()[name = string("h_143_groups_0"), val = int32(1)]; + tensor h_143_cast_fp16 = conv(dilations = h_143_dilations_0, groups = h_143_groups_0, pad = h_143_pad_0, pad_type = h_143_pad_type_0, strides = h_143_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_771_cast_fp16)[name = string("h_143_cast_fp16")]; + tensor x_547_cast_fp16 = add(x = x_545_cast_fp16, y = h_143_cast_fp16)[name = string("x_547_cast_fp16")]; + tensor key_cache_145_begin_0 = const()[name = string("key_cache_145_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_145_end_0 = const()[name = string("key_cache_145_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_145_end_mask_0 = const()[name = string("key_cache_145_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_145_cast_fp16 = slice_by_index(begin = key_cache_145_begin_0, end = key_cache_145_end_0, end_mask = key_cache_145_end_mask_0, x = layer_key_caches_29_cast_fp16)[name = string("key_cache_145_cast_fp16")]; + tensor value_cache_145_begin_0 = const()[name = string("value_cache_145_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_145_end_0 = const()[name = string("value_cache_145_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_145_end_mask_0 = const()[name = string("value_cache_145_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_145_cast_fp16 = slice_by_index(begin = value_cache_145_begin_0, end = value_cache_145_end_0, end_mask = value_cache_145_end_mask_0, x = layer_value_caches_29_cast_fp16)[name = string("value_cache_145_cast_fp16")]; + int32 var_18053 = const()[name = string("op_18053"), val = int32(2)]; + int32 var_18057 = const()[name = string("op_18057"), val = int32(3)]; + tensor var_18072_cast_fp16 = mul(x = x_547_cast_fp16, y = x_547_cast_fp16)[name = string("op_18072_cast_fp16")]; + tensor variance_603_axes_0 = const()[name = string("variance_603_axes_0"), val = tensor([1])]; + bool variance_603_keep_dims_0 = const()[name = string("variance_603_keep_dims_0"), val = bool(true)]; + tensor variance_603_cast_fp16 = reduce_mean(axes = variance_603_axes_0, keep_dims = variance_603_keep_dims_0, x = var_18072_cast_fp16)[name = string("variance_603_cast_fp16")]; + fp16 var_18075_to_fp16 = const()[name = string("op_18075_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18076_cast_fp16 = add(x = variance_603_cast_fp16, y = var_18075_to_fp16)[name = string("op_18076_cast_fp16")]; + fp32 var_18077_epsilon_0 = const()[name = string("op_18077_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18077_cast_fp16 = rsqrt(epsilon = var_18077_epsilon_0, x = var_18076_cast_fp16)[name = string("op_18077_cast_fp16")]; + tensor var_18078_cast_fp16 = mul(x = x_547_cast_fp16, y = var_18077_cast_fp16)[name = string("op_18078_cast_fp16")]; + tensor input_773_cast_fp16 = mul(x = var_18078_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_773_cast_fp16")]; + string q_433_pad_type_0 = const()[name = string("q_433_pad_type_0"), val = string("valid")]; + tensor q_433_strides_0 = const()[name = string("q_433_strides_0"), val = tensor([1, 1])]; + tensor q_433_pad_0 = const()[name = string("q_433_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_433_dilations_0 = const()[name = string("q_433_dilations_0"), val = tensor([1, 1])]; + int32 q_433_groups_0 = const()[name = string("q_433_groups_0"), val = int32(1)]; + tensor q_433_cast_fp16 = conv(dilations = q_433_dilations_0, groups = q_433_groups_0, pad = q_433_pad_0, pad_type = q_433_pad_type_0, strides = q_433_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_773_cast_fp16)[name = string("q_433_cast_fp16")]; + string k_433_pad_type_0 = const()[name = string("k_433_pad_type_0"), val = string("valid")]; + tensor k_433_strides_0 = const()[name = string("k_433_strides_0"), val = tensor([1, 1])]; + tensor k_433_pad_0 = const()[name = string("k_433_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_433_dilations_0 = const()[name = string("k_433_dilations_0"), val = tensor([1, 1])]; + int32 k_433_groups_0 = const()[name = string("k_433_groups_0"), val = int32(1)]; + tensor k_433_cast_fp16 = conv(dilations = k_433_dilations_0, groups = k_433_groups_0, pad = k_433_pad_0, pad_type = k_433_pad_type_0, strides = k_433_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_773_cast_fp16)[name = string("k_433_cast_fp16")]; + string v_145_pad_type_0 = const()[name = string("v_145_pad_type_0"), val = string("valid")]; + tensor v_145_strides_0 = const()[name = string("v_145_strides_0"), val = tensor([1, 1])]; + tensor v_145_pad_0 = const()[name = string("v_145_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_145_dilations_0 = const()[name = string("v_145_dilations_0"), val = tensor([1, 1])]; + int32 v_145_groups_0 = const()[name = string("v_145_groups_0"), val = int32(1)]; + tensor v_145_cast_fp16 = conv(dilations = v_145_dilations_0, groups = v_145_groups_0, pad = v_145_pad_0, pad_type = v_145_pad_type_0, strides = v_145_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_773_cast_fp16)[name = string("v_145_cast_fp16")]; + tensor var_18112 = const()[name = string("op_18112"), val = tensor([16, 128, 1, 1])]; + tensor x_549_cast_fp16 = reshape(shape = var_18112, x = q_433_cast_fp16)[name = string("x_549_cast_fp16")]; + tensor var_18115_cast_fp16 = mul(x = x_549_cast_fp16, y = x_549_cast_fp16)[name = string("op_18115_cast_fp16")]; + tensor variance_605_axes_0 = const()[name = string("variance_605_axes_0"), val = tensor([1])]; + bool variance_605_keep_dims_0 = const()[name = string("variance_605_keep_dims_0"), val = bool(true)]; + tensor variance_605_cast_fp16 = reduce_mean(axes = variance_605_axes_0, keep_dims = variance_605_keep_dims_0, x = var_18115_cast_fp16)[name = string("variance_605_cast_fp16")]; + fp16 var_18118_to_fp16 = const()[name = string("op_18118_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18119_cast_fp16 = add(x = variance_605_cast_fp16, y = var_18118_to_fp16)[name = string("op_18119_cast_fp16")]; + fp32 var_18120_epsilon_0 = const()[name = string("op_18120_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18120_cast_fp16 = rsqrt(epsilon = var_18120_epsilon_0, x = var_18119_cast_fp16)[name = string("op_18120_cast_fp16")]; + tensor var_18121_cast_fp16 = mul(x = x_549_cast_fp16, y = var_18120_cast_fp16)[name = string("op_18121_cast_fp16")]; + tensor q_435_cast_fp16 = mul(x = var_18121_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_435_cast_fp16")]; + tensor var_18123 = const()[name = string("op_18123"), val = tensor([8, 128, 1, 1])]; + tensor x_551_cast_fp16 = reshape(shape = var_18123, x = k_433_cast_fp16)[name = string("x_551_cast_fp16")]; + tensor var_18126_cast_fp16 = mul(x = x_551_cast_fp16, y = x_551_cast_fp16)[name = string("op_18126_cast_fp16")]; + tensor variance_607_axes_0 = const()[name = string("variance_607_axes_0"), val = tensor([1])]; + bool variance_607_keep_dims_0 = const()[name = string("variance_607_keep_dims_0"), val = bool(true)]; + tensor variance_607_cast_fp16 = reduce_mean(axes = variance_607_axes_0, keep_dims = variance_607_keep_dims_0, x = var_18126_cast_fp16)[name = string("variance_607_cast_fp16")]; + fp16 var_18129_to_fp16 = const()[name = string("op_18129_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18130_cast_fp16 = add(x = variance_607_cast_fp16, y = var_18129_to_fp16)[name = string("op_18130_cast_fp16")]; + fp32 var_18131_epsilon_0 = const()[name = string("op_18131_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18131_cast_fp16 = rsqrt(epsilon = var_18131_epsilon_0, x = var_18130_cast_fp16)[name = string("op_18131_cast_fp16")]; + tensor var_18132_cast_fp16 = mul(x = x_551_cast_fp16, y = var_18131_cast_fp16)[name = string("op_18132_cast_fp16")]; + tensor k_435_cast_fp16 = mul(x = var_18132_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_435_cast_fp16")]; + tensor var_18134 = const()[name = string("op_18134"), val = tensor([1, 16, 128, 1])]; + tensor z_289_cast_fp16 = reshape(shape = var_18134, x = q_435_cast_fp16)[name = string("z_289_cast_fp16")]; + tensor var_18136 = const()[name = string("op_18136"), val = tensor([1, 8, 128, 1])]; + tensor z_291_cast_fp16 = reshape(shape = var_18136, x = k_435_cast_fp16)[name = string("z_291_cast_fp16")]; + tensor z1_289_begin_0 = const()[name = string("z1_289_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_289_end_0 = const()[name = string("z1_289_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_289_end_mask_0 = const()[name = string("z1_289_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_289_cast_fp16 = slice_by_index(begin = z1_289_begin_0, end = z1_289_end_0, end_mask = z1_289_end_mask_0, x = z_289_cast_fp16)[name = string("z1_289_cast_fp16")]; + tensor z2_289_begin_0 = const()[name = string("z2_289_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_289_end_0 = const()[name = string("z2_289_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_289_end_mask_0 = const()[name = string("z2_289_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_289_cast_fp16 = slice_by_index(begin = z2_289_begin_0, end = z2_289_end_0, end_mask = z2_289_end_mask_0, x = z_289_cast_fp16)[name = string("z2_289_cast_fp16")]; + tensor var_18144_cast_fp16 = mul(x = z_289_cast_fp16, y = cos_141_to_fp16)[name = string("op_18144_cast_fp16")]; + fp16 const_159_promoted_to_fp16 = const()[name = string("const_159_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18145_cast_fp16 = mul(x = z2_289_cast_fp16, y = const_159_promoted_to_fp16)[name = string("op_18145_cast_fp16")]; + bool var_18147_interleave_0 = const()[name = string("op_18147_interleave_0"), val = bool(false)]; + tensor var_18147_cast_fp16 = concat(axis = var_18053, interleave = var_18147_interleave_0, values = (var_18145_cast_fp16, z1_289_cast_fp16))[name = string("op_18147_cast_fp16")]; + tensor var_18148_cast_fp16 = mul(x = var_18147_cast_fp16, y = sin_141_to_fp16)[name = string("op_18148_cast_fp16")]; + tensor q_437_cast_fp16 = add(x = var_18144_cast_fp16, y = var_18148_cast_fp16)[name = string("q_437_cast_fp16")]; + tensor z1_291_begin_0 = const()[name = string("z1_291_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_291_end_0 = const()[name = string("z1_291_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_291_end_mask_0 = const()[name = string("z1_291_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_291_cast_fp16 = slice_by_index(begin = z1_291_begin_0, end = z1_291_end_0, end_mask = z1_291_end_mask_0, x = z_291_cast_fp16)[name = string("z1_291_cast_fp16")]; + tensor z2_291_begin_0 = const()[name = string("z2_291_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_291_end_0 = const()[name = string("z2_291_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_291_end_mask_0 = const()[name = string("z2_291_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_291_cast_fp16 = slice_by_index(begin = z2_291_begin_0, end = z2_291_end_0, end_mask = z2_291_end_mask_0, x = z_291_cast_fp16)[name = string("z2_291_cast_fp16")]; + tensor var_18156_cast_fp16 = mul(x = z_291_cast_fp16, y = cos_141_to_fp16)[name = string("op_18156_cast_fp16")]; + fp16 const_160_promoted_to_fp16 = const()[name = string("const_160_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18157_cast_fp16 = mul(x = z2_291_cast_fp16, y = const_160_promoted_to_fp16)[name = string("op_18157_cast_fp16")]; + bool var_18159_interleave_0 = const()[name = string("op_18159_interleave_0"), val = bool(false)]; + tensor var_18159_cast_fp16 = concat(axis = var_18053, interleave = var_18159_interleave_0, values = (var_18157_cast_fp16, z1_291_cast_fp16))[name = string("op_18159_cast_fp16")]; + tensor var_18160_cast_fp16 = mul(x = var_18159_cast_fp16, y = sin_141_to_fp16)[name = string("op_18160_cast_fp16")]; + tensor k_437_cast_fp16 = add(x = var_18156_cast_fp16, y = var_18160_cast_fp16)[name = string("k_437_cast_fp16")]; + tensor var_18162 = const()[name = string("op_18162"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_145_cast_fp16 = reshape(shape = var_18162, x = k_437_cast_fp16)[name = string("cur_key_145_cast_fp16")]; + tensor var_18164_to_fp16 = const()[name = string("op_18164_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646336)))]; + tensor var_18165_cast_fp16 = mul(x = key_cache_145_cast_fp16, y = var_18164_to_fp16)[name = string("op_18165_cast_fp16")]; + tensor upd_145_to_fp16 = const()[name = string("upd_145_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646464)))]; + tensor var_18166_cast_fp16 = mul(x = cur_key_145_cast_fp16, y = upd_145_to_fp16)[name = string("op_18166_cast_fp16")]; + tensor key_145_cast_fp16 = add(x = var_18165_cast_fp16, y = var_18166_cast_fp16)[name = string("key_145_cast_fp16")]; + tensor var_18168_to_fp16 = const()[name = string("op_18168_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646336)))]; + tensor var_18169_cast_fp16 = mul(x = value_cache_145_cast_fp16, y = var_18168_to_fp16)[name = string("op_18169_cast_fp16")]; + tensor var_18170_cast_fp16 = mul(x = v_145_cast_fp16, y = upd_145_to_fp16)[name = string("op_18170_cast_fp16")]; + tensor value_145_cast_fp16 = add(x = var_18169_cast_fp16, y = var_18170_cast_fp16)[name = string("value_145_cast_fp16")]; + tensor var_18172 = const()[name = string("op_18172"), val = tensor([1, 8, 128, 16])]; + tensor kh_289_cast_fp16 = reshape(shape = var_18172, x = key_145_cast_fp16)[name = string("kh_289_cast_fp16")]; + tensor var_18174 = const()[name = string("op_18174"), val = tensor([1, 8, 128, 16])]; + tensor vh_289_cast_fp16 = reshape(shape = var_18174, x = value_145_cast_fp16)[name = string("vh_289_cast_fp16")]; + tensor transpose_288_perm_0 = const()[name = string("transpose_288_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_144_reps_0 = const()[name = string("tile_144_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_288_cast_fp16 = transpose(perm = transpose_288_perm_0, x = kh_289_cast_fp16)[name = string("transpose_47")]; + tensor tile_144_cast_fp16 = tile(reps = tile_144_reps_0, x = transpose_288_cast_fp16)[name = string("tile_144_cast_fp16")]; + tensor concat_361 = const()[name = string("concat_361"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_288_cast_fp16 = reshape(shape = concat_361, x = tile_144_cast_fp16)[name = string("reshape_288_cast_fp16")]; + tensor transpose_289_perm_0 = const()[name = string("transpose_289_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_362 = const()[name = string("concat_362"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_289_cast_fp16 = transpose(perm = transpose_289_perm_0, x = reshape_288_cast_fp16)[name = string("transpose_46")]; + tensor reshape_289_cast_fp16 = reshape(shape = concat_362, x = transpose_289_cast_fp16)[name = string("reshape_289_cast_fp16")]; + tensor transpose_290_perm_0 = const()[name = string("transpose_290_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_145_reps_0 = const()[name = string("tile_145_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_290_cast_fp16 = transpose(perm = transpose_290_perm_0, x = vh_289_cast_fp16)[name = string("transpose_45")]; + tensor tile_145_cast_fp16 = tile(reps = tile_145_reps_0, x = transpose_290_cast_fp16)[name = string("tile_145_cast_fp16")]; + tensor concat_363 = const()[name = string("concat_363"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_290_cast_fp16 = reshape(shape = concat_363, x = tile_145_cast_fp16)[name = string("reshape_290_cast_fp16")]; + tensor transpose_291_perm_0 = const()[name = string("transpose_291_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_364 = const()[name = string("concat_364"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_291_cast_fp16 = transpose(perm = transpose_291_perm_0, x = reshape_290_cast_fp16)[name = string("transpose_44")]; + tensor reshape_291_cast_fp16 = reshape(shape = concat_364, x = transpose_291_cast_fp16)[name = string("reshape_291_cast_fp16")]; + fp16 var_18178_to_fp16 = const()[name = string("op_18178_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_18179_cast_fp16 = mul(x = q_437_cast_fp16, y = var_18178_to_fp16)[name = string("op_18179_cast_fp16")]; + tensor transpose_605_perm_0 = const()[name = string("transpose_605_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_315_transpose_x_1 = const()[name = string("w_315_transpose_x_1"), val = bool(true)]; + bool w_315_transpose_y_1 = const()[name = string("w_315_transpose_y_1"), val = bool(false)]; + tensor transpose_605_cast_fp16 = transpose(perm = transpose_605_perm_0, x = reshape_289_cast_fp16)[name = string("transpose_43")]; + tensor w_315_cast_fp16 = matmul(transpose_x = w_315_transpose_x_1, transpose_y = w_315_transpose_y_1, x = var_18179_cast_fp16, y = transpose_605_cast_fp16)[name = string("w_315_cast_fp16")]; + tensor pad_145_to_fp16 = const()[name = string("pad_145_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646592)))]; + tensor var_18182_cast_fp16 = add(x = w_315_cast_fp16, y = pad_145_to_fp16)[name = string("op_18182_cast_fp16")]; + tensor w_317_cast_fp16 = softmax(axis = var_18057, x = var_18182_cast_fp16)[name = string("w_317_cast_fp16")]; + tensor transpose_606_perm_0 = const()[name = string("transpose_606_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_145_transpose_x_1 = const()[name = string("attn_145_transpose_x_1"), val = bool(false)]; + bool attn_145_transpose_y_1 = const()[name = string("attn_145_transpose_y_1"), val = bool(true)]; + tensor transpose_606_cast_fp16 = transpose(perm = transpose_606_perm_0, x = reshape_291_cast_fp16)[name = string("transpose_42")]; + tensor attn_145_cast_fp16 = matmul(transpose_x = attn_145_transpose_x_1, transpose_y = attn_145_transpose_y_1, x = transpose_606_cast_fp16, y = w_317_cast_fp16)[name = string("attn_145_cast_fp16")]; + tensor var_18186 = const()[name = string("op_18186"), val = tensor([1, 2048, 1, 1])]; + tensor input_775_cast_fp16 = reshape(shape = var_18186, x = attn_145_cast_fp16)[name = string("input_775_cast_fp16")]; + string attn_output_145_pad_type_0 = const()[name = string("attn_output_145_pad_type_0"), val = string("valid")]; + tensor attn_output_145_strides_0 = const()[name = string("attn_output_145_strides_0"), val = tensor([1, 1])]; + tensor attn_output_145_pad_0 = const()[name = string("attn_output_145_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_145_dilations_0 = const()[name = string("attn_output_145_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_145_groups_0 = const()[name = string("attn_output_145_groups_0"), val = int32(1)]; + tensor attn_output_145_cast_fp16 = conv(dilations = attn_output_145_dilations_0, groups = attn_output_145_groups_0, pad = attn_output_145_pad_0, pad_type = attn_output_145_pad_type_0, strides = attn_output_145_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_775_cast_fp16)[name = string("attn_output_145_cast_fp16")]; + tensor x_553_cast_fp16 = add(x = x_547_cast_fp16, y = attn_output_145_cast_fp16)[name = string("x_553_cast_fp16")]; + tensor var_18200_cast_fp16 = mul(x = x_553_cast_fp16, y = x_553_cast_fp16)[name = string("op_18200_cast_fp16")]; + tensor variance_609_axes_0 = const()[name = string("variance_609_axes_0"), val = tensor([1])]; + bool variance_609_keep_dims_0 = const()[name = string("variance_609_keep_dims_0"), val = bool(true)]; + tensor variance_609_cast_fp16 = reduce_mean(axes = variance_609_axes_0, keep_dims = variance_609_keep_dims_0, x = var_18200_cast_fp16)[name = string("variance_609_cast_fp16")]; + fp16 var_18203_to_fp16 = const()[name = string("op_18203_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18204_cast_fp16 = add(x = variance_609_cast_fp16, y = var_18203_to_fp16)[name = string("op_18204_cast_fp16")]; + fp32 var_18205_epsilon_0 = const()[name = string("op_18205_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18205_cast_fp16 = rsqrt(epsilon = var_18205_epsilon_0, x = var_18204_cast_fp16)[name = string("op_18205_cast_fp16")]; + tensor var_18206_cast_fp16 = mul(x = x_553_cast_fp16, y = var_18205_cast_fp16)[name = string("op_18206_cast_fp16")]; + tensor input_777_cast_fp16 = mul(x = var_18206_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_777_cast_fp16")]; + string input_779_pad_type_0 = const()[name = string("input_779_pad_type_0"), val = string("valid")]; + tensor input_779_strides_0 = const()[name = string("input_779_strides_0"), val = tensor([1, 1])]; + tensor input_779_pad_0 = const()[name = string("input_779_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_779_dilations_0 = const()[name = string("input_779_dilations_0"), val = tensor([1, 1])]; + int32 input_779_groups_0 = const()[name = string("input_779_groups_0"), val = int32(1)]; + tensor input_779_cast_fp16 = conv(dilations = input_779_dilations_0, groups = input_779_groups_0, pad = input_779_pad_0, pad_type = input_779_pad_type_0, strides = input_779_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_777_cast_fp16)[name = string("input_779_cast_fp16")]; + tensor var_18214_cast_fp16 = silu(x = input_779_cast_fp16)[name = string("op_18214_cast_fp16")]; + string var_18220_pad_type_0 = const()[name = string("op_18220_pad_type_0"), val = string("valid")]; + tensor var_18220_strides_0 = const()[name = string("op_18220_strides_0"), val = tensor([1, 1])]; + tensor var_18220_pad_0 = const()[name = string("op_18220_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18220_dilations_0 = const()[name = string("op_18220_dilations_0"), val = tensor([1, 1])]; + int32 var_18220_groups_0 = const()[name = string("op_18220_groups_0"), val = int32(1)]; + tensor var_18220_cast_fp16 = conv(dilations = var_18220_dilations_0, groups = var_18220_groups_0, pad = var_18220_pad_0, pad_type = var_18220_pad_type_0, strides = var_18220_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_777_cast_fp16)[name = string("op_18220_cast_fp16")]; + tensor input_781_cast_fp16 = mul(x = var_18214_cast_fp16, y = var_18220_cast_fp16)[name = string("input_781_cast_fp16")]; + string h_145_pad_type_0 = const()[name = string("h_145_pad_type_0"), val = string("valid")]; + tensor h_145_strides_0 = const()[name = string("h_145_strides_0"), val = tensor([1, 1])]; + tensor h_145_pad_0 = const()[name = string("h_145_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_145_dilations_0 = const()[name = string("h_145_dilations_0"), val = tensor([1, 1])]; + int32 h_145_groups_0 = const()[name = string("h_145_groups_0"), val = int32(1)]; + tensor h_145_cast_fp16 = conv(dilations = h_145_dilations_0, groups = h_145_groups_0, pad = h_145_pad_0, pad_type = h_145_pad_type_0, strides = h_145_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_781_cast_fp16)[name = string("h_145_cast_fp16")]; + tensor x_555_cast_fp16 = add(x = x_553_cast_fp16, y = h_145_cast_fp16)[name = string("x_555_cast_fp16")]; + tensor key_cache_147_begin_0 = const()[name = string("key_cache_147_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_147_end_0 = const()[name = string("key_cache_147_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_147_end_mask_0 = const()[name = string("key_cache_147_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_147_cast_fp16 = slice_by_index(begin = key_cache_147_begin_0, end = key_cache_147_end_0, end_mask = key_cache_147_end_mask_0, x = layer_key_caches_29_cast_fp16)[name = string("key_cache_147_cast_fp16")]; + tensor value_cache_147_begin_0 = const()[name = string("value_cache_147_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_147_end_0 = const()[name = string("value_cache_147_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_147_end_mask_0 = const()[name = string("value_cache_147_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_147_cast_fp16 = slice_by_index(begin = value_cache_147_begin_0, end = value_cache_147_end_0, end_mask = value_cache_147_end_mask_0, x = layer_value_caches_29_cast_fp16)[name = string("value_cache_147_cast_fp16")]; + int32 var_18273 = const()[name = string("op_18273"), val = int32(2)]; + int32 var_18277 = const()[name = string("op_18277"), val = int32(3)]; + tensor var_18292_cast_fp16 = mul(x = x_555_cast_fp16, y = x_555_cast_fp16)[name = string("op_18292_cast_fp16")]; + tensor variance_611_axes_0 = const()[name = string("variance_611_axes_0"), val = tensor([1])]; + bool variance_611_keep_dims_0 = const()[name = string("variance_611_keep_dims_0"), val = bool(true)]; + tensor variance_611_cast_fp16 = reduce_mean(axes = variance_611_axes_0, keep_dims = variance_611_keep_dims_0, x = var_18292_cast_fp16)[name = string("variance_611_cast_fp16")]; + fp16 var_18295_to_fp16 = const()[name = string("op_18295_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18296_cast_fp16 = add(x = variance_611_cast_fp16, y = var_18295_to_fp16)[name = string("op_18296_cast_fp16")]; + fp32 var_18297_epsilon_0 = const()[name = string("op_18297_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18297_cast_fp16 = rsqrt(epsilon = var_18297_epsilon_0, x = var_18296_cast_fp16)[name = string("op_18297_cast_fp16")]; + tensor var_18298_cast_fp16 = mul(x = x_555_cast_fp16, y = var_18297_cast_fp16)[name = string("op_18298_cast_fp16")]; + tensor input_783_cast_fp16 = mul(x = var_18298_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_783_cast_fp16")]; + string q_439_pad_type_0 = const()[name = string("q_439_pad_type_0"), val = string("valid")]; + tensor q_439_strides_0 = const()[name = string("q_439_strides_0"), val = tensor([1, 1])]; + tensor q_439_pad_0 = const()[name = string("q_439_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_439_dilations_0 = const()[name = string("q_439_dilations_0"), val = tensor([1, 1])]; + int32 q_439_groups_0 = const()[name = string("q_439_groups_0"), val = int32(1)]; + tensor q_439_cast_fp16 = conv(dilations = q_439_dilations_0, groups = q_439_groups_0, pad = q_439_pad_0, pad_type = q_439_pad_type_0, strides = q_439_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_783_cast_fp16)[name = string("q_439_cast_fp16")]; + string k_439_pad_type_0 = const()[name = string("k_439_pad_type_0"), val = string("valid")]; + tensor k_439_strides_0 = const()[name = string("k_439_strides_0"), val = tensor([1, 1])]; + tensor k_439_pad_0 = const()[name = string("k_439_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_439_dilations_0 = const()[name = string("k_439_dilations_0"), val = tensor([1, 1])]; + int32 k_439_groups_0 = const()[name = string("k_439_groups_0"), val = int32(1)]; + tensor k_439_cast_fp16 = conv(dilations = k_439_dilations_0, groups = k_439_groups_0, pad = k_439_pad_0, pad_type = k_439_pad_type_0, strides = k_439_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_783_cast_fp16)[name = string("k_439_cast_fp16")]; + string v_147_pad_type_0 = const()[name = string("v_147_pad_type_0"), val = string("valid")]; + tensor v_147_strides_0 = const()[name = string("v_147_strides_0"), val = tensor([1, 1])]; + tensor v_147_pad_0 = const()[name = string("v_147_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_147_dilations_0 = const()[name = string("v_147_dilations_0"), val = tensor([1, 1])]; + int32 v_147_groups_0 = const()[name = string("v_147_groups_0"), val = int32(1)]; + tensor v_147_cast_fp16 = conv(dilations = v_147_dilations_0, groups = v_147_groups_0, pad = v_147_pad_0, pad_type = v_147_pad_type_0, strides = v_147_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_783_cast_fp16)[name = string("v_147_cast_fp16")]; + tensor var_18332 = const()[name = string("op_18332"), val = tensor([16, 128, 1, 1])]; + tensor x_557_cast_fp16 = reshape(shape = var_18332, x = q_439_cast_fp16)[name = string("x_557_cast_fp16")]; + tensor var_18335_cast_fp16 = mul(x = x_557_cast_fp16, y = x_557_cast_fp16)[name = string("op_18335_cast_fp16")]; + tensor variance_613_axes_0 = const()[name = string("variance_613_axes_0"), val = tensor([1])]; + bool variance_613_keep_dims_0 = const()[name = string("variance_613_keep_dims_0"), val = bool(true)]; + tensor variance_613_cast_fp16 = reduce_mean(axes = variance_613_axes_0, keep_dims = variance_613_keep_dims_0, x = var_18335_cast_fp16)[name = string("variance_613_cast_fp16")]; + fp16 var_18338_to_fp16 = const()[name = string("op_18338_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18339_cast_fp16 = add(x = variance_613_cast_fp16, y = var_18338_to_fp16)[name = string("op_18339_cast_fp16")]; + fp32 var_18340_epsilon_0 = const()[name = string("op_18340_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18340_cast_fp16 = rsqrt(epsilon = var_18340_epsilon_0, x = var_18339_cast_fp16)[name = string("op_18340_cast_fp16")]; + tensor var_18341_cast_fp16 = mul(x = x_557_cast_fp16, y = var_18340_cast_fp16)[name = string("op_18341_cast_fp16")]; + tensor q_441_cast_fp16 = mul(x = var_18341_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_441_cast_fp16")]; + tensor var_18343 = const()[name = string("op_18343"), val = tensor([8, 128, 1, 1])]; + tensor x_559_cast_fp16 = reshape(shape = var_18343, x = k_439_cast_fp16)[name = string("x_559_cast_fp16")]; + tensor var_18346_cast_fp16 = mul(x = x_559_cast_fp16, y = x_559_cast_fp16)[name = string("op_18346_cast_fp16")]; + tensor variance_615_axes_0 = const()[name = string("variance_615_axes_0"), val = tensor([1])]; + bool variance_615_keep_dims_0 = const()[name = string("variance_615_keep_dims_0"), val = bool(true)]; + tensor variance_615_cast_fp16 = reduce_mean(axes = variance_615_axes_0, keep_dims = variance_615_keep_dims_0, x = var_18346_cast_fp16)[name = string("variance_615_cast_fp16")]; + fp16 var_18349_to_fp16 = const()[name = string("op_18349_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18350_cast_fp16 = add(x = variance_615_cast_fp16, y = var_18349_to_fp16)[name = string("op_18350_cast_fp16")]; + fp32 var_18351_epsilon_0 = const()[name = string("op_18351_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18351_cast_fp16 = rsqrt(epsilon = var_18351_epsilon_0, x = var_18350_cast_fp16)[name = string("op_18351_cast_fp16")]; + tensor var_18352_cast_fp16 = mul(x = x_559_cast_fp16, y = var_18351_cast_fp16)[name = string("op_18352_cast_fp16")]; + tensor k_441_cast_fp16 = mul(x = var_18352_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_441_cast_fp16")]; + tensor var_18354 = const()[name = string("op_18354"), val = tensor([1, 16, 128, 1])]; + tensor z_293_cast_fp16 = reshape(shape = var_18354, x = q_441_cast_fp16)[name = string("z_293_cast_fp16")]; + tensor var_18356 = const()[name = string("op_18356"), val = tensor([1, 8, 128, 1])]; + tensor z_295_cast_fp16 = reshape(shape = var_18356, x = k_441_cast_fp16)[name = string("z_295_cast_fp16")]; + tensor z1_293_begin_0 = const()[name = string("z1_293_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_293_end_0 = const()[name = string("z1_293_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_293_end_mask_0 = const()[name = string("z1_293_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_293_cast_fp16 = slice_by_index(begin = z1_293_begin_0, end = z1_293_end_0, end_mask = z1_293_end_mask_0, x = z_293_cast_fp16)[name = string("z1_293_cast_fp16")]; + tensor z2_293_begin_0 = const()[name = string("z2_293_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_293_end_0 = const()[name = string("z2_293_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_293_end_mask_0 = const()[name = string("z2_293_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_293_cast_fp16 = slice_by_index(begin = z2_293_begin_0, end = z2_293_end_0, end_mask = z2_293_end_mask_0, x = z_293_cast_fp16)[name = string("z2_293_cast_fp16")]; + tensor var_18364_cast_fp16 = mul(x = z_293_cast_fp16, y = cos_141_to_fp16)[name = string("op_18364_cast_fp16")]; + fp16 const_161_promoted_to_fp16 = const()[name = string("const_161_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18365_cast_fp16 = mul(x = z2_293_cast_fp16, y = const_161_promoted_to_fp16)[name = string("op_18365_cast_fp16")]; + bool var_18367_interleave_0 = const()[name = string("op_18367_interleave_0"), val = bool(false)]; + tensor var_18367_cast_fp16 = concat(axis = var_18273, interleave = var_18367_interleave_0, values = (var_18365_cast_fp16, z1_293_cast_fp16))[name = string("op_18367_cast_fp16")]; + tensor var_18368_cast_fp16 = mul(x = var_18367_cast_fp16, y = sin_141_to_fp16)[name = string("op_18368_cast_fp16")]; + tensor q_443_cast_fp16 = add(x = var_18364_cast_fp16, y = var_18368_cast_fp16)[name = string("q_443_cast_fp16")]; + tensor z1_295_begin_0 = const()[name = string("z1_295_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_295_end_0 = const()[name = string("z1_295_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_295_end_mask_0 = const()[name = string("z1_295_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_295_cast_fp16 = slice_by_index(begin = z1_295_begin_0, end = z1_295_end_0, end_mask = z1_295_end_mask_0, x = z_295_cast_fp16)[name = string("z1_295_cast_fp16")]; + tensor z2_295_begin_0 = const()[name = string("z2_295_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_295_end_0 = const()[name = string("z2_295_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_295_end_mask_0 = const()[name = string("z2_295_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_295_cast_fp16 = slice_by_index(begin = z2_295_begin_0, end = z2_295_end_0, end_mask = z2_295_end_mask_0, x = z_295_cast_fp16)[name = string("z2_295_cast_fp16")]; + tensor var_18376_cast_fp16 = mul(x = z_295_cast_fp16, y = cos_141_to_fp16)[name = string("op_18376_cast_fp16")]; + fp16 const_162_promoted_to_fp16 = const()[name = string("const_162_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18377_cast_fp16 = mul(x = z2_295_cast_fp16, y = const_162_promoted_to_fp16)[name = string("op_18377_cast_fp16")]; + bool var_18379_interleave_0 = const()[name = string("op_18379_interleave_0"), val = bool(false)]; + tensor var_18379_cast_fp16 = concat(axis = var_18273, interleave = var_18379_interleave_0, values = (var_18377_cast_fp16, z1_295_cast_fp16))[name = string("op_18379_cast_fp16")]; + tensor var_18380_cast_fp16 = mul(x = var_18379_cast_fp16, y = sin_141_to_fp16)[name = string("op_18380_cast_fp16")]; + tensor k_443_cast_fp16 = add(x = var_18376_cast_fp16, y = var_18380_cast_fp16)[name = string("k_443_cast_fp16")]; + tensor var_18382 = const()[name = string("op_18382"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_147_cast_fp16 = reshape(shape = var_18382, x = k_443_cast_fp16)[name = string("cur_key_147_cast_fp16")]; + tensor var_18384_to_fp16 = const()[name = string("op_18384_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646336)))]; + tensor var_18385_cast_fp16 = mul(x = key_cache_147_cast_fp16, y = var_18384_to_fp16)[name = string("op_18385_cast_fp16")]; + tensor upd_147_to_fp16 = const()[name = string("upd_147_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646464)))]; + tensor var_18386_cast_fp16 = mul(x = cur_key_147_cast_fp16, y = upd_147_to_fp16)[name = string("op_18386_cast_fp16")]; + tensor key_147_cast_fp16 = add(x = var_18385_cast_fp16, y = var_18386_cast_fp16)[name = string("key_147_cast_fp16")]; + tensor var_18388_to_fp16 = const()[name = string("op_18388_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646336)))]; + tensor var_18389_cast_fp16 = mul(x = value_cache_147_cast_fp16, y = var_18388_to_fp16)[name = string("op_18389_cast_fp16")]; + tensor var_18390_cast_fp16 = mul(x = v_147_cast_fp16, y = upd_147_to_fp16)[name = string("op_18390_cast_fp16")]; + tensor value_147_cast_fp16 = add(x = var_18389_cast_fp16, y = var_18390_cast_fp16)[name = string("value_147_cast_fp16")]; + tensor var_18392 = const()[name = string("op_18392"), val = tensor([1, 8, 128, 16])]; + tensor kh_293_cast_fp16 = reshape(shape = var_18392, x = key_147_cast_fp16)[name = string("kh_293_cast_fp16")]; + tensor var_18394 = const()[name = string("op_18394"), val = tensor([1, 8, 128, 16])]; + tensor vh_293_cast_fp16 = reshape(shape = var_18394, x = value_147_cast_fp16)[name = string("vh_293_cast_fp16")]; + tensor transpose_292_perm_0 = const()[name = string("transpose_292_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_146_reps_0 = const()[name = string("tile_146_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_292_cast_fp16 = transpose(perm = transpose_292_perm_0, x = kh_293_cast_fp16)[name = string("transpose_41")]; + tensor tile_146_cast_fp16 = tile(reps = tile_146_reps_0, x = transpose_292_cast_fp16)[name = string("tile_146_cast_fp16")]; + tensor concat_365 = const()[name = string("concat_365"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_292_cast_fp16 = reshape(shape = concat_365, x = tile_146_cast_fp16)[name = string("reshape_292_cast_fp16")]; + tensor transpose_293_perm_0 = const()[name = string("transpose_293_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_366 = const()[name = string("concat_366"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_293_cast_fp16 = transpose(perm = transpose_293_perm_0, x = reshape_292_cast_fp16)[name = string("transpose_40")]; + tensor reshape_293_cast_fp16 = reshape(shape = concat_366, x = transpose_293_cast_fp16)[name = string("reshape_293_cast_fp16")]; + tensor transpose_294_perm_0 = const()[name = string("transpose_294_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_147_reps_0 = const()[name = string("tile_147_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_294_cast_fp16 = transpose(perm = transpose_294_perm_0, x = vh_293_cast_fp16)[name = string("transpose_39")]; + tensor tile_147_cast_fp16 = tile(reps = tile_147_reps_0, x = transpose_294_cast_fp16)[name = string("tile_147_cast_fp16")]; + tensor concat_367 = const()[name = string("concat_367"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_294_cast_fp16 = reshape(shape = concat_367, x = tile_147_cast_fp16)[name = string("reshape_294_cast_fp16")]; + tensor transpose_295_perm_0 = const()[name = string("transpose_295_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_368 = const()[name = string("concat_368"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_295_cast_fp16 = transpose(perm = transpose_295_perm_0, x = reshape_294_cast_fp16)[name = string("transpose_38")]; + tensor reshape_295_cast_fp16 = reshape(shape = concat_368, x = transpose_295_cast_fp16)[name = string("reshape_295_cast_fp16")]; + fp16 var_18398_to_fp16 = const()[name = string("op_18398_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_18399_cast_fp16 = mul(x = q_443_cast_fp16, y = var_18398_to_fp16)[name = string("op_18399_cast_fp16")]; + tensor transpose_609_perm_0 = const()[name = string("transpose_609_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_319_transpose_x_1 = const()[name = string("w_319_transpose_x_1"), val = bool(true)]; + bool w_319_transpose_y_1 = const()[name = string("w_319_transpose_y_1"), val = bool(false)]; + tensor transpose_609_cast_fp16 = transpose(perm = transpose_609_perm_0, x = reshape_293_cast_fp16)[name = string("transpose_37")]; + tensor w_319_cast_fp16 = matmul(transpose_x = w_319_transpose_x_1, transpose_y = w_319_transpose_y_1, x = var_18399_cast_fp16, y = transpose_609_cast_fp16)[name = string("w_319_cast_fp16")]; + tensor pad_147_to_fp16 = const()[name = string("pad_147_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646592)))]; + tensor var_18402_cast_fp16 = add(x = w_319_cast_fp16, y = pad_147_to_fp16)[name = string("op_18402_cast_fp16")]; + tensor w_321_cast_fp16 = softmax(axis = var_18277, x = var_18402_cast_fp16)[name = string("w_321_cast_fp16")]; + tensor transpose_610_perm_0 = const()[name = string("transpose_610_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_147_transpose_x_1 = const()[name = string("attn_147_transpose_x_1"), val = bool(false)]; + bool attn_147_transpose_y_1 = const()[name = string("attn_147_transpose_y_1"), val = bool(true)]; + tensor transpose_610_cast_fp16 = transpose(perm = transpose_610_perm_0, x = reshape_295_cast_fp16)[name = string("transpose_36")]; + tensor attn_147_cast_fp16 = matmul(transpose_x = attn_147_transpose_x_1, transpose_y = attn_147_transpose_y_1, x = transpose_610_cast_fp16, y = w_321_cast_fp16)[name = string("attn_147_cast_fp16")]; + tensor var_18406 = const()[name = string("op_18406"), val = tensor([1, 2048, 1, 1])]; + tensor input_785_cast_fp16 = reshape(shape = var_18406, x = attn_147_cast_fp16)[name = string("input_785_cast_fp16")]; + string attn_output_147_pad_type_0 = const()[name = string("attn_output_147_pad_type_0"), val = string("valid")]; + tensor attn_output_147_strides_0 = const()[name = string("attn_output_147_strides_0"), val = tensor([1, 1])]; + tensor attn_output_147_pad_0 = const()[name = string("attn_output_147_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_147_dilations_0 = const()[name = string("attn_output_147_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_147_groups_0 = const()[name = string("attn_output_147_groups_0"), val = int32(1)]; + tensor attn_output_147_cast_fp16 = conv(dilations = attn_output_147_dilations_0, groups = attn_output_147_groups_0, pad = attn_output_147_pad_0, pad_type = attn_output_147_pad_type_0, strides = attn_output_147_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_785_cast_fp16)[name = string("attn_output_147_cast_fp16")]; + tensor x_561_cast_fp16 = add(x = x_555_cast_fp16, y = attn_output_147_cast_fp16)[name = string("x_561_cast_fp16")]; + tensor var_18420_cast_fp16 = mul(x = x_561_cast_fp16, y = x_561_cast_fp16)[name = string("op_18420_cast_fp16")]; + tensor variance_617_axes_0 = const()[name = string("variance_617_axes_0"), val = tensor([1])]; + bool variance_617_keep_dims_0 = const()[name = string("variance_617_keep_dims_0"), val = bool(true)]; + tensor variance_617_cast_fp16 = reduce_mean(axes = variance_617_axes_0, keep_dims = variance_617_keep_dims_0, x = var_18420_cast_fp16)[name = string("variance_617_cast_fp16")]; + fp16 var_18423_to_fp16 = const()[name = string("op_18423_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18424_cast_fp16 = add(x = variance_617_cast_fp16, y = var_18423_to_fp16)[name = string("op_18424_cast_fp16")]; + fp32 var_18425_epsilon_0 = const()[name = string("op_18425_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18425_cast_fp16 = rsqrt(epsilon = var_18425_epsilon_0, x = var_18424_cast_fp16)[name = string("op_18425_cast_fp16")]; + tensor var_18426_cast_fp16 = mul(x = x_561_cast_fp16, y = var_18425_cast_fp16)[name = string("op_18426_cast_fp16")]; + tensor input_787_cast_fp16 = mul(x = var_18426_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_787_cast_fp16")]; + string input_789_pad_type_0 = const()[name = string("input_789_pad_type_0"), val = string("valid")]; + tensor input_789_strides_0 = const()[name = string("input_789_strides_0"), val = tensor([1, 1])]; + tensor input_789_pad_0 = const()[name = string("input_789_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_789_dilations_0 = const()[name = string("input_789_dilations_0"), val = tensor([1, 1])]; + int32 input_789_groups_0 = const()[name = string("input_789_groups_0"), val = int32(1)]; + tensor input_789_cast_fp16 = conv(dilations = input_789_dilations_0, groups = input_789_groups_0, pad = input_789_pad_0, pad_type = input_789_pad_type_0, strides = input_789_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_787_cast_fp16)[name = string("input_789_cast_fp16")]; + tensor var_18434_cast_fp16 = silu(x = input_789_cast_fp16)[name = string("op_18434_cast_fp16")]; + string var_18440_pad_type_0 = const()[name = string("op_18440_pad_type_0"), val = string("valid")]; + tensor var_18440_strides_0 = const()[name = string("op_18440_strides_0"), val = tensor([1, 1])]; + tensor var_18440_pad_0 = const()[name = string("op_18440_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18440_dilations_0 = const()[name = string("op_18440_dilations_0"), val = tensor([1, 1])]; + int32 var_18440_groups_0 = const()[name = string("op_18440_groups_0"), val = int32(1)]; + tensor var_18440_cast_fp16 = conv(dilations = var_18440_dilations_0, groups = var_18440_groups_0, pad = var_18440_pad_0, pad_type = var_18440_pad_type_0, strides = var_18440_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_787_cast_fp16)[name = string("op_18440_cast_fp16")]; + tensor input_791_cast_fp16 = mul(x = var_18434_cast_fp16, y = var_18440_cast_fp16)[name = string("input_791_cast_fp16")]; + string h_147_pad_type_0 = const()[name = string("h_147_pad_type_0"), val = string("valid")]; + tensor h_147_strides_0 = const()[name = string("h_147_strides_0"), val = tensor([1, 1])]; + tensor h_147_pad_0 = const()[name = string("h_147_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_147_dilations_0 = const()[name = string("h_147_dilations_0"), val = tensor([1, 1])]; + int32 h_147_groups_0 = const()[name = string("h_147_groups_0"), val = int32(1)]; + tensor h_147_cast_fp16 = conv(dilations = h_147_dilations_0, groups = h_147_groups_0, pad = h_147_pad_0, pad_type = h_147_pad_type_0, strides = h_147_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_791_cast_fp16)[name = string("h_147_cast_fp16")]; + tensor x_563_cast_fp16 = add(x = x_561_cast_fp16, y = h_147_cast_fp16)[name = string("x_563_cast_fp16")]; + tensor key_cache_149_begin_0 = const()[name = string("key_cache_149_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_149_end_0 = const()[name = string("key_cache_149_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_149_end_mask_0 = const()[name = string("key_cache_149_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_149_cast_fp16 = slice_by_index(begin = key_cache_149_begin_0, end = key_cache_149_end_0, end_mask = key_cache_149_end_mask_0, x = layer_key_caches_29_cast_fp16)[name = string("key_cache_149_cast_fp16")]; + tensor value_cache_149_begin_0 = const()[name = string("value_cache_149_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_149_end_0 = const()[name = string("value_cache_149_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_149_end_mask_0 = const()[name = string("value_cache_149_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_149_cast_fp16 = slice_by_index(begin = value_cache_149_begin_0, end = value_cache_149_end_0, end_mask = value_cache_149_end_mask_0, x = layer_value_caches_29_cast_fp16)[name = string("value_cache_149_cast_fp16")]; + int32 var_18493 = const()[name = string("op_18493"), val = int32(2)]; + int32 var_18497 = const()[name = string("op_18497"), val = int32(3)]; + tensor var_18512_cast_fp16 = mul(x = x_563_cast_fp16, y = x_563_cast_fp16)[name = string("op_18512_cast_fp16")]; + tensor variance_619_axes_0 = const()[name = string("variance_619_axes_0"), val = tensor([1])]; + bool variance_619_keep_dims_0 = const()[name = string("variance_619_keep_dims_0"), val = bool(true)]; + tensor variance_619_cast_fp16 = reduce_mean(axes = variance_619_axes_0, keep_dims = variance_619_keep_dims_0, x = var_18512_cast_fp16)[name = string("variance_619_cast_fp16")]; + fp16 var_18515_to_fp16 = const()[name = string("op_18515_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18516_cast_fp16 = add(x = variance_619_cast_fp16, y = var_18515_to_fp16)[name = string("op_18516_cast_fp16")]; + fp32 var_18517_epsilon_0 = const()[name = string("op_18517_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18517_cast_fp16 = rsqrt(epsilon = var_18517_epsilon_0, x = var_18516_cast_fp16)[name = string("op_18517_cast_fp16")]; + tensor var_18518_cast_fp16 = mul(x = x_563_cast_fp16, y = var_18517_cast_fp16)[name = string("op_18518_cast_fp16")]; + tensor input_793_cast_fp16 = mul(x = var_18518_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_793_cast_fp16")]; + string q_445_pad_type_0 = const()[name = string("q_445_pad_type_0"), val = string("valid")]; + tensor q_445_strides_0 = const()[name = string("q_445_strides_0"), val = tensor([1, 1])]; + tensor q_445_pad_0 = const()[name = string("q_445_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_445_dilations_0 = const()[name = string("q_445_dilations_0"), val = tensor([1, 1])]; + int32 q_445_groups_0 = const()[name = string("q_445_groups_0"), val = int32(1)]; + tensor q_445_cast_fp16 = conv(dilations = q_445_dilations_0, groups = q_445_groups_0, pad = q_445_pad_0, pad_type = q_445_pad_type_0, strides = q_445_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_793_cast_fp16)[name = string("q_445_cast_fp16")]; + string k_445_pad_type_0 = const()[name = string("k_445_pad_type_0"), val = string("valid")]; + tensor k_445_strides_0 = const()[name = string("k_445_strides_0"), val = tensor([1, 1])]; + tensor k_445_pad_0 = const()[name = string("k_445_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_445_dilations_0 = const()[name = string("k_445_dilations_0"), val = tensor([1, 1])]; + int32 k_445_groups_0 = const()[name = string("k_445_groups_0"), val = int32(1)]; + tensor k_445_cast_fp16 = conv(dilations = k_445_dilations_0, groups = k_445_groups_0, pad = k_445_pad_0, pad_type = k_445_pad_type_0, strides = k_445_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_793_cast_fp16)[name = string("k_445_cast_fp16")]; + string v_149_pad_type_0 = const()[name = string("v_149_pad_type_0"), val = string("valid")]; + tensor v_149_strides_0 = const()[name = string("v_149_strides_0"), val = tensor([1, 1])]; + tensor v_149_pad_0 = const()[name = string("v_149_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_149_dilations_0 = const()[name = string("v_149_dilations_0"), val = tensor([1, 1])]; + int32 v_149_groups_0 = const()[name = string("v_149_groups_0"), val = int32(1)]; + tensor v_149_cast_fp16 = conv(dilations = v_149_dilations_0, groups = v_149_groups_0, pad = v_149_pad_0, pad_type = v_149_pad_type_0, strides = v_149_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_793_cast_fp16)[name = string("v_149_cast_fp16")]; + tensor var_18552 = const()[name = string("op_18552"), val = tensor([16, 128, 1, 1])]; + tensor x_565_cast_fp16 = reshape(shape = var_18552, x = q_445_cast_fp16)[name = string("x_565_cast_fp16")]; + tensor var_18555_cast_fp16 = mul(x = x_565_cast_fp16, y = x_565_cast_fp16)[name = string("op_18555_cast_fp16")]; + tensor variance_621_axes_0 = const()[name = string("variance_621_axes_0"), val = tensor([1])]; + bool variance_621_keep_dims_0 = const()[name = string("variance_621_keep_dims_0"), val = bool(true)]; + tensor variance_621_cast_fp16 = reduce_mean(axes = variance_621_axes_0, keep_dims = variance_621_keep_dims_0, x = var_18555_cast_fp16)[name = string("variance_621_cast_fp16")]; + fp16 var_18558_to_fp16 = const()[name = string("op_18558_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18559_cast_fp16 = add(x = variance_621_cast_fp16, y = var_18558_to_fp16)[name = string("op_18559_cast_fp16")]; + fp32 var_18560_epsilon_0 = const()[name = string("op_18560_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18560_cast_fp16 = rsqrt(epsilon = var_18560_epsilon_0, x = var_18559_cast_fp16)[name = string("op_18560_cast_fp16")]; + tensor var_18561_cast_fp16 = mul(x = x_565_cast_fp16, y = var_18560_cast_fp16)[name = string("op_18561_cast_fp16")]; + tensor q_447_cast_fp16 = mul(x = var_18561_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_447_cast_fp16")]; + tensor var_18563 = const()[name = string("op_18563"), val = tensor([8, 128, 1, 1])]; + tensor x_567_cast_fp16 = reshape(shape = var_18563, x = k_445_cast_fp16)[name = string("x_567_cast_fp16")]; + tensor var_18566_cast_fp16 = mul(x = x_567_cast_fp16, y = x_567_cast_fp16)[name = string("op_18566_cast_fp16")]; + tensor variance_623_axes_0 = const()[name = string("variance_623_axes_0"), val = tensor([1])]; + bool variance_623_keep_dims_0 = const()[name = string("variance_623_keep_dims_0"), val = bool(true)]; + tensor variance_623_cast_fp16 = reduce_mean(axes = variance_623_axes_0, keep_dims = variance_623_keep_dims_0, x = var_18566_cast_fp16)[name = string("variance_623_cast_fp16")]; + fp16 var_18569_to_fp16 = const()[name = string("op_18569_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18570_cast_fp16 = add(x = variance_623_cast_fp16, y = var_18569_to_fp16)[name = string("op_18570_cast_fp16")]; + fp32 var_18571_epsilon_0 = const()[name = string("op_18571_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18571_cast_fp16 = rsqrt(epsilon = var_18571_epsilon_0, x = var_18570_cast_fp16)[name = string("op_18571_cast_fp16")]; + tensor var_18572_cast_fp16 = mul(x = x_567_cast_fp16, y = var_18571_cast_fp16)[name = string("op_18572_cast_fp16")]; + tensor k_447_cast_fp16 = mul(x = var_18572_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_447_cast_fp16")]; + tensor var_18574 = const()[name = string("op_18574"), val = tensor([1, 16, 128, 1])]; + tensor z_297_cast_fp16 = reshape(shape = var_18574, x = q_447_cast_fp16)[name = string("z_297_cast_fp16")]; + tensor var_18576 = const()[name = string("op_18576"), val = tensor([1, 8, 128, 1])]; + tensor z_299_cast_fp16 = reshape(shape = var_18576, x = k_447_cast_fp16)[name = string("z_299_cast_fp16")]; + tensor z1_297_begin_0 = const()[name = string("z1_297_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_297_end_0 = const()[name = string("z1_297_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_297_end_mask_0 = const()[name = string("z1_297_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_297_cast_fp16 = slice_by_index(begin = z1_297_begin_0, end = z1_297_end_0, end_mask = z1_297_end_mask_0, x = z_297_cast_fp16)[name = string("z1_297_cast_fp16")]; + tensor z2_297_begin_0 = const()[name = string("z2_297_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_297_end_0 = const()[name = string("z2_297_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_297_end_mask_0 = const()[name = string("z2_297_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_297_cast_fp16 = slice_by_index(begin = z2_297_begin_0, end = z2_297_end_0, end_mask = z2_297_end_mask_0, x = z_297_cast_fp16)[name = string("z2_297_cast_fp16")]; + tensor var_18584_cast_fp16 = mul(x = z_297_cast_fp16, y = cos_141_to_fp16)[name = string("op_18584_cast_fp16")]; + fp16 const_163_promoted_to_fp16 = const()[name = string("const_163_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18585_cast_fp16 = mul(x = z2_297_cast_fp16, y = const_163_promoted_to_fp16)[name = string("op_18585_cast_fp16")]; + bool var_18587_interleave_0 = const()[name = string("op_18587_interleave_0"), val = bool(false)]; + tensor var_18587_cast_fp16 = concat(axis = var_18493, interleave = var_18587_interleave_0, values = (var_18585_cast_fp16, z1_297_cast_fp16))[name = string("op_18587_cast_fp16")]; + tensor var_18588_cast_fp16 = mul(x = var_18587_cast_fp16, y = sin_141_to_fp16)[name = string("op_18588_cast_fp16")]; + tensor q_449_cast_fp16 = add(x = var_18584_cast_fp16, y = var_18588_cast_fp16)[name = string("q_449_cast_fp16")]; + tensor z1_299_begin_0 = const()[name = string("z1_299_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_299_end_0 = const()[name = string("z1_299_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_299_end_mask_0 = const()[name = string("z1_299_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_299_cast_fp16 = slice_by_index(begin = z1_299_begin_0, end = z1_299_end_0, end_mask = z1_299_end_mask_0, x = z_299_cast_fp16)[name = string("z1_299_cast_fp16")]; + tensor z2_299_begin_0 = const()[name = string("z2_299_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_299_end_0 = const()[name = string("z2_299_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_299_end_mask_0 = const()[name = string("z2_299_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_299_cast_fp16 = slice_by_index(begin = z2_299_begin_0, end = z2_299_end_0, end_mask = z2_299_end_mask_0, x = z_299_cast_fp16)[name = string("z2_299_cast_fp16")]; + tensor var_18596_cast_fp16 = mul(x = z_299_cast_fp16, y = cos_141_to_fp16)[name = string("op_18596_cast_fp16")]; + fp16 const_164_promoted_to_fp16 = const()[name = string("const_164_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18597_cast_fp16 = mul(x = z2_299_cast_fp16, y = const_164_promoted_to_fp16)[name = string("op_18597_cast_fp16")]; + bool var_18599_interleave_0 = const()[name = string("op_18599_interleave_0"), val = bool(false)]; + tensor var_18599_cast_fp16 = concat(axis = var_18493, interleave = var_18599_interleave_0, values = (var_18597_cast_fp16, z1_299_cast_fp16))[name = string("op_18599_cast_fp16")]; + tensor var_18600_cast_fp16 = mul(x = var_18599_cast_fp16, y = sin_141_to_fp16)[name = string("op_18600_cast_fp16")]; + tensor k_449_cast_fp16 = add(x = var_18596_cast_fp16, y = var_18600_cast_fp16)[name = string("k_449_cast_fp16")]; + tensor var_18602 = const()[name = string("op_18602"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_149_cast_fp16 = reshape(shape = var_18602, x = k_449_cast_fp16)[name = string("cur_key_149_cast_fp16")]; + tensor var_18604_to_fp16 = const()[name = string("op_18604_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646336)))]; + tensor var_18605_cast_fp16 = mul(x = key_cache_149_cast_fp16, y = var_18604_to_fp16)[name = string("op_18605_cast_fp16")]; + tensor upd_149_to_fp16 = const()[name = string("upd_149_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646464)))]; + tensor var_18606_cast_fp16 = mul(x = cur_key_149_cast_fp16, y = upd_149_to_fp16)[name = string("op_18606_cast_fp16")]; + tensor key_149_cast_fp16 = add(x = var_18605_cast_fp16, y = var_18606_cast_fp16)[name = string("key_149_cast_fp16")]; + tensor var_18608_to_fp16 = const()[name = string("op_18608_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646336)))]; + tensor var_18609_cast_fp16 = mul(x = value_cache_149_cast_fp16, y = var_18608_to_fp16)[name = string("op_18609_cast_fp16")]; + tensor var_18610_cast_fp16 = mul(x = v_149_cast_fp16, y = upd_149_to_fp16)[name = string("op_18610_cast_fp16")]; + tensor value_149_cast_fp16 = add(x = var_18609_cast_fp16, y = var_18610_cast_fp16)[name = string("value_149_cast_fp16")]; + tensor var_18612 = const()[name = string("op_18612"), val = tensor([1, 8, 128, 16])]; + tensor kh_297_cast_fp16 = reshape(shape = var_18612, x = key_149_cast_fp16)[name = string("kh_297_cast_fp16")]; + tensor var_18614 = const()[name = string("op_18614"), val = tensor([1, 8, 128, 16])]; + tensor vh_297_cast_fp16 = reshape(shape = var_18614, x = value_149_cast_fp16)[name = string("vh_297_cast_fp16")]; + tensor transpose_296_perm_0 = const()[name = string("transpose_296_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_148_reps_0 = const()[name = string("tile_148_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_296_cast_fp16 = transpose(perm = transpose_296_perm_0, x = kh_297_cast_fp16)[name = string("transpose_35")]; + tensor tile_148_cast_fp16 = tile(reps = tile_148_reps_0, x = transpose_296_cast_fp16)[name = string("tile_148_cast_fp16")]; + tensor concat_369 = const()[name = string("concat_369"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_296_cast_fp16 = reshape(shape = concat_369, x = tile_148_cast_fp16)[name = string("reshape_296_cast_fp16")]; + tensor transpose_297_perm_0 = const()[name = string("transpose_297_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_370 = const()[name = string("concat_370"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_297_cast_fp16 = transpose(perm = transpose_297_perm_0, x = reshape_296_cast_fp16)[name = string("transpose_34")]; + tensor reshape_297_cast_fp16 = reshape(shape = concat_370, x = transpose_297_cast_fp16)[name = string("reshape_297_cast_fp16")]; + tensor transpose_298_perm_0 = const()[name = string("transpose_298_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_149_reps_0 = const()[name = string("tile_149_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_298_cast_fp16 = transpose(perm = transpose_298_perm_0, x = vh_297_cast_fp16)[name = string("transpose_33")]; + tensor tile_149_cast_fp16 = tile(reps = tile_149_reps_0, x = transpose_298_cast_fp16)[name = string("tile_149_cast_fp16")]; + tensor concat_371 = const()[name = string("concat_371"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_298_cast_fp16 = reshape(shape = concat_371, x = tile_149_cast_fp16)[name = string("reshape_298_cast_fp16")]; + tensor transpose_299_perm_0 = const()[name = string("transpose_299_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_372 = const()[name = string("concat_372"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_299_cast_fp16 = transpose(perm = transpose_299_perm_0, x = reshape_298_cast_fp16)[name = string("transpose_32")]; + tensor reshape_299_cast_fp16 = reshape(shape = concat_372, x = transpose_299_cast_fp16)[name = string("reshape_299_cast_fp16")]; + fp16 var_18618_to_fp16 = const()[name = string("op_18618_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_18619_cast_fp16 = mul(x = q_449_cast_fp16, y = var_18618_to_fp16)[name = string("op_18619_cast_fp16")]; + tensor transpose_613_perm_0 = const()[name = string("transpose_613_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_323_transpose_x_1 = const()[name = string("w_323_transpose_x_1"), val = bool(true)]; + bool w_323_transpose_y_1 = const()[name = string("w_323_transpose_y_1"), val = bool(false)]; + tensor transpose_613_cast_fp16 = transpose(perm = transpose_613_perm_0, x = reshape_297_cast_fp16)[name = string("transpose_31")]; + tensor w_323_cast_fp16 = matmul(transpose_x = w_323_transpose_x_1, transpose_y = w_323_transpose_y_1, x = var_18619_cast_fp16, y = transpose_613_cast_fp16)[name = string("w_323_cast_fp16")]; + tensor pad_149_to_fp16 = const()[name = string("pad_149_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646592)))]; + tensor var_18622_cast_fp16 = add(x = w_323_cast_fp16, y = pad_149_to_fp16)[name = string("op_18622_cast_fp16")]; + tensor w_325_cast_fp16 = softmax(axis = var_18497, x = var_18622_cast_fp16)[name = string("w_325_cast_fp16")]; + tensor transpose_614_perm_0 = const()[name = string("transpose_614_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_149_transpose_x_1 = const()[name = string("attn_149_transpose_x_1"), val = bool(false)]; + bool attn_149_transpose_y_1 = const()[name = string("attn_149_transpose_y_1"), val = bool(true)]; + tensor transpose_614_cast_fp16 = transpose(perm = transpose_614_perm_0, x = reshape_299_cast_fp16)[name = string("transpose_30")]; + tensor attn_149_cast_fp16 = matmul(transpose_x = attn_149_transpose_x_1, transpose_y = attn_149_transpose_y_1, x = transpose_614_cast_fp16, y = w_325_cast_fp16)[name = string("attn_149_cast_fp16")]; + tensor var_18626 = const()[name = string("op_18626"), val = tensor([1, 2048, 1, 1])]; + tensor input_795_cast_fp16 = reshape(shape = var_18626, x = attn_149_cast_fp16)[name = string("input_795_cast_fp16")]; + string attn_output_149_pad_type_0 = const()[name = string("attn_output_149_pad_type_0"), val = string("valid")]; + tensor attn_output_149_strides_0 = const()[name = string("attn_output_149_strides_0"), val = tensor([1, 1])]; + tensor attn_output_149_pad_0 = const()[name = string("attn_output_149_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_149_dilations_0 = const()[name = string("attn_output_149_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_149_groups_0 = const()[name = string("attn_output_149_groups_0"), val = int32(1)]; + tensor attn_output_149_cast_fp16 = conv(dilations = attn_output_149_dilations_0, groups = attn_output_149_groups_0, pad = attn_output_149_pad_0, pad_type = attn_output_149_pad_type_0, strides = attn_output_149_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_795_cast_fp16)[name = string("attn_output_149_cast_fp16")]; + tensor x_569_cast_fp16 = add(x = x_563_cast_fp16, y = attn_output_149_cast_fp16)[name = string("x_569_cast_fp16")]; + tensor var_18640_cast_fp16 = mul(x = x_569_cast_fp16, y = x_569_cast_fp16)[name = string("op_18640_cast_fp16")]; + tensor variance_625_axes_0 = const()[name = string("variance_625_axes_0"), val = tensor([1])]; + bool variance_625_keep_dims_0 = const()[name = string("variance_625_keep_dims_0"), val = bool(true)]; + tensor variance_625_cast_fp16 = reduce_mean(axes = variance_625_axes_0, keep_dims = variance_625_keep_dims_0, x = var_18640_cast_fp16)[name = string("variance_625_cast_fp16")]; + fp16 var_18643_to_fp16 = const()[name = string("op_18643_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18644_cast_fp16 = add(x = variance_625_cast_fp16, y = var_18643_to_fp16)[name = string("op_18644_cast_fp16")]; + fp32 var_18645_epsilon_0 = const()[name = string("op_18645_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18645_cast_fp16 = rsqrt(epsilon = var_18645_epsilon_0, x = var_18644_cast_fp16)[name = string("op_18645_cast_fp16")]; + tensor var_18646_cast_fp16 = mul(x = x_569_cast_fp16, y = var_18645_cast_fp16)[name = string("op_18646_cast_fp16")]; + tensor input_797_cast_fp16 = mul(x = var_18646_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_797_cast_fp16")]; + string input_799_pad_type_0 = const()[name = string("input_799_pad_type_0"), val = string("valid")]; + tensor input_799_strides_0 = const()[name = string("input_799_strides_0"), val = tensor([1, 1])]; + tensor input_799_pad_0 = const()[name = string("input_799_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_799_dilations_0 = const()[name = string("input_799_dilations_0"), val = tensor([1, 1])]; + int32 input_799_groups_0 = const()[name = string("input_799_groups_0"), val = int32(1)]; + tensor input_799_cast_fp16 = conv(dilations = input_799_dilations_0, groups = input_799_groups_0, pad = input_799_pad_0, pad_type = input_799_pad_type_0, strides = input_799_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_797_cast_fp16)[name = string("input_799_cast_fp16")]; + tensor var_18654_cast_fp16 = silu(x = input_799_cast_fp16)[name = string("op_18654_cast_fp16")]; + string var_18660_pad_type_0 = const()[name = string("op_18660_pad_type_0"), val = string("valid")]; + tensor var_18660_strides_0 = const()[name = string("op_18660_strides_0"), val = tensor([1, 1])]; + tensor var_18660_pad_0 = const()[name = string("op_18660_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18660_dilations_0 = const()[name = string("op_18660_dilations_0"), val = tensor([1, 1])]; + int32 var_18660_groups_0 = const()[name = string("op_18660_groups_0"), val = int32(1)]; + tensor var_18660_cast_fp16 = conv(dilations = var_18660_dilations_0, groups = var_18660_groups_0, pad = var_18660_pad_0, pad_type = var_18660_pad_type_0, strides = var_18660_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_797_cast_fp16)[name = string("op_18660_cast_fp16")]; + tensor input_801_cast_fp16 = mul(x = var_18654_cast_fp16, y = var_18660_cast_fp16)[name = string("input_801_cast_fp16")]; + string h_149_pad_type_0 = const()[name = string("h_149_pad_type_0"), val = string("valid")]; + tensor h_149_strides_0 = const()[name = string("h_149_strides_0"), val = tensor([1, 1])]; + tensor h_149_pad_0 = const()[name = string("h_149_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_149_dilations_0 = const()[name = string("h_149_dilations_0"), val = tensor([1, 1])]; + int32 h_149_groups_0 = const()[name = string("h_149_groups_0"), val = int32(1)]; + tensor h_149_cast_fp16 = conv(dilations = h_149_dilations_0, groups = h_149_groups_0, pad = h_149_pad_0, pad_type = h_149_pad_type_0, strides = h_149_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_801_cast_fp16)[name = string("h_149_cast_fp16")]; + tensor inputs_27_cast_fp16 = add(x = x_569_cast_fp16, y = h_149_cast_fp16)[name = string("inputs_27_cast_fp16")]; + int32 var_18688 = const()[name = string("op_18688"), val = int32(1)]; + bool layer_key_caches_interleave_0 = const()[name = string("layer_key_caches_interleave_0"), val = bool(false)]; + tensor layer_key_caches_cast_fp16 = concat(axis = var_18688, interleave = layer_key_caches_interleave_0, values = (key_141_cast_fp16, key_143_cast_fp16, key_145_cast_fp16, key_147_cast_fp16, key_149_cast_fp16))[name = string("layer_key_caches_cast_fp16")]; + int32 var_18691 = const()[name = string("op_18691"), val = int32(1)]; + bool layer_value_caches_interleave_0 = const()[name = string("layer_value_caches_interleave_0"), val = bool(false)]; + tensor layer_value_caches_cast_fp16 = concat(axis = var_18691, interleave = layer_value_caches_interleave_0, values = (value_141_cast_fp16, value_143_cast_fp16, value_145_cast_fp16, value_147_cast_fp16, value_149_cast_fp16))[name = string("layer_value_caches_cast_fp16")]; + tensor inputs_sq_27_cast_fp16 = mul(x = inputs_27_cast_fp16, y = inputs_27_cast_fp16)[name = string("inputs_sq_27_cast_fp16")]; + tensor variance_627_axes_0 = const()[name = string("variance_627_axes_0"), val = tensor([1])]; + bool variance_627_keep_dims_0 = const()[name = string("variance_627_keep_dims_0"), val = bool(true)]; + tensor variance_627_cast_fp16 = reduce_mean(axes = variance_627_axes_0, keep_dims = variance_627_keep_dims_0, x = inputs_sq_27_cast_fp16)[name = string("variance_627_cast_fp16")]; + fp16 var_18701_to_fp16 = const()[name = string("op_18701_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18702_cast_fp16 = add(x = variance_627_cast_fp16, y = var_18701_to_fp16)[name = string("op_18702_cast_fp16")]; + fp32 var_18703_epsilon_0 = const()[name = string("op_18703_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18703_cast_fp16 = rsqrt(epsilon = var_18703_epsilon_0, x = var_18702_cast_fp16)[name = string("op_18703_cast_fp16")]; + tensor hidden_states_27_cast_fp16 = mul(x = inputs_27_cast_fp16, y = var_18703_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; + tensor input_803_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_27_cast_fp16)[name = string("input_803_cast_fp16")]; + string logits_53_pad_type_0 = const()[name = string("logits_53_pad_type_0"), val = string("valid")]; + tensor logits_53_strides_0 = const()[name = string("logits_53_strides_0"), val = tensor([1, 1])]; + tensor logits_53_pad_0 = const()[name = string("logits_53_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_53_dilations_0 = const()[name = string("logits_53_dilations_0"), val = tensor([1, 1])]; + int32 logits_53_groups_0 = const()[name = string("logits_53_groups_0"), val = int32(1)]; + tensor lm_heads_13_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105977984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108075200))))[name = string("lm_heads_13_weight_to_fp16_palettized")]; + tensor logits_53_cast_fp16 = conv(dilations = logits_53_dilations_0, groups = logits_53_groups_0, pad = logits_53_pad_0, pad_type = logits_53_pad_type_0, strides = logits_53_strides_0, weight = lm_heads_13_weight_to_fp16_palettized, x = input_803_cast_fp16)[name = string("logits_53_cast_fp16")]; + tensor var_18721 = const()[name = string("op_18721"), val = tensor([1, 2048])]; + tensor logits_55_cast_fp16 = reshape(shape = var_18721, x = logits_53_cast_fp16)[name = string("logits_55_cast_fp16")]; + tensor scaled_logits_27_cast_fp16 = real_div(x = logits_55_cast_fp16, y = temperature)[name = string("scaled_logits_27_cast_fp16")]; + int32 var_18731 = const()[name = string("op_18731"), val = int32(100)]; + int32 top_values_27_axis_0 = const()[name = string("top_values_27_axis_0"), val = int32(1)]; + bool top_values_27_ascending_0 = const()[name = string("top_values_27_ascending_0"), val = bool(false)]; + bool top_values_27_sort_0 = const()[name = string("top_values_27_sort_0"), val = bool(true)]; + bool top_values_27_return_indices_0 = const()[name = string("top_values_27_return_indices_0"), val = bool(true)]; + string top_values_27_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_27_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_27_cast_fp16_cast_uint16_0, tensor top_values_27_cast_fp16_cast_uint16_1 = topk(ascending = top_values_27_ascending_0, axis = top_values_27_axis_0, k = var_18731, output_indices_dtype = top_values_27_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_27_return_indices_0, sort = top_values_27_sort_0, x = scaled_logits_27_cast_fp16)[name = string("top_values_27_cast_fp16_cast_uint16")]; + tensor var_18737_cast_fp16 = mul(x = top_values_27_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_18737_cast_fp16")]; + tensor var_18741_cast_fp16 = add(x = var_18737_cast_fp16, y = var_2438_cast_fp16)[name = string("op_18741_cast_fp16")]; + tensor reduce_min_13_axes_0 = const()[name = string("reduce_min_13_axes_0"), val = tensor([1])]; + bool reduce_min_13_keep_dims_0 = const()[name = string("reduce_min_13_keep_dims_0"), val = bool(true)]; + tensor reduce_min_13_cast_fp16 = reduce_min(axes = reduce_min_13_axes_0, keep_dims = reduce_min_13_keep_dims_0, x = var_18741_cast_fp16)[name = string("reduce_min_13_cast_fp16")]; + tensor var_18744_cast_fp16 = greater_equal(x = scaled_logits_27_cast_fp16, y = reduce_min_13_cast_fp16)[name = string("op_18744_cast_fp16")]; + fp16 var_18745_value_0_to_fp16 = const()[name = string("op_18745_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_18745_cast_fp16 = fill_like(ref_tensor = scaled_logits_27_cast_fp16, value = var_18745_value_0_to_fp16)[name = string("op_18745_cast_fp16")]; + tensor masked_logits_27_cast_fp16 = select(a = scaled_logits_27_cast_fp16, b = var_18745_cast_fp16, cond = var_18744_cast_fp16)[name = string("masked_logits_27_cast_fp16")]; + tensor var_18749_begin_0 = const()[name = string("op_18749_begin_0"), val = tensor([13, 0])]; + tensor var_18749_end_0 = const()[name = string("op_18749_end_0"), val = tensor([14, 2048])]; + tensor var_18749_end_mask_0 = const()[name = string("op_18749_end_mask_0"), val = tensor([false, true])]; + tensor var_18749_squeeze_mask_0 = const()[name = string("op_18749_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_18749_cast_fp16 = slice_by_index(begin = var_18749_begin_0, end = var_18749_end_0, end_mask = var_18749_end_mask_0, squeeze_mask = var_18749_squeeze_mask_0, x = gumbel)[name = string("op_18749_cast_fp16")]; + tensor var_18752 = const()[name = string("op_18752"), val = tensor([1, 2048])]; + tensor var_18753_cast_fp16 = reshape(shape = var_18752, x = var_18749_cast_fp16)[name = string("op_18753_cast_fp16")]; + tensor noisy_logits_27_cast_fp16 = add(x = masked_logits_27_cast_fp16, y = var_18753_cast_fp16)[name = string("noisy_logits_27_cast_fp16")]; + int32 code_27_axis_0 = const()[name = string("code_27_axis_0"), val = int32(1)]; + bool code_27_keep_dims_0 = const()[name = string("code_27_keep_dims_0"), val = bool(false)]; + string code_27_output_dtype_0 = const()[name = string("code_27_output_dtype_0"), val = string("int32")]; + tensor code_27_cast_fp16 = reduce_argmax(axis = code_27_axis_0, keep_dims = code_27_keep_dims_0, output_dtype = code_27_output_dtype_0, x = noisy_logits_27_cast_fp16)[name = string("code_27_cast_fp16")]; + int32 var_18764 = const()[name = string("op_18764"), val = int32(26624)]; + tensor input_805 = add(x = code_27_cast_fp16, y = var_18764)[name = string("input_805")]; + int32 code_embed_53_axis_0 = const()[name = string("code_embed_53_axis_0"), val = int32(0)]; + int32 code_embed_53_batch_dims_0 = const()[name = string("code_embed_53_batch_dims_0"), val = int32(0)]; + bool code_embed_53_validate_indices_0 = const()[name = string("code_embed_53_validate_indices_0"), val = bool(false)]; + string input_805_to_uint16_dtype_0 = const()[name = string("input_805_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_805_to_uint16 = cast(dtype = input_805_to_uint16_dtype_0, x = input_805)[name = string("cast_1")]; + tensor code_embed_53_cast_fp16_cast_uint16 = gather(axis = code_embed_53_axis_0, batch_dims = code_embed_53_batch_dims_0, indices = input_805_to_uint16, validate_indices = code_embed_53_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_53_cast_fp16_cast_uint16")]; + tensor var_18768 = const()[name = string("op_18768"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_55_cast_fp16 = reshape(shape = var_18768, x = code_embed_53_cast_fp16_cast_uint16)[name = string("code_embed_55_cast_fp16")]; + tensor embed_sum_cast_fp16 = add(x = embed_sum_27_cast_fp16, y = code_embed_55_cast_fp16)[name = string("embed_sum_cast_fp16")]; + tensor key_cache_151_begin_0 = const()[name = string("key_cache_151_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor key_cache_151_end_0 = const()[name = string("key_cache_151_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor key_cache_151_end_mask_0 = const()[name = string("key_cache_151_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_151_cast_fp16 = slice_by_index(begin = key_cache_151_begin_0, end = key_cache_151_end_0, end_mask = key_cache_151_end_mask_0, x = layer_key_caches_cast_fp16)[name = string("key_cache_151_cast_fp16")]; + tensor value_cache_151_begin_0 = const()[name = string("value_cache_151_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor value_cache_151_end_0 = const()[name = string("value_cache_151_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor value_cache_151_end_mask_0 = const()[name = string("value_cache_151_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_151_cast_fp16 = slice_by_index(begin = value_cache_151_begin_0, end = value_cache_151_end_0, end_mask = value_cache_151_end_mask_0, x = layer_value_caches_cast_fp16)[name = string("value_cache_151_cast_fp16")]; + int32 var_18853 = const()[name = string("op_18853"), val = int32(2)]; + int32 var_18857 = const()[name = string("op_18857"), val = int32(3)]; + tensor var_18872_cast_fp16 = mul(x = code_embed_55_cast_fp16, y = code_embed_55_cast_fp16)[name = string("op_18872_cast_fp16")]; + tensor variance_629_axes_0 = const()[name = string("variance_629_axes_0"), val = tensor([1])]; + bool variance_629_keep_dims_0 = const()[name = string("variance_629_keep_dims_0"), val = bool(true)]; + tensor variance_629_cast_fp16 = reduce_mean(axes = variance_629_axes_0, keep_dims = variance_629_keep_dims_0, x = var_18872_cast_fp16)[name = string("variance_629_cast_fp16")]; + fp16 var_18875_to_fp16 = const()[name = string("op_18875_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18876_cast_fp16 = add(x = variance_629_cast_fp16, y = var_18875_to_fp16)[name = string("op_18876_cast_fp16")]; + fp32 var_18877_epsilon_0 = const()[name = string("op_18877_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18877_cast_fp16 = rsqrt(epsilon = var_18877_epsilon_0, x = var_18876_cast_fp16)[name = string("op_18877_cast_fp16")]; + tensor var_18878_cast_fp16 = mul(x = code_embed_55_cast_fp16, y = var_18877_cast_fp16)[name = string("op_18878_cast_fp16")]; + tensor input_807_cast_fp16 = mul(x = var_18878_cast_fp16, y = layers_0_input_layernorm_weight_to_fp16)[name = string("input_807_cast_fp16")]; + string q_451_pad_type_0 = const()[name = string("q_451_pad_type_0"), val = string("valid")]; + tensor q_451_strides_0 = const()[name = string("q_451_strides_0"), val = tensor([1, 1])]; + tensor q_451_pad_0 = const()[name = string("q_451_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_451_dilations_0 = const()[name = string("q_451_dilations_0"), val = tensor([1, 1])]; + int32 q_451_groups_0 = const()[name = string("q_451_groups_0"), val = int32(1)]; + tensor q_451_cast_fp16 = conv(dilations = q_451_dilations_0, groups = q_451_groups_0, pad = q_451_pad_0, pad_type = q_451_pad_type_0, strides = q_451_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = input_807_cast_fp16)[name = string("q_451_cast_fp16")]; + string k_451_pad_type_0 = const()[name = string("k_451_pad_type_0"), val = string("valid")]; + tensor k_451_strides_0 = const()[name = string("k_451_strides_0"), val = tensor([1, 1])]; + tensor k_451_pad_0 = const()[name = string("k_451_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_451_dilations_0 = const()[name = string("k_451_dilations_0"), val = tensor([1, 1])]; + int32 k_451_groups_0 = const()[name = string("k_451_groups_0"), val = int32(1)]; + tensor k_451_cast_fp16 = conv(dilations = k_451_dilations_0, groups = k_451_groups_0, pad = k_451_pad_0, pad_type = k_451_pad_type_0, strides = k_451_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = input_807_cast_fp16)[name = string("k_451_cast_fp16")]; + string v_151_pad_type_0 = const()[name = string("v_151_pad_type_0"), val = string("valid")]; + tensor v_151_strides_0 = const()[name = string("v_151_strides_0"), val = tensor([1, 1])]; + tensor v_151_pad_0 = const()[name = string("v_151_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_151_dilations_0 = const()[name = string("v_151_dilations_0"), val = tensor([1, 1])]; + int32 v_151_groups_0 = const()[name = string("v_151_groups_0"), val = int32(1)]; + tensor v_151_cast_fp16 = conv(dilations = v_151_dilations_0, groups = v_151_groups_0, pad = v_151_pad_0, pad_type = v_151_pad_type_0, strides = v_151_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = input_807_cast_fp16)[name = string("v_151_cast_fp16")]; + tensor var_18912 = const()[name = string("op_18912"), val = tensor([16, 128, 1, 1])]; + tensor x_571_cast_fp16 = reshape(shape = var_18912, x = q_451_cast_fp16)[name = string("x_571_cast_fp16")]; + tensor var_18915_cast_fp16 = mul(x = x_571_cast_fp16, y = x_571_cast_fp16)[name = string("op_18915_cast_fp16")]; + tensor variance_631_axes_0 = const()[name = string("variance_631_axes_0"), val = tensor([1])]; + bool variance_631_keep_dims_0 = const()[name = string("variance_631_keep_dims_0"), val = bool(true)]; + tensor variance_631_cast_fp16 = reduce_mean(axes = variance_631_axes_0, keep_dims = variance_631_keep_dims_0, x = var_18915_cast_fp16)[name = string("variance_631_cast_fp16")]; + fp16 var_18918_to_fp16 = const()[name = string("op_18918_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18919_cast_fp16 = add(x = variance_631_cast_fp16, y = var_18918_to_fp16)[name = string("op_18919_cast_fp16")]; + fp32 var_18920_epsilon_0 = const()[name = string("op_18920_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18920_cast_fp16 = rsqrt(epsilon = var_18920_epsilon_0, x = var_18919_cast_fp16)[name = string("op_18920_cast_fp16")]; + tensor var_18921_cast_fp16 = mul(x = x_571_cast_fp16, y = var_18920_cast_fp16)[name = string("op_18921_cast_fp16")]; + tensor q_453_cast_fp16 = mul(x = var_18921_cast_fp16, y = layers_0_self_attn_q_norm_weight_to_fp16)[name = string("q_453_cast_fp16")]; + tensor var_18923 = const()[name = string("op_18923"), val = tensor([8, 128, 1, 1])]; + tensor x_573_cast_fp16 = reshape(shape = var_18923, x = k_451_cast_fp16)[name = string("x_573_cast_fp16")]; + tensor var_18926_cast_fp16 = mul(x = x_573_cast_fp16, y = x_573_cast_fp16)[name = string("op_18926_cast_fp16")]; + tensor variance_633_axes_0 = const()[name = string("variance_633_axes_0"), val = tensor([1])]; + bool variance_633_keep_dims_0 = const()[name = string("variance_633_keep_dims_0"), val = bool(true)]; + tensor variance_633_cast_fp16 = reduce_mean(axes = variance_633_axes_0, keep_dims = variance_633_keep_dims_0, x = var_18926_cast_fp16)[name = string("variance_633_cast_fp16")]; + fp16 var_18929_to_fp16 = const()[name = string("op_18929_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18930_cast_fp16 = add(x = variance_633_cast_fp16, y = var_18929_to_fp16)[name = string("op_18930_cast_fp16")]; + fp32 var_18931_epsilon_0 = const()[name = string("op_18931_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18931_cast_fp16 = rsqrt(epsilon = var_18931_epsilon_0, x = var_18930_cast_fp16)[name = string("op_18931_cast_fp16")]; + tensor var_18932_cast_fp16 = mul(x = x_573_cast_fp16, y = var_18931_cast_fp16)[name = string("op_18932_cast_fp16")]; + tensor k_453_cast_fp16 = mul(x = var_18932_cast_fp16, y = layers_0_self_attn_k_norm_weight_to_fp16)[name = string("k_453_cast_fp16")]; + tensor var_18934 = const()[name = string("op_18934"), val = tensor([1, 16, 128, 1])]; + tensor z_301_cast_fp16 = reshape(shape = var_18934, x = q_453_cast_fp16)[name = string("z_301_cast_fp16")]; + tensor var_18936 = const()[name = string("op_18936"), val = tensor([1, 8, 128, 1])]; + tensor z_303_cast_fp16 = reshape(shape = var_18936, x = k_453_cast_fp16)[name = string("z_303_cast_fp16")]; + tensor z1_301_begin_0 = const()[name = string("z1_301_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_301_end_0 = const()[name = string("z1_301_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_301_end_mask_0 = const()[name = string("z1_301_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_301_cast_fp16 = slice_by_index(begin = z1_301_begin_0, end = z1_301_end_0, end_mask = z1_301_end_mask_0, x = z_301_cast_fp16)[name = string("z1_301_cast_fp16")]; + tensor z2_301_begin_0 = const()[name = string("z2_301_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_301_end_0 = const()[name = string("z2_301_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_301_end_mask_0 = const()[name = string("z2_301_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_301_cast_fp16 = slice_by_index(begin = z2_301_begin_0, end = z2_301_end_0, end_mask = z2_301_end_mask_0, x = z_301_cast_fp16)[name = string("z2_301_cast_fp16")]; + tensor cos_151_to_fp16 = const()[name = string("cos_151_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141646720)))]; + tensor var_18944_cast_fp16 = mul(x = z_301_cast_fp16, y = cos_151_to_fp16)[name = string("op_18944_cast_fp16")]; + fp16 const_166_promoted_to_fp16 = const()[name = string("const_166_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18945_cast_fp16 = mul(x = z2_301_cast_fp16, y = const_166_promoted_to_fp16)[name = string("op_18945_cast_fp16")]; + bool var_18947_interleave_0 = const()[name = string("op_18947_interleave_0"), val = bool(false)]; + tensor var_18947_cast_fp16 = concat(axis = var_18853, interleave = var_18947_interleave_0, values = (var_18945_cast_fp16, z1_301_cast_fp16))[name = string("op_18947_cast_fp16")]; + tensor sin_151_to_fp16 = const()[name = string("sin_151_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647040)))]; + tensor var_18948_cast_fp16 = mul(x = var_18947_cast_fp16, y = sin_151_to_fp16)[name = string("op_18948_cast_fp16")]; + tensor q_455_cast_fp16 = add(x = var_18944_cast_fp16, y = var_18948_cast_fp16)[name = string("q_455_cast_fp16")]; + tensor z1_303_begin_0 = const()[name = string("z1_303_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_303_end_0 = const()[name = string("z1_303_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_303_end_mask_0 = const()[name = string("z1_303_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_303_cast_fp16 = slice_by_index(begin = z1_303_begin_0, end = z1_303_end_0, end_mask = z1_303_end_mask_0, x = z_303_cast_fp16)[name = string("z1_303_cast_fp16")]; + tensor z2_303_begin_0 = const()[name = string("z2_303_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_303_end_0 = const()[name = string("z2_303_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_303_end_mask_0 = const()[name = string("z2_303_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_303_cast_fp16 = slice_by_index(begin = z2_303_begin_0, end = z2_303_end_0, end_mask = z2_303_end_mask_0, x = z_303_cast_fp16)[name = string("z2_303_cast_fp16")]; + tensor var_18956_cast_fp16 = mul(x = z_303_cast_fp16, y = cos_151_to_fp16)[name = string("op_18956_cast_fp16")]; + fp16 const_167_promoted_to_fp16 = const()[name = string("const_167_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18957_cast_fp16 = mul(x = z2_303_cast_fp16, y = const_167_promoted_to_fp16)[name = string("op_18957_cast_fp16")]; + bool var_18959_interleave_0 = const()[name = string("op_18959_interleave_0"), val = bool(false)]; + tensor var_18959_cast_fp16 = concat(axis = var_18853, interleave = var_18959_interleave_0, values = (var_18957_cast_fp16, z1_303_cast_fp16))[name = string("op_18959_cast_fp16")]; + tensor var_18960_cast_fp16 = mul(x = var_18959_cast_fp16, y = sin_151_to_fp16)[name = string("op_18960_cast_fp16")]; + tensor k_455_cast_fp16 = add(x = var_18956_cast_fp16, y = var_18960_cast_fp16)[name = string("k_455_cast_fp16")]; + tensor var_18962 = const()[name = string("op_18962"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_151_cast_fp16 = reshape(shape = var_18962, x = k_455_cast_fp16)[name = string("cur_key_151_cast_fp16")]; + tensor var_18964_to_fp16 = const()[name = string("op_18964_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647360)))]; + tensor var_18965_cast_fp16 = mul(x = key_cache_151_cast_fp16, y = var_18964_to_fp16)[name = string("op_18965_cast_fp16")]; + tensor upd_151_to_fp16 = const()[name = string("upd_151_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647488)))]; + tensor var_18966_cast_fp16 = mul(x = cur_key_151_cast_fp16, y = upd_151_to_fp16)[name = string("op_18966_cast_fp16")]; + tensor key_151_cast_fp16 = add(x = var_18965_cast_fp16, y = var_18966_cast_fp16)[name = string("key_151_cast_fp16")]; + tensor var_18968_to_fp16 = const()[name = string("op_18968_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647360)))]; + tensor var_18969_cast_fp16 = mul(x = value_cache_151_cast_fp16, y = var_18968_to_fp16)[name = string("op_18969_cast_fp16")]; + tensor var_18970_cast_fp16 = mul(x = v_151_cast_fp16, y = upd_151_to_fp16)[name = string("op_18970_cast_fp16")]; + tensor value_151_cast_fp16 = add(x = var_18969_cast_fp16, y = var_18970_cast_fp16)[name = string("value_151_cast_fp16")]; + tensor var_18972 = const()[name = string("op_18972"), val = tensor([1, 8, 128, 16])]; + tensor kh_301_cast_fp16 = reshape(shape = var_18972, x = key_151_cast_fp16)[name = string("kh_301_cast_fp16")]; + tensor var_18974 = const()[name = string("op_18974"), val = tensor([1, 8, 128, 16])]; + tensor vh_301_cast_fp16 = reshape(shape = var_18974, x = value_151_cast_fp16)[name = string("vh_301_cast_fp16")]; + tensor transpose_300_perm_0 = const()[name = string("transpose_300_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_150_reps_0 = const()[name = string("tile_150_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_300_cast_fp16 = transpose(perm = transpose_300_perm_0, x = kh_301_cast_fp16)[name = string("transpose_29")]; + tensor tile_150_cast_fp16 = tile(reps = tile_150_reps_0, x = transpose_300_cast_fp16)[name = string("tile_150_cast_fp16")]; + tensor concat_378 = const()[name = string("concat_378"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_300_cast_fp16 = reshape(shape = concat_378, x = tile_150_cast_fp16)[name = string("reshape_300_cast_fp16")]; + tensor transpose_301_perm_0 = const()[name = string("transpose_301_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_379 = const()[name = string("concat_379"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_301_cast_fp16 = transpose(perm = transpose_301_perm_0, x = reshape_300_cast_fp16)[name = string("transpose_28")]; + tensor reshape_301_cast_fp16 = reshape(shape = concat_379, x = transpose_301_cast_fp16)[name = string("reshape_301_cast_fp16")]; + tensor transpose_302_perm_0 = const()[name = string("transpose_302_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_151_reps_0 = const()[name = string("tile_151_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_302_cast_fp16 = transpose(perm = transpose_302_perm_0, x = vh_301_cast_fp16)[name = string("transpose_27")]; + tensor tile_151_cast_fp16 = tile(reps = tile_151_reps_0, x = transpose_302_cast_fp16)[name = string("tile_151_cast_fp16")]; + tensor concat_380 = const()[name = string("concat_380"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_302_cast_fp16 = reshape(shape = concat_380, x = tile_151_cast_fp16)[name = string("reshape_302_cast_fp16")]; + tensor transpose_303_perm_0 = const()[name = string("transpose_303_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_381 = const()[name = string("concat_381"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_303_cast_fp16 = transpose(perm = transpose_303_perm_0, x = reshape_302_cast_fp16)[name = string("transpose_26")]; + tensor reshape_303_cast_fp16 = reshape(shape = concat_381, x = transpose_303_cast_fp16)[name = string("reshape_303_cast_fp16")]; + fp16 var_18978_to_fp16 = const()[name = string("op_18978_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_18979_cast_fp16 = mul(x = q_455_cast_fp16, y = var_18978_to_fp16)[name = string("op_18979_cast_fp16")]; + tensor transpose_617_perm_0 = const()[name = string("transpose_617_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_329_transpose_x_1 = const()[name = string("w_329_transpose_x_1"), val = bool(true)]; + bool w_329_transpose_y_1 = const()[name = string("w_329_transpose_y_1"), val = bool(false)]; + tensor transpose_617_cast_fp16 = transpose(perm = transpose_617_perm_0, x = reshape_301_cast_fp16)[name = string("transpose_25")]; + tensor w_329_cast_fp16 = matmul(transpose_x = w_329_transpose_x_1, transpose_y = w_329_transpose_y_1, x = var_18979_cast_fp16, y = transpose_617_cast_fp16)[name = string("w_329_cast_fp16")]; + tensor w_331_cast_fp16 = softmax(axis = var_18857, x = w_329_cast_fp16)[name = string("w_331_cast_fp16")]; + tensor transpose_618_perm_0 = const()[name = string("transpose_618_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_151_transpose_x_1 = const()[name = string("attn_151_transpose_x_1"), val = bool(false)]; + bool attn_151_transpose_y_1 = const()[name = string("attn_151_transpose_y_1"), val = bool(true)]; + tensor transpose_618_cast_fp16 = transpose(perm = transpose_618_perm_0, x = reshape_303_cast_fp16)[name = string("transpose_24")]; + tensor attn_151_cast_fp16 = matmul(transpose_x = attn_151_transpose_x_1, transpose_y = attn_151_transpose_y_1, x = transpose_618_cast_fp16, y = w_331_cast_fp16)[name = string("attn_151_cast_fp16")]; + tensor var_18986 = const()[name = string("op_18986"), val = tensor([1, 2048, 1, 1])]; + tensor input_809_cast_fp16 = reshape(shape = var_18986, x = attn_151_cast_fp16)[name = string("input_809_cast_fp16")]; + string attn_output_151_pad_type_0 = const()[name = string("attn_output_151_pad_type_0"), val = string("valid")]; + tensor attn_output_151_strides_0 = const()[name = string("attn_output_151_strides_0"), val = tensor([1, 1])]; + tensor attn_output_151_pad_0 = const()[name = string("attn_output_151_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_151_dilations_0 = const()[name = string("attn_output_151_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_151_groups_0 = const()[name = string("attn_output_151_groups_0"), val = int32(1)]; + tensor attn_output_151_cast_fp16 = conv(dilations = attn_output_151_dilations_0, groups = attn_output_151_groups_0, pad = attn_output_151_pad_0, pad_type = attn_output_151_pad_type_0, strides = attn_output_151_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_809_cast_fp16)[name = string("attn_output_151_cast_fp16")]; + tensor x_575_cast_fp16 = add(x = code_embed_55_cast_fp16, y = attn_output_151_cast_fp16)[name = string("x_575_cast_fp16")]; + tensor var_19000_cast_fp16 = mul(x = x_575_cast_fp16, y = x_575_cast_fp16)[name = string("op_19000_cast_fp16")]; + tensor variance_635_axes_0 = const()[name = string("variance_635_axes_0"), val = tensor([1])]; + bool variance_635_keep_dims_0 = const()[name = string("variance_635_keep_dims_0"), val = bool(true)]; + tensor variance_635_cast_fp16 = reduce_mean(axes = variance_635_axes_0, keep_dims = variance_635_keep_dims_0, x = var_19000_cast_fp16)[name = string("variance_635_cast_fp16")]; + fp16 var_19003_to_fp16 = const()[name = string("op_19003_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19004_cast_fp16 = add(x = variance_635_cast_fp16, y = var_19003_to_fp16)[name = string("op_19004_cast_fp16")]; + fp32 var_19005_epsilon_0 = const()[name = string("op_19005_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19005_cast_fp16 = rsqrt(epsilon = var_19005_epsilon_0, x = var_19004_cast_fp16)[name = string("op_19005_cast_fp16")]; + tensor var_19006_cast_fp16 = mul(x = x_575_cast_fp16, y = var_19005_cast_fp16)[name = string("op_19006_cast_fp16")]; + tensor input_811_cast_fp16 = mul(x = var_19006_cast_fp16, y = layers_0_post_attention_layernorm_weight_to_fp16)[name = string("input_811_cast_fp16")]; + string input_813_pad_type_0 = const()[name = string("input_813_pad_type_0"), val = string("valid")]; + tensor input_813_strides_0 = const()[name = string("input_813_strides_0"), val = tensor([1, 1])]; + tensor input_813_pad_0 = const()[name = string("input_813_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_813_dilations_0 = const()[name = string("input_813_dilations_0"), val = tensor([1, 1])]; + int32 input_813_groups_0 = const()[name = string("input_813_groups_0"), val = int32(1)]; + tensor input_813_cast_fp16 = conv(dilations = input_813_dilations_0, groups = input_813_groups_0, pad = input_813_pad_0, pad_type = input_813_pad_type_0, strides = input_813_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_811_cast_fp16)[name = string("input_813_cast_fp16")]; + tensor var_19014_cast_fp16 = silu(x = input_813_cast_fp16)[name = string("op_19014_cast_fp16")]; + string var_19020_pad_type_0 = const()[name = string("op_19020_pad_type_0"), val = string("valid")]; + tensor var_19020_strides_0 = const()[name = string("op_19020_strides_0"), val = tensor([1, 1])]; + tensor var_19020_pad_0 = const()[name = string("op_19020_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19020_dilations_0 = const()[name = string("op_19020_dilations_0"), val = tensor([1, 1])]; + int32 var_19020_groups_0 = const()[name = string("op_19020_groups_0"), val = int32(1)]; + tensor var_19020_cast_fp16 = conv(dilations = var_19020_dilations_0, groups = var_19020_groups_0, pad = var_19020_pad_0, pad_type = var_19020_pad_type_0, strides = var_19020_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_811_cast_fp16)[name = string("op_19020_cast_fp16")]; + tensor input_815_cast_fp16 = mul(x = var_19014_cast_fp16, y = var_19020_cast_fp16)[name = string("input_815_cast_fp16")]; + string h_151_pad_type_0 = const()[name = string("h_151_pad_type_0"), val = string("valid")]; + tensor h_151_strides_0 = const()[name = string("h_151_strides_0"), val = tensor([1, 1])]; + tensor h_151_pad_0 = const()[name = string("h_151_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_151_dilations_0 = const()[name = string("h_151_dilations_0"), val = tensor([1, 1])]; + int32 h_151_groups_0 = const()[name = string("h_151_groups_0"), val = int32(1)]; + tensor h_151_cast_fp16 = conv(dilations = h_151_dilations_0, groups = h_151_groups_0, pad = h_151_pad_0, pad_type = h_151_pad_type_0, strides = h_151_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_815_cast_fp16)[name = string("h_151_cast_fp16")]; + tensor x_577_cast_fp16 = add(x = x_575_cast_fp16, y = h_151_cast_fp16)[name = string("x_577_cast_fp16")]; + tensor key_cache_153_begin_0 = const()[name = string("key_cache_153_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor key_cache_153_end_0 = const()[name = string("key_cache_153_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor key_cache_153_end_mask_0 = const()[name = string("key_cache_153_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_153_cast_fp16 = slice_by_index(begin = key_cache_153_begin_0, end = key_cache_153_end_0, end_mask = key_cache_153_end_mask_0, x = layer_key_caches_cast_fp16)[name = string("key_cache_153_cast_fp16")]; + tensor value_cache_153_begin_0 = const()[name = string("value_cache_153_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor value_cache_153_end_0 = const()[name = string("value_cache_153_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor value_cache_153_end_mask_0 = const()[name = string("value_cache_153_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_153_cast_fp16 = slice_by_index(begin = value_cache_153_begin_0, end = value_cache_153_end_0, end_mask = value_cache_153_end_mask_0, x = layer_value_caches_cast_fp16)[name = string("value_cache_153_cast_fp16")]; + int32 var_19059 = const()[name = string("op_19059"), val = int32(2)]; + int32 var_19063 = const()[name = string("op_19063"), val = int32(3)]; + tensor var_19078_cast_fp16 = mul(x = x_577_cast_fp16, y = x_577_cast_fp16)[name = string("op_19078_cast_fp16")]; + tensor variance_637_axes_0 = const()[name = string("variance_637_axes_0"), val = tensor([1])]; + bool variance_637_keep_dims_0 = const()[name = string("variance_637_keep_dims_0"), val = bool(true)]; + tensor variance_637_cast_fp16 = reduce_mean(axes = variance_637_axes_0, keep_dims = variance_637_keep_dims_0, x = var_19078_cast_fp16)[name = string("variance_637_cast_fp16")]; + fp16 var_19081_to_fp16 = const()[name = string("op_19081_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19082_cast_fp16 = add(x = variance_637_cast_fp16, y = var_19081_to_fp16)[name = string("op_19082_cast_fp16")]; + fp32 var_19083_epsilon_0 = const()[name = string("op_19083_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19083_cast_fp16 = rsqrt(epsilon = var_19083_epsilon_0, x = var_19082_cast_fp16)[name = string("op_19083_cast_fp16")]; + tensor var_19084_cast_fp16 = mul(x = x_577_cast_fp16, y = var_19083_cast_fp16)[name = string("op_19084_cast_fp16")]; + tensor input_817_cast_fp16 = mul(x = var_19084_cast_fp16, y = layers_1_input_layernorm_weight_to_fp16)[name = string("input_817_cast_fp16")]; + string q_457_pad_type_0 = const()[name = string("q_457_pad_type_0"), val = string("valid")]; + tensor q_457_strides_0 = const()[name = string("q_457_strides_0"), val = tensor([1, 1])]; + tensor q_457_pad_0 = const()[name = string("q_457_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_457_dilations_0 = const()[name = string("q_457_dilations_0"), val = tensor([1, 1])]; + int32 q_457_groups_0 = const()[name = string("q_457_groups_0"), val = int32(1)]; + tensor q_457_cast_fp16 = conv(dilations = q_457_dilations_0, groups = q_457_groups_0, pad = q_457_pad_0, pad_type = q_457_pad_type_0, strides = q_457_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = input_817_cast_fp16)[name = string("q_457_cast_fp16")]; + string k_457_pad_type_0 = const()[name = string("k_457_pad_type_0"), val = string("valid")]; + tensor k_457_strides_0 = const()[name = string("k_457_strides_0"), val = tensor([1, 1])]; + tensor k_457_pad_0 = const()[name = string("k_457_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_457_dilations_0 = const()[name = string("k_457_dilations_0"), val = tensor([1, 1])]; + int32 k_457_groups_0 = const()[name = string("k_457_groups_0"), val = int32(1)]; + tensor k_457_cast_fp16 = conv(dilations = k_457_dilations_0, groups = k_457_groups_0, pad = k_457_pad_0, pad_type = k_457_pad_type_0, strides = k_457_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = input_817_cast_fp16)[name = string("k_457_cast_fp16")]; + string v_153_pad_type_0 = const()[name = string("v_153_pad_type_0"), val = string("valid")]; + tensor v_153_strides_0 = const()[name = string("v_153_strides_0"), val = tensor([1, 1])]; + tensor v_153_pad_0 = const()[name = string("v_153_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_153_dilations_0 = const()[name = string("v_153_dilations_0"), val = tensor([1, 1])]; + int32 v_153_groups_0 = const()[name = string("v_153_groups_0"), val = int32(1)]; + tensor v_153_cast_fp16 = conv(dilations = v_153_dilations_0, groups = v_153_groups_0, pad = v_153_pad_0, pad_type = v_153_pad_type_0, strides = v_153_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = input_817_cast_fp16)[name = string("v_153_cast_fp16")]; + tensor var_19118 = const()[name = string("op_19118"), val = tensor([16, 128, 1, 1])]; + tensor x_579_cast_fp16 = reshape(shape = var_19118, x = q_457_cast_fp16)[name = string("x_579_cast_fp16")]; + tensor var_19121_cast_fp16 = mul(x = x_579_cast_fp16, y = x_579_cast_fp16)[name = string("op_19121_cast_fp16")]; + tensor variance_639_axes_0 = const()[name = string("variance_639_axes_0"), val = tensor([1])]; + bool variance_639_keep_dims_0 = const()[name = string("variance_639_keep_dims_0"), val = bool(true)]; + tensor variance_639_cast_fp16 = reduce_mean(axes = variance_639_axes_0, keep_dims = variance_639_keep_dims_0, x = var_19121_cast_fp16)[name = string("variance_639_cast_fp16")]; + fp16 var_19124_to_fp16 = const()[name = string("op_19124_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19125_cast_fp16 = add(x = variance_639_cast_fp16, y = var_19124_to_fp16)[name = string("op_19125_cast_fp16")]; + fp32 var_19126_epsilon_0 = const()[name = string("op_19126_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19126_cast_fp16 = rsqrt(epsilon = var_19126_epsilon_0, x = var_19125_cast_fp16)[name = string("op_19126_cast_fp16")]; + tensor var_19127_cast_fp16 = mul(x = x_579_cast_fp16, y = var_19126_cast_fp16)[name = string("op_19127_cast_fp16")]; + tensor q_459_cast_fp16 = mul(x = var_19127_cast_fp16, y = layers_1_self_attn_q_norm_weight_to_fp16)[name = string("q_459_cast_fp16")]; + tensor var_19129 = const()[name = string("op_19129"), val = tensor([8, 128, 1, 1])]; + tensor x_581_cast_fp16 = reshape(shape = var_19129, x = k_457_cast_fp16)[name = string("x_581_cast_fp16")]; + tensor var_19132_cast_fp16 = mul(x = x_581_cast_fp16, y = x_581_cast_fp16)[name = string("op_19132_cast_fp16")]; + tensor variance_641_axes_0 = const()[name = string("variance_641_axes_0"), val = tensor([1])]; + bool variance_641_keep_dims_0 = const()[name = string("variance_641_keep_dims_0"), val = bool(true)]; + tensor variance_641_cast_fp16 = reduce_mean(axes = variance_641_axes_0, keep_dims = variance_641_keep_dims_0, x = var_19132_cast_fp16)[name = string("variance_641_cast_fp16")]; + fp16 var_19135_to_fp16 = const()[name = string("op_19135_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19136_cast_fp16 = add(x = variance_641_cast_fp16, y = var_19135_to_fp16)[name = string("op_19136_cast_fp16")]; + fp32 var_19137_epsilon_0 = const()[name = string("op_19137_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19137_cast_fp16 = rsqrt(epsilon = var_19137_epsilon_0, x = var_19136_cast_fp16)[name = string("op_19137_cast_fp16")]; + tensor var_19138_cast_fp16 = mul(x = x_581_cast_fp16, y = var_19137_cast_fp16)[name = string("op_19138_cast_fp16")]; + tensor k_459_cast_fp16 = mul(x = var_19138_cast_fp16, y = layers_1_self_attn_k_norm_weight_to_fp16)[name = string("k_459_cast_fp16")]; + tensor var_19140 = const()[name = string("op_19140"), val = tensor([1, 16, 128, 1])]; + tensor z_305_cast_fp16 = reshape(shape = var_19140, x = q_459_cast_fp16)[name = string("z_305_cast_fp16")]; + tensor var_19142 = const()[name = string("op_19142"), val = tensor([1, 8, 128, 1])]; + tensor z_307_cast_fp16 = reshape(shape = var_19142, x = k_459_cast_fp16)[name = string("z_307_cast_fp16")]; + tensor z1_305_begin_0 = const()[name = string("z1_305_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_305_end_0 = const()[name = string("z1_305_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_305_end_mask_0 = const()[name = string("z1_305_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_305_cast_fp16 = slice_by_index(begin = z1_305_begin_0, end = z1_305_end_0, end_mask = z1_305_end_mask_0, x = z_305_cast_fp16)[name = string("z1_305_cast_fp16")]; + tensor z2_305_begin_0 = const()[name = string("z2_305_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_305_end_0 = const()[name = string("z2_305_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_305_end_mask_0 = const()[name = string("z2_305_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_305_cast_fp16 = slice_by_index(begin = z2_305_begin_0, end = z2_305_end_0, end_mask = z2_305_end_mask_0, x = z_305_cast_fp16)[name = string("z2_305_cast_fp16")]; + tensor var_19150_cast_fp16 = mul(x = z_305_cast_fp16, y = cos_151_to_fp16)[name = string("op_19150_cast_fp16")]; + fp16 const_168_promoted_to_fp16 = const()[name = string("const_168_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19151_cast_fp16 = mul(x = z2_305_cast_fp16, y = const_168_promoted_to_fp16)[name = string("op_19151_cast_fp16")]; + bool var_19153_interleave_0 = const()[name = string("op_19153_interleave_0"), val = bool(false)]; + tensor var_19153_cast_fp16 = concat(axis = var_19059, interleave = var_19153_interleave_0, values = (var_19151_cast_fp16, z1_305_cast_fp16))[name = string("op_19153_cast_fp16")]; + tensor var_19154_cast_fp16 = mul(x = var_19153_cast_fp16, y = sin_151_to_fp16)[name = string("op_19154_cast_fp16")]; + tensor q_461_cast_fp16 = add(x = var_19150_cast_fp16, y = var_19154_cast_fp16)[name = string("q_461_cast_fp16")]; + tensor z1_307_begin_0 = const()[name = string("z1_307_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_307_end_0 = const()[name = string("z1_307_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_307_end_mask_0 = const()[name = string("z1_307_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_307_cast_fp16 = slice_by_index(begin = z1_307_begin_0, end = z1_307_end_0, end_mask = z1_307_end_mask_0, x = z_307_cast_fp16)[name = string("z1_307_cast_fp16")]; + tensor z2_307_begin_0 = const()[name = string("z2_307_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_307_end_0 = const()[name = string("z2_307_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_307_end_mask_0 = const()[name = string("z2_307_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_307_cast_fp16 = slice_by_index(begin = z2_307_begin_0, end = z2_307_end_0, end_mask = z2_307_end_mask_0, x = z_307_cast_fp16)[name = string("z2_307_cast_fp16")]; + tensor var_19162_cast_fp16 = mul(x = z_307_cast_fp16, y = cos_151_to_fp16)[name = string("op_19162_cast_fp16")]; + fp16 const_169_promoted_to_fp16 = const()[name = string("const_169_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19163_cast_fp16 = mul(x = z2_307_cast_fp16, y = const_169_promoted_to_fp16)[name = string("op_19163_cast_fp16")]; + bool var_19165_interleave_0 = const()[name = string("op_19165_interleave_0"), val = bool(false)]; + tensor var_19165_cast_fp16 = concat(axis = var_19059, interleave = var_19165_interleave_0, values = (var_19163_cast_fp16, z1_307_cast_fp16))[name = string("op_19165_cast_fp16")]; + tensor var_19166_cast_fp16 = mul(x = var_19165_cast_fp16, y = sin_151_to_fp16)[name = string("op_19166_cast_fp16")]; + tensor k_461_cast_fp16 = add(x = var_19162_cast_fp16, y = var_19166_cast_fp16)[name = string("k_461_cast_fp16")]; + tensor var_19168 = const()[name = string("op_19168"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_153_cast_fp16 = reshape(shape = var_19168, x = k_461_cast_fp16)[name = string("cur_key_153_cast_fp16")]; + tensor var_19170_to_fp16 = const()[name = string("op_19170_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647360)))]; + tensor var_19171_cast_fp16 = mul(x = key_cache_153_cast_fp16, y = var_19170_to_fp16)[name = string("op_19171_cast_fp16")]; + tensor upd_153_to_fp16 = const()[name = string("upd_153_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647488)))]; + tensor var_19172_cast_fp16 = mul(x = cur_key_153_cast_fp16, y = upd_153_to_fp16)[name = string("op_19172_cast_fp16")]; + tensor key_153_cast_fp16 = add(x = var_19171_cast_fp16, y = var_19172_cast_fp16)[name = string("key_153_cast_fp16")]; + tensor var_19174_to_fp16 = const()[name = string("op_19174_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647360)))]; + tensor var_19175_cast_fp16 = mul(x = value_cache_153_cast_fp16, y = var_19174_to_fp16)[name = string("op_19175_cast_fp16")]; + tensor var_19176_cast_fp16 = mul(x = v_153_cast_fp16, y = upd_153_to_fp16)[name = string("op_19176_cast_fp16")]; + tensor value_153_cast_fp16 = add(x = var_19175_cast_fp16, y = var_19176_cast_fp16)[name = string("value_153_cast_fp16")]; + tensor var_19178 = const()[name = string("op_19178"), val = tensor([1, 8, 128, 16])]; + tensor kh_305_cast_fp16 = reshape(shape = var_19178, x = key_153_cast_fp16)[name = string("kh_305_cast_fp16")]; + tensor var_19180 = const()[name = string("op_19180"), val = tensor([1, 8, 128, 16])]; + tensor vh_305_cast_fp16 = reshape(shape = var_19180, x = value_153_cast_fp16)[name = string("vh_305_cast_fp16")]; + tensor transpose_304_perm_0 = const()[name = string("transpose_304_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_152_reps_0 = const()[name = string("tile_152_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_304_cast_fp16 = transpose(perm = transpose_304_perm_0, x = kh_305_cast_fp16)[name = string("transpose_23")]; + tensor tile_152_cast_fp16 = tile(reps = tile_152_reps_0, x = transpose_304_cast_fp16)[name = string("tile_152_cast_fp16")]; + tensor concat_382 = const()[name = string("concat_382"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_304_cast_fp16 = reshape(shape = concat_382, x = tile_152_cast_fp16)[name = string("reshape_304_cast_fp16")]; + tensor transpose_305_perm_0 = const()[name = string("transpose_305_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_383 = const()[name = string("concat_383"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_305_cast_fp16 = transpose(perm = transpose_305_perm_0, x = reshape_304_cast_fp16)[name = string("transpose_22")]; + tensor reshape_305_cast_fp16 = reshape(shape = concat_383, x = transpose_305_cast_fp16)[name = string("reshape_305_cast_fp16")]; + tensor transpose_306_perm_0 = const()[name = string("transpose_306_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_153_reps_0 = const()[name = string("tile_153_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_306_cast_fp16 = transpose(perm = transpose_306_perm_0, x = vh_305_cast_fp16)[name = string("transpose_21")]; + tensor tile_153_cast_fp16 = tile(reps = tile_153_reps_0, x = transpose_306_cast_fp16)[name = string("tile_153_cast_fp16")]; + tensor concat_384 = const()[name = string("concat_384"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_306_cast_fp16 = reshape(shape = concat_384, x = tile_153_cast_fp16)[name = string("reshape_306_cast_fp16")]; + tensor transpose_307_perm_0 = const()[name = string("transpose_307_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_385 = const()[name = string("concat_385"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_307_cast_fp16 = transpose(perm = transpose_307_perm_0, x = reshape_306_cast_fp16)[name = string("transpose_20")]; + tensor reshape_307_cast_fp16 = reshape(shape = concat_385, x = transpose_307_cast_fp16)[name = string("reshape_307_cast_fp16")]; + fp16 var_19184_to_fp16 = const()[name = string("op_19184_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_19185_cast_fp16 = mul(x = q_461_cast_fp16, y = var_19184_to_fp16)[name = string("op_19185_cast_fp16")]; + tensor transpose_621_perm_0 = const()[name = string("transpose_621_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_333_transpose_x_1 = const()[name = string("w_333_transpose_x_1"), val = bool(true)]; + bool w_333_transpose_y_1 = const()[name = string("w_333_transpose_y_1"), val = bool(false)]; + tensor transpose_621_cast_fp16 = transpose(perm = transpose_621_perm_0, x = reshape_305_cast_fp16)[name = string("transpose_19")]; + tensor w_333_cast_fp16 = matmul(transpose_x = w_333_transpose_x_1, transpose_y = w_333_transpose_y_1, x = var_19185_cast_fp16, y = transpose_621_cast_fp16)[name = string("w_333_cast_fp16")]; + tensor w_335_cast_fp16 = softmax(axis = var_19063, x = w_333_cast_fp16)[name = string("w_335_cast_fp16")]; + tensor transpose_622_perm_0 = const()[name = string("transpose_622_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_153_transpose_x_1 = const()[name = string("attn_153_transpose_x_1"), val = bool(false)]; + bool attn_153_transpose_y_1 = const()[name = string("attn_153_transpose_y_1"), val = bool(true)]; + tensor transpose_622_cast_fp16 = transpose(perm = transpose_622_perm_0, x = reshape_307_cast_fp16)[name = string("transpose_18")]; + tensor attn_153_cast_fp16 = matmul(transpose_x = attn_153_transpose_x_1, transpose_y = attn_153_transpose_y_1, x = transpose_622_cast_fp16, y = w_335_cast_fp16)[name = string("attn_153_cast_fp16")]; + tensor var_19192 = const()[name = string("op_19192"), val = tensor([1, 2048, 1, 1])]; + tensor input_819_cast_fp16 = reshape(shape = var_19192, x = attn_153_cast_fp16)[name = string("input_819_cast_fp16")]; + string attn_output_153_pad_type_0 = const()[name = string("attn_output_153_pad_type_0"), val = string("valid")]; + tensor attn_output_153_strides_0 = const()[name = string("attn_output_153_strides_0"), val = tensor([1, 1])]; + tensor attn_output_153_pad_0 = const()[name = string("attn_output_153_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_153_dilations_0 = const()[name = string("attn_output_153_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_153_groups_0 = const()[name = string("attn_output_153_groups_0"), val = int32(1)]; + tensor attn_output_153_cast_fp16 = conv(dilations = attn_output_153_dilations_0, groups = attn_output_153_groups_0, pad = attn_output_153_pad_0, pad_type = attn_output_153_pad_type_0, strides = attn_output_153_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_819_cast_fp16)[name = string("attn_output_153_cast_fp16")]; + tensor x_583_cast_fp16 = add(x = x_577_cast_fp16, y = attn_output_153_cast_fp16)[name = string("x_583_cast_fp16")]; + tensor var_19206_cast_fp16 = mul(x = x_583_cast_fp16, y = x_583_cast_fp16)[name = string("op_19206_cast_fp16")]; + tensor variance_643_axes_0 = const()[name = string("variance_643_axes_0"), val = tensor([1])]; + bool variance_643_keep_dims_0 = const()[name = string("variance_643_keep_dims_0"), val = bool(true)]; + tensor variance_643_cast_fp16 = reduce_mean(axes = variance_643_axes_0, keep_dims = variance_643_keep_dims_0, x = var_19206_cast_fp16)[name = string("variance_643_cast_fp16")]; + fp16 var_19209_to_fp16 = const()[name = string("op_19209_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19210_cast_fp16 = add(x = variance_643_cast_fp16, y = var_19209_to_fp16)[name = string("op_19210_cast_fp16")]; + fp32 var_19211_epsilon_0 = const()[name = string("op_19211_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19211_cast_fp16 = rsqrt(epsilon = var_19211_epsilon_0, x = var_19210_cast_fp16)[name = string("op_19211_cast_fp16")]; + tensor var_19212_cast_fp16 = mul(x = x_583_cast_fp16, y = var_19211_cast_fp16)[name = string("op_19212_cast_fp16")]; + tensor input_821_cast_fp16 = mul(x = var_19212_cast_fp16, y = layers_1_post_attention_layernorm_weight_to_fp16)[name = string("input_821_cast_fp16")]; + string input_823_pad_type_0 = const()[name = string("input_823_pad_type_0"), val = string("valid")]; + tensor input_823_strides_0 = const()[name = string("input_823_strides_0"), val = tensor([1, 1])]; + tensor input_823_pad_0 = const()[name = string("input_823_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_823_dilations_0 = const()[name = string("input_823_dilations_0"), val = tensor([1, 1])]; + int32 input_823_groups_0 = const()[name = string("input_823_groups_0"), val = int32(1)]; + tensor input_823_cast_fp16 = conv(dilations = input_823_dilations_0, groups = input_823_groups_0, pad = input_823_pad_0, pad_type = input_823_pad_type_0, strides = input_823_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_821_cast_fp16)[name = string("input_823_cast_fp16")]; + tensor var_19220_cast_fp16 = silu(x = input_823_cast_fp16)[name = string("op_19220_cast_fp16")]; + string var_19226_pad_type_0 = const()[name = string("op_19226_pad_type_0"), val = string("valid")]; + tensor var_19226_strides_0 = const()[name = string("op_19226_strides_0"), val = tensor([1, 1])]; + tensor var_19226_pad_0 = const()[name = string("op_19226_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19226_dilations_0 = const()[name = string("op_19226_dilations_0"), val = tensor([1, 1])]; + int32 var_19226_groups_0 = const()[name = string("op_19226_groups_0"), val = int32(1)]; + tensor var_19226_cast_fp16 = conv(dilations = var_19226_dilations_0, groups = var_19226_groups_0, pad = var_19226_pad_0, pad_type = var_19226_pad_type_0, strides = var_19226_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_821_cast_fp16)[name = string("op_19226_cast_fp16")]; + tensor input_825_cast_fp16 = mul(x = var_19220_cast_fp16, y = var_19226_cast_fp16)[name = string("input_825_cast_fp16")]; + string h_153_pad_type_0 = const()[name = string("h_153_pad_type_0"), val = string("valid")]; + tensor h_153_strides_0 = const()[name = string("h_153_strides_0"), val = tensor([1, 1])]; + tensor h_153_pad_0 = const()[name = string("h_153_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_153_dilations_0 = const()[name = string("h_153_dilations_0"), val = tensor([1, 1])]; + int32 h_153_groups_0 = const()[name = string("h_153_groups_0"), val = int32(1)]; + tensor h_153_cast_fp16 = conv(dilations = h_153_dilations_0, groups = h_153_groups_0, pad = h_153_pad_0, pad_type = h_153_pad_type_0, strides = h_153_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_825_cast_fp16)[name = string("h_153_cast_fp16")]; + tensor x_585_cast_fp16 = add(x = x_583_cast_fp16, y = h_153_cast_fp16)[name = string("x_585_cast_fp16")]; + tensor key_cache_155_begin_0 = const()[name = string("key_cache_155_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor key_cache_155_end_0 = const()[name = string("key_cache_155_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor key_cache_155_end_mask_0 = const()[name = string("key_cache_155_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_155_cast_fp16 = slice_by_index(begin = key_cache_155_begin_0, end = key_cache_155_end_0, end_mask = key_cache_155_end_mask_0, x = layer_key_caches_cast_fp16)[name = string("key_cache_155_cast_fp16")]; + tensor value_cache_155_begin_0 = const()[name = string("value_cache_155_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor value_cache_155_end_0 = const()[name = string("value_cache_155_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor value_cache_155_end_mask_0 = const()[name = string("value_cache_155_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_155_cast_fp16 = slice_by_index(begin = value_cache_155_begin_0, end = value_cache_155_end_0, end_mask = value_cache_155_end_mask_0, x = layer_value_caches_cast_fp16)[name = string("value_cache_155_cast_fp16")]; + int32 var_19265 = const()[name = string("op_19265"), val = int32(2)]; + int32 var_19269 = const()[name = string("op_19269"), val = int32(3)]; + tensor var_19284_cast_fp16 = mul(x = x_585_cast_fp16, y = x_585_cast_fp16)[name = string("op_19284_cast_fp16")]; + tensor variance_645_axes_0 = const()[name = string("variance_645_axes_0"), val = tensor([1])]; + bool variance_645_keep_dims_0 = const()[name = string("variance_645_keep_dims_0"), val = bool(true)]; + tensor variance_645_cast_fp16 = reduce_mean(axes = variance_645_axes_0, keep_dims = variance_645_keep_dims_0, x = var_19284_cast_fp16)[name = string("variance_645_cast_fp16")]; + fp16 var_19287_to_fp16 = const()[name = string("op_19287_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19288_cast_fp16 = add(x = variance_645_cast_fp16, y = var_19287_to_fp16)[name = string("op_19288_cast_fp16")]; + fp32 var_19289_epsilon_0 = const()[name = string("op_19289_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19289_cast_fp16 = rsqrt(epsilon = var_19289_epsilon_0, x = var_19288_cast_fp16)[name = string("op_19289_cast_fp16")]; + tensor var_19290_cast_fp16 = mul(x = x_585_cast_fp16, y = var_19289_cast_fp16)[name = string("op_19290_cast_fp16")]; + tensor input_827_cast_fp16 = mul(x = var_19290_cast_fp16, y = layers_2_input_layernorm_weight_to_fp16)[name = string("input_827_cast_fp16")]; + string q_463_pad_type_0 = const()[name = string("q_463_pad_type_0"), val = string("valid")]; + tensor q_463_strides_0 = const()[name = string("q_463_strides_0"), val = tensor([1, 1])]; + tensor q_463_pad_0 = const()[name = string("q_463_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_463_dilations_0 = const()[name = string("q_463_dilations_0"), val = tensor([1, 1])]; + int32 q_463_groups_0 = const()[name = string("q_463_groups_0"), val = int32(1)]; + tensor q_463_cast_fp16 = conv(dilations = q_463_dilations_0, groups = q_463_groups_0, pad = q_463_pad_0, pad_type = q_463_pad_type_0, strides = q_463_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = input_827_cast_fp16)[name = string("q_463_cast_fp16")]; + string k_463_pad_type_0 = const()[name = string("k_463_pad_type_0"), val = string("valid")]; + tensor k_463_strides_0 = const()[name = string("k_463_strides_0"), val = tensor([1, 1])]; + tensor k_463_pad_0 = const()[name = string("k_463_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_463_dilations_0 = const()[name = string("k_463_dilations_0"), val = tensor([1, 1])]; + int32 k_463_groups_0 = const()[name = string("k_463_groups_0"), val = int32(1)]; + tensor k_463_cast_fp16 = conv(dilations = k_463_dilations_0, groups = k_463_groups_0, pad = k_463_pad_0, pad_type = k_463_pad_type_0, strides = k_463_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = input_827_cast_fp16)[name = string("k_463_cast_fp16")]; + string v_155_pad_type_0 = const()[name = string("v_155_pad_type_0"), val = string("valid")]; + tensor v_155_strides_0 = const()[name = string("v_155_strides_0"), val = tensor([1, 1])]; + tensor v_155_pad_0 = const()[name = string("v_155_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_155_dilations_0 = const()[name = string("v_155_dilations_0"), val = tensor([1, 1])]; + int32 v_155_groups_0 = const()[name = string("v_155_groups_0"), val = int32(1)]; + tensor v_155_cast_fp16 = conv(dilations = v_155_dilations_0, groups = v_155_groups_0, pad = v_155_pad_0, pad_type = v_155_pad_type_0, strides = v_155_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = input_827_cast_fp16)[name = string("v_155_cast_fp16")]; + tensor var_19324 = const()[name = string("op_19324"), val = tensor([16, 128, 1, 1])]; + tensor x_587_cast_fp16 = reshape(shape = var_19324, x = q_463_cast_fp16)[name = string("x_587_cast_fp16")]; + tensor var_19327_cast_fp16 = mul(x = x_587_cast_fp16, y = x_587_cast_fp16)[name = string("op_19327_cast_fp16")]; + tensor variance_647_axes_0 = const()[name = string("variance_647_axes_0"), val = tensor([1])]; + bool variance_647_keep_dims_0 = const()[name = string("variance_647_keep_dims_0"), val = bool(true)]; + tensor variance_647_cast_fp16 = reduce_mean(axes = variance_647_axes_0, keep_dims = variance_647_keep_dims_0, x = var_19327_cast_fp16)[name = string("variance_647_cast_fp16")]; + fp16 var_19330_to_fp16 = const()[name = string("op_19330_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19331_cast_fp16 = add(x = variance_647_cast_fp16, y = var_19330_to_fp16)[name = string("op_19331_cast_fp16")]; + fp32 var_19332_epsilon_0 = const()[name = string("op_19332_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19332_cast_fp16 = rsqrt(epsilon = var_19332_epsilon_0, x = var_19331_cast_fp16)[name = string("op_19332_cast_fp16")]; + tensor var_19333_cast_fp16 = mul(x = x_587_cast_fp16, y = var_19332_cast_fp16)[name = string("op_19333_cast_fp16")]; + tensor q_465_cast_fp16 = mul(x = var_19333_cast_fp16, y = layers_2_self_attn_q_norm_weight_to_fp16)[name = string("q_465_cast_fp16")]; + tensor var_19335 = const()[name = string("op_19335"), val = tensor([8, 128, 1, 1])]; + tensor x_589_cast_fp16 = reshape(shape = var_19335, x = k_463_cast_fp16)[name = string("x_589_cast_fp16")]; + tensor var_19338_cast_fp16 = mul(x = x_589_cast_fp16, y = x_589_cast_fp16)[name = string("op_19338_cast_fp16")]; + tensor variance_649_axes_0 = const()[name = string("variance_649_axes_0"), val = tensor([1])]; + bool variance_649_keep_dims_0 = const()[name = string("variance_649_keep_dims_0"), val = bool(true)]; + tensor variance_649_cast_fp16 = reduce_mean(axes = variance_649_axes_0, keep_dims = variance_649_keep_dims_0, x = var_19338_cast_fp16)[name = string("variance_649_cast_fp16")]; + fp16 var_19341_to_fp16 = const()[name = string("op_19341_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19342_cast_fp16 = add(x = variance_649_cast_fp16, y = var_19341_to_fp16)[name = string("op_19342_cast_fp16")]; + fp32 var_19343_epsilon_0 = const()[name = string("op_19343_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19343_cast_fp16 = rsqrt(epsilon = var_19343_epsilon_0, x = var_19342_cast_fp16)[name = string("op_19343_cast_fp16")]; + tensor var_19344_cast_fp16 = mul(x = x_589_cast_fp16, y = var_19343_cast_fp16)[name = string("op_19344_cast_fp16")]; + tensor k_465_cast_fp16 = mul(x = var_19344_cast_fp16, y = layers_2_self_attn_k_norm_weight_to_fp16)[name = string("k_465_cast_fp16")]; + tensor var_19346 = const()[name = string("op_19346"), val = tensor([1, 16, 128, 1])]; + tensor z_309_cast_fp16 = reshape(shape = var_19346, x = q_465_cast_fp16)[name = string("z_309_cast_fp16")]; + tensor var_19348 = const()[name = string("op_19348"), val = tensor([1, 8, 128, 1])]; + tensor z_311_cast_fp16 = reshape(shape = var_19348, x = k_465_cast_fp16)[name = string("z_311_cast_fp16")]; + tensor z1_309_begin_0 = const()[name = string("z1_309_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_309_end_0 = const()[name = string("z1_309_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_309_end_mask_0 = const()[name = string("z1_309_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_309_cast_fp16 = slice_by_index(begin = z1_309_begin_0, end = z1_309_end_0, end_mask = z1_309_end_mask_0, x = z_309_cast_fp16)[name = string("z1_309_cast_fp16")]; + tensor z2_309_begin_0 = const()[name = string("z2_309_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_309_end_0 = const()[name = string("z2_309_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_309_end_mask_0 = const()[name = string("z2_309_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_309_cast_fp16 = slice_by_index(begin = z2_309_begin_0, end = z2_309_end_0, end_mask = z2_309_end_mask_0, x = z_309_cast_fp16)[name = string("z2_309_cast_fp16")]; + tensor var_19356_cast_fp16 = mul(x = z_309_cast_fp16, y = cos_151_to_fp16)[name = string("op_19356_cast_fp16")]; + fp16 const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19357_cast_fp16 = mul(x = z2_309_cast_fp16, y = const_170_promoted_to_fp16)[name = string("op_19357_cast_fp16")]; + bool var_19359_interleave_0 = const()[name = string("op_19359_interleave_0"), val = bool(false)]; + tensor var_19359_cast_fp16 = concat(axis = var_19265, interleave = var_19359_interleave_0, values = (var_19357_cast_fp16, z1_309_cast_fp16))[name = string("op_19359_cast_fp16")]; + tensor var_19360_cast_fp16 = mul(x = var_19359_cast_fp16, y = sin_151_to_fp16)[name = string("op_19360_cast_fp16")]; + tensor q_467_cast_fp16 = add(x = var_19356_cast_fp16, y = var_19360_cast_fp16)[name = string("q_467_cast_fp16")]; + tensor z1_311_begin_0 = const()[name = string("z1_311_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_311_end_0 = const()[name = string("z1_311_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_311_end_mask_0 = const()[name = string("z1_311_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_311_cast_fp16 = slice_by_index(begin = z1_311_begin_0, end = z1_311_end_0, end_mask = z1_311_end_mask_0, x = z_311_cast_fp16)[name = string("z1_311_cast_fp16")]; + tensor z2_311_begin_0 = const()[name = string("z2_311_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_311_end_0 = const()[name = string("z2_311_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_311_end_mask_0 = const()[name = string("z2_311_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_311_cast_fp16 = slice_by_index(begin = z2_311_begin_0, end = z2_311_end_0, end_mask = z2_311_end_mask_0, x = z_311_cast_fp16)[name = string("z2_311_cast_fp16")]; + tensor var_19368_cast_fp16 = mul(x = z_311_cast_fp16, y = cos_151_to_fp16)[name = string("op_19368_cast_fp16")]; + fp16 const_171_promoted_to_fp16 = const()[name = string("const_171_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19369_cast_fp16 = mul(x = z2_311_cast_fp16, y = const_171_promoted_to_fp16)[name = string("op_19369_cast_fp16")]; + bool var_19371_interleave_0 = const()[name = string("op_19371_interleave_0"), val = bool(false)]; + tensor var_19371_cast_fp16 = concat(axis = var_19265, interleave = var_19371_interleave_0, values = (var_19369_cast_fp16, z1_311_cast_fp16))[name = string("op_19371_cast_fp16")]; + tensor var_19372_cast_fp16 = mul(x = var_19371_cast_fp16, y = sin_151_to_fp16)[name = string("op_19372_cast_fp16")]; + tensor k_467_cast_fp16 = add(x = var_19368_cast_fp16, y = var_19372_cast_fp16)[name = string("k_467_cast_fp16")]; + tensor var_19374 = const()[name = string("op_19374"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_155_cast_fp16 = reshape(shape = var_19374, x = k_467_cast_fp16)[name = string("cur_key_155_cast_fp16")]; + tensor var_19376_to_fp16 = const()[name = string("op_19376_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647360)))]; + tensor var_19377_cast_fp16 = mul(x = key_cache_155_cast_fp16, y = var_19376_to_fp16)[name = string("op_19377_cast_fp16")]; + tensor upd_155_to_fp16 = const()[name = string("upd_155_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647488)))]; + tensor var_19378_cast_fp16 = mul(x = cur_key_155_cast_fp16, y = upd_155_to_fp16)[name = string("op_19378_cast_fp16")]; + tensor key_155_cast_fp16 = add(x = var_19377_cast_fp16, y = var_19378_cast_fp16)[name = string("key_155_cast_fp16")]; + tensor var_19380_to_fp16 = const()[name = string("op_19380_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647360)))]; + tensor var_19381_cast_fp16 = mul(x = value_cache_155_cast_fp16, y = var_19380_to_fp16)[name = string("op_19381_cast_fp16")]; + tensor var_19382_cast_fp16 = mul(x = v_155_cast_fp16, y = upd_155_to_fp16)[name = string("op_19382_cast_fp16")]; + tensor value_155_cast_fp16 = add(x = var_19381_cast_fp16, y = var_19382_cast_fp16)[name = string("value_155_cast_fp16")]; + tensor var_19384 = const()[name = string("op_19384"), val = tensor([1, 8, 128, 16])]; + tensor kh_309_cast_fp16 = reshape(shape = var_19384, x = key_155_cast_fp16)[name = string("kh_309_cast_fp16")]; + tensor var_19386 = const()[name = string("op_19386"), val = tensor([1, 8, 128, 16])]; + tensor vh_309_cast_fp16 = reshape(shape = var_19386, x = value_155_cast_fp16)[name = string("vh_309_cast_fp16")]; + tensor transpose_308_perm_0 = const()[name = string("transpose_308_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_154_reps_0 = const()[name = string("tile_154_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_308_cast_fp16 = transpose(perm = transpose_308_perm_0, x = kh_309_cast_fp16)[name = string("transpose_17")]; + tensor tile_154_cast_fp16 = tile(reps = tile_154_reps_0, x = transpose_308_cast_fp16)[name = string("tile_154_cast_fp16")]; + tensor concat_386 = const()[name = string("concat_386"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_308_cast_fp16 = reshape(shape = concat_386, x = tile_154_cast_fp16)[name = string("reshape_308_cast_fp16")]; + tensor transpose_309_perm_0 = const()[name = string("transpose_309_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_387 = const()[name = string("concat_387"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_309_cast_fp16 = transpose(perm = transpose_309_perm_0, x = reshape_308_cast_fp16)[name = string("transpose_16")]; + tensor reshape_309_cast_fp16 = reshape(shape = concat_387, x = transpose_309_cast_fp16)[name = string("reshape_309_cast_fp16")]; + tensor transpose_310_perm_0 = const()[name = string("transpose_310_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_155_reps_0 = const()[name = string("tile_155_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_310_cast_fp16 = transpose(perm = transpose_310_perm_0, x = vh_309_cast_fp16)[name = string("transpose_15")]; + tensor tile_155_cast_fp16 = tile(reps = tile_155_reps_0, x = transpose_310_cast_fp16)[name = string("tile_155_cast_fp16")]; + tensor concat_388 = const()[name = string("concat_388"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_310_cast_fp16 = reshape(shape = concat_388, x = tile_155_cast_fp16)[name = string("reshape_310_cast_fp16")]; + tensor transpose_311_perm_0 = const()[name = string("transpose_311_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_389 = const()[name = string("concat_389"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_311_cast_fp16 = transpose(perm = transpose_311_perm_0, x = reshape_310_cast_fp16)[name = string("transpose_14")]; + tensor reshape_311_cast_fp16 = reshape(shape = concat_389, x = transpose_311_cast_fp16)[name = string("reshape_311_cast_fp16")]; + fp16 var_19390_to_fp16 = const()[name = string("op_19390_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_19391_cast_fp16 = mul(x = q_467_cast_fp16, y = var_19390_to_fp16)[name = string("op_19391_cast_fp16")]; + tensor transpose_625_perm_0 = const()[name = string("transpose_625_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_337_transpose_x_1 = const()[name = string("w_337_transpose_x_1"), val = bool(true)]; + bool w_337_transpose_y_1 = const()[name = string("w_337_transpose_y_1"), val = bool(false)]; + tensor transpose_625_cast_fp16 = transpose(perm = transpose_625_perm_0, x = reshape_309_cast_fp16)[name = string("transpose_13")]; + tensor w_337_cast_fp16 = matmul(transpose_x = w_337_transpose_x_1, transpose_y = w_337_transpose_y_1, x = var_19391_cast_fp16, y = transpose_625_cast_fp16)[name = string("w_337_cast_fp16")]; + tensor w_339_cast_fp16 = softmax(axis = var_19269, x = w_337_cast_fp16)[name = string("w_339_cast_fp16")]; + tensor transpose_626_perm_0 = const()[name = string("transpose_626_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_155_transpose_x_1 = const()[name = string("attn_155_transpose_x_1"), val = bool(false)]; + bool attn_155_transpose_y_1 = const()[name = string("attn_155_transpose_y_1"), val = bool(true)]; + tensor transpose_626_cast_fp16 = transpose(perm = transpose_626_perm_0, x = reshape_311_cast_fp16)[name = string("transpose_12")]; + tensor attn_155_cast_fp16 = matmul(transpose_x = attn_155_transpose_x_1, transpose_y = attn_155_transpose_y_1, x = transpose_626_cast_fp16, y = w_339_cast_fp16)[name = string("attn_155_cast_fp16")]; + tensor var_19398 = const()[name = string("op_19398"), val = tensor([1, 2048, 1, 1])]; + tensor input_829_cast_fp16 = reshape(shape = var_19398, x = attn_155_cast_fp16)[name = string("input_829_cast_fp16")]; + string attn_output_155_pad_type_0 = const()[name = string("attn_output_155_pad_type_0"), val = string("valid")]; + tensor attn_output_155_strides_0 = const()[name = string("attn_output_155_strides_0"), val = tensor([1, 1])]; + tensor attn_output_155_pad_0 = const()[name = string("attn_output_155_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_155_dilations_0 = const()[name = string("attn_output_155_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_155_groups_0 = const()[name = string("attn_output_155_groups_0"), val = int32(1)]; + tensor attn_output_155_cast_fp16 = conv(dilations = attn_output_155_dilations_0, groups = attn_output_155_groups_0, pad = attn_output_155_pad_0, pad_type = attn_output_155_pad_type_0, strides = attn_output_155_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_829_cast_fp16)[name = string("attn_output_155_cast_fp16")]; + tensor x_591_cast_fp16 = add(x = x_585_cast_fp16, y = attn_output_155_cast_fp16)[name = string("x_591_cast_fp16")]; + tensor var_19412_cast_fp16 = mul(x = x_591_cast_fp16, y = x_591_cast_fp16)[name = string("op_19412_cast_fp16")]; + tensor variance_651_axes_0 = const()[name = string("variance_651_axes_0"), val = tensor([1])]; + bool variance_651_keep_dims_0 = const()[name = string("variance_651_keep_dims_0"), val = bool(true)]; + tensor variance_651_cast_fp16 = reduce_mean(axes = variance_651_axes_0, keep_dims = variance_651_keep_dims_0, x = var_19412_cast_fp16)[name = string("variance_651_cast_fp16")]; + fp16 var_19415_to_fp16 = const()[name = string("op_19415_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19416_cast_fp16 = add(x = variance_651_cast_fp16, y = var_19415_to_fp16)[name = string("op_19416_cast_fp16")]; + fp32 var_19417_epsilon_0 = const()[name = string("op_19417_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19417_cast_fp16 = rsqrt(epsilon = var_19417_epsilon_0, x = var_19416_cast_fp16)[name = string("op_19417_cast_fp16")]; + tensor var_19418_cast_fp16 = mul(x = x_591_cast_fp16, y = var_19417_cast_fp16)[name = string("op_19418_cast_fp16")]; + tensor input_831_cast_fp16 = mul(x = var_19418_cast_fp16, y = layers_2_post_attention_layernorm_weight_to_fp16)[name = string("input_831_cast_fp16")]; + string input_833_pad_type_0 = const()[name = string("input_833_pad_type_0"), val = string("valid")]; + tensor input_833_strides_0 = const()[name = string("input_833_strides_0"), val = tensor([1, 1])]; + tensor input_833_pad_0 = const()[name = string("input_833_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_833_dilations_0 = const()[name = string("input_833_dilations_0"), val = tensor([1, 1])]; + int32 input_833_groups_0 = const()[name = string("input_833_groups_0"), val = int32(1)]; + tensor input_833_cast_fp16 = conv(dilations = input_833_dilations_0, groups = input_833_groups_0, pad = input_833_pad_0, pad_type = input_833_pad_type_0, strides = input_833_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_831_cast_fp16)[name = string("input_833_cast_fp16")]; + tensor var_19426_cast_fp16 = silu(x = input_833_cast_fp16)[name = string("op_19426_cast_fp16")]; + string var_19432_pad_type_0 = const()[name = string("op_19432_pad_type_0"), val = string("valid")]; + tensor var_19432_strides_0 = const()[name = string("op_19432_strides_0"), val = tensor([1, 1])]; + tensor var_19432_pad_0 = const()[name = string("op_19432_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19432_dilations_0 = const()[name = string("op_19432_dilations_0"), val = tensor([1, 1])]; + int32 var_19432_groups_0 = const()[name = string("op_19432_groups_0"), val = int32(1)]; + tensor var_19432_cast_fp16 = conv(dilations = var_19432_dilations_0, groups = var_19432_groups_0, pad = var_19432_pad_0, pad_type = var_19432_pad_type_0, strides = var_19432_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_831_cast_fp16)[name = string("op_19432_cast_fp16")]; + tensor input_835_cast_fp16 = mul(x = var_19426_cast_fp16, y = var_19432_cast_fp16)[name = string("input_835_cast_fp16")]; + string h_155_pad_type_0 = const()[name = string("h_155_pad_type_0"), val = string("valid")]; + tensor h_155_strides_0 = const()[name = string("h_155_strides_0"), val = tensor([1, 1])]; + tensor h_155_pad_0 = const()[name = string("h_155_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_155_dilations_0 = const()[name = string("h_155_dilations_0"), val = tensor([1, 1])]; + int32 h_155_groups_0 = const()[name = string("h_155_groups_0"), val = int32(1)]; + tensor h_155_cast_fp16 = conv(dilations = h_155_dilations_0, groups = h_155_groups_0, pad = h_155_pad_0, pad_type = h_155_pad_type_0, strides = h_155_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_835_cast_fp16)[name = string("h_155_cast_fp16")]; + tensor x_593_cast_fp16 = add(x = x_591_cast_fp16, y = h_155_cast_fp16)[name = string("x_593_cast_fp16")]; + tensor key_cache_157_begin_0 = const()[name = string("key_cache_157_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor key_cache_157_end_0 = const()[name = string("key_cache_157_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor key_cache_157_end_mask_0 = const()[name = string("key_cache_157_end_mask_0"), val = tensor([true, false, true, true])]; + tensor key_cache_157_cast_fp16 = slice_by_index(begin = key_cache_157_begin_0, end = key_cache_157_end_0, end_mask = key_cache_157_end_mask_0, x = layer_key_caches_cast_fp16)[name = string("key_cache_157_cast_fp16")]; + tensor value_cache_157_begin_0 = const()[name = string("value_cache_157_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor value_cache_157_end_0 = const()[name = string("value_cache_157_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor value_cache_157_end_mask_0 = const()[name = string("value_cache_157_end_mask_0"), val = tensor([true, false, true, true])]; + tensor value_cache_157_cast_fp16 = slice_by_index(begin = value_cache_157_begin_0, end = value_cache_157_end_0, end_mask = value_cache_157_end_mask_0, x = layer_value_caches_cast_fp16)[name = string("value_cache_157_cast_fp16")]; + int32 var_19471 = const()[name = string("op_19471"), val = int32(2)]; + int32 var_19475 = const()[name = string("op_19475"), val = int32(3)]; + tensor var_19490_cast_fp16 = mul(x = x_593_cast_fp16, y = x_593_cast_fp16)[name = string("op_19490_cast_fp16")]; + tensor variance_653_axes_0 = const()[name = string("variance_653_axes_0"), val = tensor([1])]; + bool variance_653_keep_dims_0 = const()[name = string("variance_653_keep_dims_0"), val = bool(true)]; + tensor variance_653_cast_fp16 = reduce_mean(axes = variance_653_axes_0, keep_dims = variance_653_keep_dims_0, x = var_19490_cast_fp16)[name = string("variance_653_cast_fp16")]; + fp16 var_19493_to_fp16 = const()[name = string("op_19493_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19494_cast_fp16 = add(x = variance_653_cast_fp16, y = var_19493_to_fp16)[name = string("op_19494_cast_fp16")]; + fp32 var_19495_epsilon_0 = const()[name = string("op_19495_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19495_cast_fp16 = rsqrt(epsilon = var_19495_epsilon_0, x = var_19494_cast_fp16)[name = string("op_19495_cast_fp16")]; + tensor var_19496_cast_fp16 = mul(x = x_593_cast_fp16, y = var_19495_cast_fp16)[name = string("op_19496_cast_fp16")]; + tensor input_837_cast_fp16 = mul(x = var_19496_cast_fp16, y = layers_3_input_layernorm_weight_to_fp16)[name = string("input_837_cast_fp16")]; + string q_469_pad_type_0 = const()[name = string("q_469_pad_type_0"), val = string("valid")]; + tensor q_469_strides_0 = const()[name = string("q_469_strides_0"), val = tensor([1, 1])]; + tensor q_469_pad_0 = const()[name = string("q_469_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_469_dilations_0 = const()[name = string("q_469_dilations_0"), val = tensor([1, 1])]; + int32 q_469_groups_0 = const()[name = string("q_469_groups_0"), val = int32(1)]; + tensor q_469_cast_fp16 = conv(dilations = q_469_dilations_0, groups = q_469_groups_0, pad = q_469_pad_0, pad_type = q_469_pad_type_0, strides = q_469_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = input_837_cast_fp16)[name = string("q_469_cast_fp16")]; + string k_469_pad_type_0 = const()[name = string("k_469_pad_type_0"), val = string("valid")]; + tensor k_469_strides_0 = const()[name = string("k_469_strides_0"), val = tensor([1, 1])]; + tensor k_469_pad_0 = const()[name = string("k_469_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_469_dilations_0 = const()[name = string("k_469_dilations_0"), val = tensor([1, 1])]; + int32 k_469_groups_0 = const()[name = string("k_469_groups_0"), val = int32(1)]; + tensor k_469_cast_fp16 = conv(dilations = k_469_dilations_0, groups = k_469_groups_0, pad = k_469_pad_0, pad_type = k_469_pad_type_0, strides = k_469_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = input_837_cast_fp16)[name = string("k_469_cast_fp16")]; + string v_157_pad_type_0 = const()[name = string("v_157_pad_type_0"), val = string("valid")]; + tensor v_157_strides_0 = const()[name = string("v_157_strides_0"), val = tensor([1, 1])]; + tensor v_157_pad_0 = const()[name = string("v_157_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_157_dilations_0 = const()[name = string("v_157_dilations_0"), val = tensor([1, 1])]; + int32 v_157_groups_0 = const()[name = string("v_157_groups_0"), val = int32(1)]; + tensor v_157_cast_fp16 = conv(dilations = v_157_dilations_0, groups = v_157_groups_0, pad = v_157_pad_0, pad_type = v_157_pad_type_0, strides = v_157_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = input_837_cast_fp16)[name = string("v_157_cast_fp16")]; + tensor var_19530 = const()[name = string("op_19530"), val = tensor([16, 128, 1, 1])]; + tensor x_595_cast_fp16 = reshape(shape = var_19530, x = q_469_cast_fp16)[name = string("x_595_cast_fp16")]; + tensor var_19533_cast_fp16 = mul(x = x_595_cast_fp16, y = x_595_cast_fp16)[name = string("op_19533_cast_fp16")]; + tensor variance_655_axes_0 = const()[name = string("variance_655_axes_0"), val = tensor([1])]; + bool variance_655_keep_dims_0 = const()[name = string("variance_655_keep_dims_0"), val = bool(true)]; + tensor variance_655_cast_fp16 = reduce_mean(axes = variance_655_axes_0, keep_dims = variance_655_keep_dims_0, x = var_19533_cast_fp16)[name = string("variance_655_cast_fp16")]; + fp16 var_19536_to_fp16 = const()[name = string("op_19536_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19537_cast_fp16 = add(x = variance_655_cast_fp16, y = var_19536_to_fp16)[name = string("op_19537_cast_fp16")]; + fp32 var_19538_epsilon_0 = const()[name = string("op_19538_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19538_cast_fp16 = rsqrt(epsilon = var_19538_epsilon_0, x = var_19537_cast_fp16)[name = string("op_19538_cast_fp16")]; + tensor var_19539_cast_fp16 = mul(x = x_595_cast_fp16, y = var_19538_cast_fp16)[name = string("op_19539_cast_fp16")]; + tensor q_471_cast_fp16 = mul(x = var_19539_cast_fp16, y = layers_3_self_attn_q_norm_weight_to_fp16)[name = string("q_471_cast_fp16")]; + tensor var_19541 = const()[name = string("op_19541"), val = tensor([8, 128, 1, 1])]; + tensor x_597_cast_fp16 = reshape(shape = var_19541, x = k_469_cast_fp16)[name = string("x_597_cast_fp16")]; + tensor var_19544_cast_fp16 = mul(x = x_597_cast_fp16, y = x_597_cast_fp16)[name = string("op_19544_cast_fp16")]; + tensor variance_657_axes_0 = const()[name = string("variance_657_axes_0"), val = tensor([1])]; + bool variance_657_keep_dims_0 = const()[name = string("variance_657_keep_dims_0"), val = bool(true)]; + tensor variance_657_cast_fp16 = reduce_mean(axes = variance_657_axes_0, keep_dims = variance_657_keep_dims_0, x = var_19544_cast_fp16)[name = string("variance_657_cast_fp16")]; + fp16 var_19547_to_fp16 = const()[name = string("op_19547_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19548_cast_fp16 = add(x = variance_657_cast_fp16, y = var_19547_to_fp16)[name = string("op_19548_cast_fp16")]; + fp32 var_19549_epsilon_0 = const()[name = string("op_19549_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19549_cast_fp16 = rsqrt(epsilon = var_19549_epsilon_0, x = var_19548_cast_fp16)[name = string("op_19549_cast_fp16")]; + tensor var_19550_cast_fp16 = mul(x = x_597_cast_fp16, y = var_19549_cast_fp16)[name = string("op_19550_cast_fp16")]; + tensor k_471_cast_fp16 = mul(x = var_19550_cast_fp16, y = layers_3_self_attn_k_norm_weight_to_fp16)[name = string("k_471_cast_fp16")]; + tensor var_19552 = const()[name = string("op_19552"), val = tensor([1, 16, 128, 1])]; + tensor z_313_cast_fp16 = reshape(shape = var_19552, x = q_471_cast_fp16)[name = string("z_313_cast_fp16")]; + tensor var_19554 = const()[name = string("op_19554"), val = tensor([1, 8, 128, 1])]; + tensor z_315_cast_fp16 = reshape(shape = var_19554, x = k_471_cast_fp16)[name = string("z_315_cast_fp16")]; + tensor z1_313_begin_0 = const()[name = string("z1_313_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_313_end_0 = const()[name = string("z1_313_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_313_end_mask_0 = const()[name = string("z1_313_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_313_cast_fp16 = slice_by_index(begin = z1_313_begin_0, end = z1_313_end_0, end_mask = z1_313_end_mask_0, x = z_313_cast_fp16)[name = string("z1_313_cast_fp16")]; + tensor z2_313_begin_0 = const()[name = string("z2_313_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_313_end_0 = const()[name = string("z2_313_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_313_end_mask_0 = const()[name = string("z2_313_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_313_cast_fp16 = slice_by_index(begin = z2_313_begin_0, end = z2_313_end_0, end_mask = z2_313_end_mask_0, x = z_313_cast_fp16)[name = string("z2_313_cast_fp16")]; + tensor var_19562_cast_fp16 = mul(x = z_313_cast_fp16, y = cos_151_to_fp16)[name = string("op_19562_cast_fp16")]; + fp16 const_172_promoted_to_fp16 = const()[name = string("const_172_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19563_cast_fp16 = mul(x = z2_313_cast_fp16, y = const_172_promoted_to_fp16)[name = string("op_19563_cast_fp16")]; + bool var_19565_interleave_0 = const()[name = string("op_19565_interleave_0"), val = bool(false)]; + tensor var_19565_cast_fp16 = concat(axis = var_19471, interleave = var_19565_interleave_0, values = (var_19563_cast_fp16, z1_313_cast_fp16))[name = string("op_19565_cast_fp16")]; + tensor var_19566_cast_fp16 = mul(x = var_19565_cast_fp16, y = sin_151_to_fp16)[name = string("op_19566_cast_fp16")]; + tensor q_473_cast_fp16 = add(x = var_19562_cast_fp16, y = var_19566_cast_fp16)[name = string("q_473_cast_fp16")]; + tensor z1_315_begin_0 = const()[name = string("z1_315_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_315_end_0 = const()[name = string("z1_315_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_315_end_mask_0 = const()[name = string("z1_315_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_315_cast_fp16 = slice_by_index(begin = z1_315_begin_0, end = z1_315_end_0, end_mask = z1_315_end_mask_0, x = z_315_cast_fp16)[name = string("z1_315_cast_fp16")]; + tensor z2_315_begin_0 = const()[name = string("z2_315_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_315_end_0 = const()[name = string("z2_315_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_315_end_mask_0 = const()[name = string("z2_315_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_315_cast_fp16 = slice_by_index(begin = z2_315_begin_0, end = z2_315_end_0, end_mask = z2_315_end_mask_0, x = z_315_cast_fp16)[name = string("z2_315_cast_fp16")]; + tensor var_19574_cast_fp16 = mul(x = z_315_cast_fp16, y = cos_151_to_fp16)[name = string("op_19574_cast_fp16")]; + fp16 const_173_promoted_to_fp16 = const()[name = string("const_173_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19575_cast_fp16 = mul(x = z2_315_cast_fp16, y = const_173_promoted_to_fp16)[name = string("op_19575_cast_fp16")]; + bool var_19577_interleave_0 = const()[name = string("op_19577_interleave_0"), val = bool(false)]; + tensor var_19577_cast_fp16 = concat(axis = var_19471, interleave = var_19577_interleave_0, values = (var_19575_cast_fp16, z1_315_cast_fp16))[name = string("op_19577_cast_fp16")]; + tensor var_19578_cast_fp16 = mul(x = var_19577_cast_fp16, y = sin_151_to_fp16)[name = string("op_19578_cast_fp16")]; + tensor k_473_cast_fp16 = add(x = var_19574_cast_fp16, y = var_19578_cast_fp16)[name = string("k_473_cast_fp16")]; + tensor var_19580 = const()[name = string("op_19580"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_157_cast_fp16 = reshape(shape = var_19580, x = k_473_cast_fp16)[name = string("cur_key_157_cast_fp16")]; + tensor var_19582_to_fp16 = const()[name = string("op_19582_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647360)))]; + tensor var_19583_cast_fp16 = mul(x = key_cache_157_cast_fp16, y = var_19582_to_fp16)[name = string("op_19583_cast_fp16")]; + tensor upd_157_to_fp16 = const()[name = string("upd_157_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647488)))]; + tensor var_19584_cast_fp16 = mul(x = cur_key_157_cast_fp16, y = upd_157_to_fp16)[name = string("op_19584_cast_fp16")]; + tensor key_157_cast_fp16 = add(x = var_19583_cast_fp16, y = var_19584_cast_fp16)[name = string("key_157_cast_fp16")]; + tensor var_19586_to_fp16 = const()[name = string("op_19586_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647360)))]; + tensor var_19587_cast_fp16 = mul(x = value_cache_157_cast_fp16, y = var_19586_to_fp16)[name = string("op_19587_cast_fp16")]; + tensor var_19588_cast_fp16 = mul(x = v_157_cast_fp16, y = upd_157_to_fp16)[name = string("op_19588_cast_fp16")]; + tensor value_157_cast_fp16 = add(x = var_19587_cast_fp16, y = var_19588_cast_fp16)[name = string("value_157_cast_fp16")]; + tensor var_19590 = const()[name = string("op_19590"), val = tensor([1, 8, 128, 16])]; + tensor kh_313_cast_fp16 = reshape(shape = var_19590, x = key_157_cast_fp16)[name = string("kh_313_cast_fp16")]; + tensor var_19592 = const()[name = string("op_19592"), val = tensor([1, 8, 128, 16])]; + tensor vh_313_cast_fp16 = reshape(shape = var_19592, x = value_157_cast_fp16)[name = string("vh_313_cast_fp16")]; + tensor transpose_312_perm_0 = const()[name = string("transpose_312_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_156_reps_0 = const()[name = string("tile_156_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_312_cast_fp16 = transpose(perm = transpose_312_perm_0, x = kh_313_cast_fp16)[name = string("transpose_11")]; + tensor tile_156_cast_fp16 = tile(reps = tile_156_reps_0, x = transpose_312_cast_fp16)[name = string("tile_156_cast_fp16")]; + tensor concat_390 = const()[name = string("concat_390"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_312_cast_fp16 = reshape(shape = concat_390, x = tile_156_cast_fp16)[name = string("reshape_312_cast_fp16")]; + tensor transpose_313_perm_0 = const()[name = string("transpose_313_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_391 = const()[name = string("concat_391"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_313_cast_fp16 = transpose(perm = transpose_313_perm_0, x = reshape_312_cast_fp16)[name = string("transpose_10")]; + tensor reshape_313_cast_fp16 = reshape(shape = concat_391, x = transpose_313_cast_fp16)[name = string("reshape_313_cast_fp16")]; + tensor transpose_314_perm_0 = const()[name = string("transpose_314_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_157_reps_0 = const()[name = string("tile_157_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_314_cast_fp16 = transpose(perm = transpose_314_perm_0, x = vh_313_cast_fp16)[name = string("transpose_9")]; + tensor tile_157_cast_fp16 = tile(reps = tile_157_reps_0, x = transpose_314_cast_fp16)[name = string("tile_157_cast_fp16")]; + tensor concat_392 = const()[name = string("concat_392"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_314_cast_fp16 = reshape(shape = concat_392, x = tile_157_cast_fp16)[name = string("reshape_314_cast_fp16")]; + tensor transpose_315_perm_0 = const()[name = string("transpose_315_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_393 = const()[name = string("concat_393"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_315_cast_fp16 = transpose(perm = transpose_315_perm_0, x = reshape_314_cast_fp16)[name = string("transpose_8")]; + tensor reshape_315_cast_fp16 = reshape(shape = concat_393, x = transpose_315_cast_fp16)[name = string("reshape_315_cast_fp16")]; + fp16 var_19596_to_fp16 = const()[name = string("op_19596_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_19597_cast_fp16 = mul(x = q_473_cast_fp16, y = var_19596_to_fp16)[name = string("op_19597_cast_fp16")]; + tensor transpose_629_perm_0 = const()[name = string("transpose_629_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_341_transpose_x_1 = const()[name = string("w_341_transpose_x_1"), val = bool(true)]; + bool w_341_transpose_y_1 = const()[name = string("w_341_transpose_y_1"), val = bool(false)]; + tensor transpose_629_cast_fp16 = transpose(perm = transpose_629_perm_0, x = reshape_313_cast_fp16)[name = string("transpose_7")]; + tensor w_341_cast_fp16 = matmul(transpose_x = w_341_transpose_x_1, transpose_y = w_341_transpose_y_1, x = var_19597_cast_fp16, y = transpose_629_cast_fp16)[name = string("w_341_cast_fp16")]; + tensor w_343_cast_fp16 = softmax(axis = var_19475, x = w_341_cast_fp16)[name = string("w_343_cast_fp16")]; + tensor transpose_630_perm_0 = const()[name = string("transpose_630_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_157_transpose_x_1 = const()[name = string("attn_157_transpose_x_1"), val = bool(false)]; + bool attn_157_transpose_y_1 = const()[name = string("attn_157_transpose_y_1"), val = bool(true)]; + tensor transpose_630_cast_fp16 = transpose(perm = transpose_630_perm_0, x = reshape_315_cast_fp16)[name = string("transpose_6")]; + tensor attn_157_cast_fp16 = matmul(transpose_x = attn_157_transpose_x_1, transpose_y = attn_157_transpose_y_1, x = transpose_630_cast_fp16, y = w_343_cast_fp16)[name = string("attn_157_cast_fp16")]; + tensor var_19604 = const()[name = string("op_19604"), val = tensor([1, 2048, 1, 1])]; + tensor input_839_cast_fp16 = reshape(shape = var_19604, x = attn_157_cast_fp16)[name = string("input_839_cast_fp16")]; + string attn_output_157_pad_type_0 = const()[name = string("attn_output_157_pad_type_0"), val = string("valid")]; + tensor attn_output_157_strides_0 = const()[name = string("attn_output_157_strides_0"), val = tensor([1, 1])]; + tensor attn_output_157_pad_0 = const()[name = string("attn_output_157_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_157_dilations_0 = const()[name = string("attn_output_157_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_157_groups_0 = const()[name = string("attn_output_157_groups_0"), val = int32(1)]; + tensor attn_output_157_cast_fp16 = conv(dilations = attn_output_157_dilations_0, groups = attn_output_157_groups_0, pad = attn_output_157_pad_0, pad_type = attn_output_157_pad_type_0, strides = attn_output_157_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_839_cast_fp16)[name = string("attn_output_157_cast_fp16")]; + tensor x_599_cast_fp16 = add(x = x_593_cast_fp16, y = attn_output_157_cast_fp16)[name = string("x_599_cast_fp16")]; + tensor var_19618_cast_fp16 = mul(x = x_599_cast_fp16, y = x_599_cast_fp16)[name = string("op_19618_cast_fp16")]; + tensor variance_659_axes_0 = const()[name = string("variance_659_axes_0"), val = tensor([1])]; + bool variance_659_keep_dims_0 = const()[name = string("variance_659_keep_dims_0"), val = bool(true)]; + tensor variance_659_cast_fp16 = reduce_mean(axes = variance_659_axes_0, keep_dims = variance_659_keep_dims_0, x = var_19618_cast_fp16)[name = string("variance_659_cast_fp16")]; + fp16 var_19621_to_fp16 = const()[name = string("op_19621_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19622_cast_fp16 = add(x = variance_659_cast_fp16, y = var_19621_to_fp16)[name = string("op_19622_cast_fp16")]; + fp32 var_19623_epsilon_0 = const()[name = string("op_19623_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19623_cast_fp16 = rsqrt(epsilon = var_19623_epsilon_0, x = var_19622_cast_fp16)[name = string("op_19623_cast_fp16")]; + tensor var_19624_cast_fp16 = mul(x = x_599_cast_fp16, y = var_19623_cast_fp16)[name = string("op_19624_cast_fp16")]; + tensor input_841_cast_fp16 = mul(x = var_19624_cast_fp16, y = layers_3_post_attention_layernorm_weight_to_fp16)[name = string("input_841_cast_fp16")]; + string input_843_pad_type_0 = const()[name = string("input_843_pad_type_0"), val = string("valid")]; + tensor input_843_strides_0 = const()[name = string("input_843_strides_0"), val = tensor([1, 1])]; + tensor input_843_pad_0 = const()[name = string("input_843_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_843_dilations_0 = const()[name = string("input_843_dilations_0"), val = tensor([1, 1])]; + int32 input_843_groups_0 = const()[name = string("input_843_groups_0"), val = int32(1)]; + tensor input_843_cast_fp16 = conv(dilations = input_843_dilations_0, groups = input_843_groups_0, pad = input_843_pad_0, pad_type = input_843_pad_type_0, strides = input_843_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_841_cast_fp16)[name = string("input_843_cast_fp16")]; + tensor var_19632_cast_fp16 = silu(x = input_843_cast_fp16)[name = string("op_19632_cast_fp16")]; + string var_19638_pad_type_0 = const()[name = string("op_19638_pad_type_0"), val = string("valid")]; + tensor var_19638_strides_0 = const()[name = string("op_19638_strides_0"), val = tensor([1, 1])]; + tensor var_19638_pad_0 = const()[name = string("op_19638_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19638_dilations_0 = const()[name = string("op_19638_dilations_0"), val = tensor([1, 1])]; + int32 var_19638_groups_0 = const()[name = string("op_19638_groups_0"), val = int32(1)]; + tensor var_19638_cast_fp16 = conv(dilations = var_19638_dilations_0, groups = var_19638_groups_0, pad = var_19638_pad_0, pad_type = var_19638_pad_type_0, strides = var_19638_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_841_cast_fp16)[name = string("op_19638_cast_fp16")]; + tensor input_845_cast_fp16 = mul(x = var_19632_cast_fp16, y = var_19638_cast_fp16)[name = string("input_845_cast_fp16")]; + string h_157_pad_type_0 = const()[name = string("h_157_pad_type_0"), val = string("valid")]; + tensor h_157_strides_0 = const()[name = string("h_157_strides_0"), val = tensor([1, 1])]; + tensor h_157_pad_0 = const()[name = string("h_157_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_157_dilations_0 = const()[name = string("h_157_dilations_0"), val = tensor([1, 1])]; + int32 h_157_groups_0 = const()[name = string("h_157_groups_0"), val = int32(1)]; + tensor h_157_cast_fp16 = conv(dilations = h_157_dilations_0, groups = h_157_groups_0, pad = h_157_pad_0, pad_type = h_157_pad_type_0, strides = h_157_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_845_cast_fp16)[name = string("h_157_cast_fp16")]; + tensor x_601_cast_fp16 = add(x = x_599_cast_fp16, y = h_157_cast_fp16)[name = string("x_601_cast_fp16")]; + tensor key_cache_begin_0 = const()[name = string("key_cache_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor key_cache_end_0 = const()[name = string("key_cache_end_0"), val = tensor([1, 1, 1, 16])]; + tensor key_cache_end_mask_0 = const()[name = string("key_cache_end_mask_0"), val = tensor([true, true, true, true])]; + tensor key_cache_cast_fp16 = slice_by_index(begin = key_cache_begin_0, end = key_cache_end_0, end_mask = key_cache_end_mask_0, x = layer_key_caches_cast_fp16)[name = string("key_cache_cast_fp16")]; + tensor value_cache_begin_0 = const()[name = string("value_cache_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor value_cache_end_0 = const()[name = string("value_cache_end_0"), val = tensor([1, 1, 1, 16])]; + tensor value_cache_end_mask_0 = const()[name = string("value_cache_end_mask_0"), val = tensor([true, true, true, true])]; + tensor value_cache_cast_fp16 = slice_by_index(begin = value_cache_begin_0, end = value_cache_end_0, end_mask = value_cache_end_mask_0, x = layer_value_caches_cast_fp16)[name = string("value_cache_cast_fp16")]; + int32 var_19677 = const()[name = string("op_19677"), val = int32(2)]; + int32 var_19681 = const()[name = string("op_19681"), val = int32(3)]; + tensor var_19696_cast_fp16 = mul(x = x_601_cast_fp16, y = x_601_cast_fp16)[name = string("op_19696_cast_fp16")]; + tensor variance_661_axes_0 = const()[name = string("variance_661_axes_0"), val = tensor([1])]; + bool variance_661_keep_dims_0 = const()[name = string("variance_661_keep_dims_0"), val = bool(true)]; + tensor variance_661_cast_fp16 = reduce_mean(axes = variance_661_axes_0, keep_dims = variance_661_keep_dims_0, x = var_19696_cast_fp16)[name = string("variance_661_cast_fp16")]; + fp16 var_19699_to_fp16 = const()[name = string("op_19699_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19700_cast_fp16 = add(x = variance_661_cast_fp16, y = var_19699_to_fp16)[name = string("op_19700_cast_fp16")]; + fp32 var_19701_epsilon_0 = const()[name = string("op_19701_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19701_cast_fp16 = rsqrt(epsilon = var_19701_epsilon_0, x = var_19700_cast_fp16)[name = string("op_19701_cast_fp16")]; + tensor var_19702_cast_fp16 = mul(x = x_601_cast_fp16, y = var_19701_cast_fp16)[name = string("op_19702_cast_fp16")]; + tensor input_847_cast_fp16 = mul(x = var_19702_cast_fp16, y = layers_4_input_layernorm_weight_to_fp16)[name = string("input_847_cast_fp16")]; + string q_475_pad_type_0 = const()[name = string("q_475_pad_type_0"), val = string("valid")]; + tensor q_475_strides_0 = const()[name = string("q_475_strides_0"), val = tensor([1, 1])]; + tensor q_475_pad_0 = const()[name = string("q_475_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor q_475_dilations_0 = const()[name = string("q_475_dilations_0"), val = tensor([1, 1])]; + int32 q_475_groups_0 = const()[name = string("q_475_groups_0"), val = int32(1)]; + tensor q_475_cast_fp16 = conv(dilations = q_475_dilations_0, groups = q_475_groups_0, pad = q_475_pad_0, pad_type = q_475_pad_type_0, strides = q_475_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = input_847_cast_fp16)[name = string("q_475_cast_fp16")]; + string k_475_pad_type_0 = const()[name = string("k_475_pad_type_0"), val = string("valid")]; + tensor k_475_strides_0 = const()[name = string("k_475_strides_0"), val = tensor([1, 1])]; + tensor k_475_pad_0 = const()[name = string("k_475_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor k_475_dilations_0 = const()[name = string("k_475_dilations_0"), val = tensor([1, 1])]; + int32 k_475_groups_0 = const()[name = string("k_475_groups_0"), val = int32(1)]; + tensor k_475_cast_fp16 = conv(dilations = k_475_dilations_0, groups = k_475_groups_0, pad = k_475_pad_0, pad_type = k_475_pad_type_0, strides = k_475_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = input_847_cast_fp16)[name = string("k_475_cast_fp16")]; + string v_pad_type_0 = const()[name = string("v_pad_type_0"), val = string("valid")]; + tensor v_strides_0 = const()[name = string("v_strides_0"), val = tensor([1, 1])]; + tensor v_pad_0 = const()[name = string("v_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor v_dilations_0 = const()[name = string("v_dilations_0"), val = tensor([1, 1])]; + int32 v_groups_0 = const()[name = string("v_groups_0"), val = int32(1)]; + tensor v_cast_fp16 = conv(dilations = v_dilations_0, groups = v_groups_0, pad = v_pad_0, pad_type = v_pad_type_0, strides = v_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = input_847_cast_fp16)[name = string("v_cast_fp16")]; + tensor var_19736 = const()[name = string("op_19736"), val = tensor([16, 128, 1, 1])]; + tensor x_603_cast_fp16 = reshape(shape = var_19736, x = q_475_cast_fp16)[name = string("x_603_cast_fp16")]; + tensor var_19739_cast_fp16 = mul(x = x_603_cast_fp16, y = x_603_cast_fp16)[name = string("op_19739_cast_fp16")]; + tensor variance_663_axes_0 = const()[name = string("variance_663_axes_0"), val = tensor([1])]; + bool variance_663_keep_dims_0 = const()[name = string("variance_663_keep_dims_0"), val = bool(true)]; + tensor variance_663_cast_fp16 = reduce_mean(axes = variance_663_axes_0, keep_dims = variance_663_keep_dims_0, x = var_19739_cast_fp16)[name = string("variance_663_cast_fp16")]; + fp16 var_19742_to_fp16 = const()[name = string("op_19742_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19743_cast_fp16 = add(x = variance_663_cast_fp16, y = var_19742_to_fp16)[name = string("op_19743_cast_fp16")]; + fp32 var_19744_epsilon_0 = const()[name = string("op_19744_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19744_cast_fp16 = rsqrt(epsilon = var_19744_epsilon_0, x = var_19743_cast_fp16)[name = string("op_19744_cast_fp16")]; + tensor var_19745_cast_fp16 = mul(x = x_603_cast_fp16, y = var_19744_cast_fp16)[name = string("op_19745_cast_fp16")]; + tensor q_477_cast_fp16 = mul(x = var_19745_cast_fp16, y = layers_4_self_attn_q_norm_weight_to_fp16)[name = string("q_477_cast_fp16")]; + tensor var_19747 = const()[name = string("op_19747"), val = tensor([8, 128, 1, 1])]; + tensor x_605_cast_fp16 = reshape(shape = var_19747, x = k_475_cast_fp16)[name = string("x_605_cast_fp16")]; + tensor var_19750_cast_fp16 = mul(x = x_605_cast_fp16, y = x_605_cast_fp16)[name = string("op_19750_cast_fp16")]; + tensor variance_665_axes_0 = const()[name = string("variance_665_axes_0"), val = tensor([1])]; + bool variance_665_keep_dims_0 = const()[name = string("variance_665_keep_dims_0"), val = bool(true)]; + tensor variance_665_cast_fp16 = reduce_mean(axes = variance_665_axes_0, keep_dims = variance_665_keep_dims_0, x = var_19750_cast_fp16)[name = string("variance_665_cast_fp16")]; + fp16 var_19753_to_fp16 = const()[name = string("op_19753_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19754_cast_fp16 = add(x = variance_665_cast_fp16, y = var_19753_to_fp16)[name = string("op_19754_cast_fp16")]; + fp32 var_19755_epsilon_0 = const()[name = string("op_19755_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19755_cast_fp16 = rsqrt(epsilon = var_19755_epsilon_0, x = var_19754_cast_fp16)[name = string("op_19755_cast_fp16")]; + tensor var_19756_cast_fp16 = mul(x = x_605_cast_fp16, y = var_19755_cast_fp16)[name = string("op_19756_cast_fp16")]; + tensor k_477_cast_fp16 = mul(x = var_19756_cast_fp16, y = layers_4_self_attn_k_norm_weight_to_fp16)[name = string("k_477_cast_fp16")]; + tensor var_19758 = const()[name = string("op_19758"), val = tensor([1, 16, 128, 1])]; + tensor z_317_cast_fp16 = reshape(shape = var_19758, x = q_477_cast_fp16)[name = string("z_317_cast_fp16")]; + tensor var_19760 = const()[name = string("op_19760"), val = tensor([1, 8, 128, 1])]; + tensor z_cast_fp16 = reshape(shape = var_19760, x = k_477_cast_fp16)[name = string("z_cast_fp16")]; + tensor z1_317_begin_0 = const()[name = string("z1_317_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_317_end_0 = const()[name = string("z1_317_end_0"), val = tensor([1, 16, 64, 1])]; + tensor z1_317_end_mask_0 = const()[name = string("z1_317_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_317_cast_fp16 = slice_by_index(begin = z1_317_begin_0, end = z1_317_end_0, end_mask = z1_317_end_mask_0, x = z_317_cast_fp16)[name = string("z1_317_cast_fp16")]; + tensor z2_317_begin_0 = const()[name = string("z2_317_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_317_end_0 = const()[name = string("z2_317_end_0"), val = tensor([1, 16, 128, 1])]; + tensor z2_317_end_mask_0 = const()[name = string("z2_317_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_317_cast_fp16 = slice_by_index(begin = z2_317_begin_0, end = z2_317_end_0, end_mask = z2_317_end_mask_0, x = z_317_cast_fp16)[name = string("z2_317_cast_fp16")]; + tensor var_19768_cast_fp16 = mul(x = z_317_cast_fp16, y = cos_151_to_fp16)[name = string("op_19768_cast_fp16")]; + fp16 const_174_promoted_to_fp16 = const()[name = string("const_174_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19769_cast_fp16 = mul(x = z2_317_cast_fp16, y = const_174_promoted_to_fp16)[name = string("op_19769_cast_fp16")]; + bool var_19771_interleave_0 = const()[name = string("op_19771_interleave_0"), val = bool(false)]; + tensor var_19771_cast_fp16 = concat(axis = var_19677, interleave = var_19771_interleave_0, values = (var_19769_cast_fp16, z1_317_cast_fp16))[name = string("op_19771_cast_fp16")]; + tensor var_19772_cast_fp16 = mul(x = var_19771_cast_fp16, y = sin_151_to_fp16)[name = string("op_19772_cast_fp16")]; + tensor q_cast_fp16 = add(x = var_19768_cast_fp16, y = var_19772_cast_fp16)[name = string("q_cast_fp16")]; + tensor z1_begin_0 = const()[name = string("z1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor z1_end_0 = const()[name = string("z1_end_0"), val = tensor([1, 8, 64, 1])]; + tensor z1_end_mask_0 = const()[name = string("z1_end_mask_0"), val = tensor([true, true, false, true])]; + tensor z1_cast_fp16 = slice_by_index(begin = z1_begin_0, end = z1_end_0, end_mask = z1_end_mask_0, x = z_cast_fp16)[name = string("z1_cast_fp16")]; + tensor z2_begin_0 = const()[name = string("z2_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor z2_end_0 = const()[name = string("z2_end_0"), val = tensor([1, 8, 128, 1])]; + tensor z2_end_mask_0 = const()[name = string("z2_end_mask_0"), val = tensor([true, true, true, true])]; + tensor z2_cast_fp16 = slice_by_index(begin = z2_begin_0, end = z2_end_0, end_mask = z2_end_mask_0, x = z_cast_fp16)[name = string("z2_cast_fp16")]; + tensor var_19780_cast_fp16 = mul(x = z_cast_fp16, y = cos_151_to_fp16)[name = string("op_19780_cast_fp16")]; + fp16 const_175_promoted_to_fp16 = const()[name = string("const_175_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19781_cast_fp16 = mul(x = z2_cast_fp16, y = const_175_promoted_to_fp16)[name = string("op_19781_cast_fp16")]; + bool var_19783_interleave_0 = const()[name = string("op_19783_interleave_0"), val = bool(false)]; + tensor var_19783_cast_fp16 = concat(axis = var_19677, interleave = var_19783_interleave_0, values = (var_19781_cast_fp16, z1_cast_fp16))[name = string("op_19783_cast_fp16")]; + tensor var_19784_cast_fp16 = mul(x = var_19783_cast_fp16, y = sin_151_to_fp16)[name = string("op_19784_cast_fp16")]; + tensor k_cast_fp16 = add(x = var_19780_cast_fp16, y = var_19784_cast_fp16)[name = string("k_cast_fp16")]; + tensor var_19786 = const()[name = string("op_19786"), val = tensor([1, 1024, 1, 1])]; + tensor cur_key_cast_fp16 = reshape(shape = var_19786, x = k_cast_fp16)[name = string("cur_key_cast_fp16")]; + tensor var_19788_to_fp16 = const()[name = string("op_19788_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647360)))]; + tensor var_19789_cast_fp16 = mul(x = key_cache_cast_fp16, y = var_19788_to_fp16)[name = string("op_19789_cast_fp16")]; + tensor upd_to_fp16 = const()[name = string("upd_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647488)))]; + tensor var_19790_cast_fp16 = mul(x = cur_key_cast_fp16, y = upd_to_fp16)[name = string("op_19790_cast_fp16")]; + tensor key_cast_fp16 = add(x = var_19789_cast_fp16, y = var_19790_cast_fp16)[name = string("key_cast_fp16")]; + tensor var_19792_to_fp16 = const()[name = string("op_19792_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141647360)))]; + tensor var_19793_cast_fp16 = mul(x = value_cache_cast_fp16, y = var_19792_to_fp16)[name = string("op_19793_cast_fp16")]; + tensor var_19794_cast_fp16 = mul(x = v_cast_fp16, y = upd_to_fp16)[name = string("op_19794_cast_fp16")]; + tensor value_cast_fp16 = add(x = var_19793_cast_fp16, y = var_19794_cast_fp16)[name = string("value_cast_fp16")]; + tensor var_19796 = const()[name = string("op_19796"), val = tensor([1, 8, 128, 16])]; + tensor kh_317_cast_fp16 = reshape(shape = var_19796, x = key_cast_fp16)[name = string("kh_317_cast_fp16")]; + tensor var_19798 = const()[name = string("op_19798"), val = tensor([1, 8, 128, 16])]; + tensor vh_317_cast_fp16 = reshape(shape = var_19798, x = value_cast_fp16)[name = string("vh_317_cast_fp16")]; + tensor transpose_316_perm_0 = const()[name = string("transpose_316_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_158_reps_0 = const()[name = string("tile_158_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_316_cast_fp16 = transpose(perm = transpose_316_perm_0, x = kh_317_cast_fp16)[name = string("transpose_5")]; + tensor tile_158_cast_fp16 = tile(reps = tile_158_reps_0, x = transpose_316_cast_fp16)[name = string("tile_158_cast_fp16")]; + tensor concat_394 = const()[name = string("concat_394"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_316_cast_fp16 = reshape(shape = concat_394, x = tile_158_cast_fp16)[name = string("reshape_316_cast_fp16")]; + tensor transpose_317_perm_0 = const()[name = string("transpose_317_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_395 = const()[name = string("concat_395"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_317_cast_fp16 = transpose(perm = transpose_317_perm_0, x = reshape_316_cast_fp16)[name = string("transpose_4")]; + tensor reshape_317_cast_fp16 = reshape(shape = concat_395, x = transpose_317_cast_fp16)[name = string("reshape_317_cast_fp16")]; + tensor transpose_318_perm_0 = const()[name = string("transpose_318_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_159_reps_0 = const()[name = string("tile_159_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_318_cast_fp16 = transpose(perm = transpose_318_perm_0, x = vh_317_cast_fp16)[name = string("transpose_3")]; + tensor tile_159_cast_fp16 = tile(reps = tile_159_reps_0, x = transpose_318_cast_fp16)[name = string("tile_159_cast_fp16")]; + tensor concat_396 = const()[name = string("concat_396"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_318_cast_fp16 = reshape(shape = concat_396, x = tile_159_cast_fp16)[name = string("reshape_318_cast_fp16")]; + tensor transpose_319_perm_0 = const()[name = string("transpose_319_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_397 = const()[name = string("concat_397"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_319_cast_fp16 = transpose(perm = transpose_319_perm_0, x = reshape_318_cast_fp16)[name = string("transpose_2")]; + tensor reshape_319_cast_fp16 = reshape(shape = concat_397, x = transpose_319_cast_fp16)[name = string("reshape_319_cast_fp16")]; + fp16 var_19802_to_fp16 = const()[name = string("op_19802_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_19803_cast_fp16 = mul(x = q_cast_fp16, y = var_19802_to_fp16)[name = string("op_19803_cast_fp16")]; + tensor transpose_633_perm_0 = const()[name = string("transpose_633_perm_0"), val = tensor([1, 0, -2, -1])]; + bool w_345_transpose_x_1 = const()[name = string("w_345_transpose_x_1"), val = bool(true)]; + bool w_345_transpose_y_1 = const()[name = string("w_345_transpose_y_1"), val = bool(false)]; + tensor transpose_633_cast_fp16 = transpose(perm = transpose_633_perm_0, x = reshape_317_cast_fp16)[name = string("transpose_1")]; + tensor w_345_cast_fp16 = matmul(transpose_x = w_345_transpose_x_1, transpose_y = w_345_transpose_y_1, x = var_19803_cast_fp16, y = transpose_633_cast_fp16)[name = string("w_345_cast_fp16")]; + tensor w_347_cast_fp16 = softmax(axis = var_19681, x = w_345_cast_fp16)[name = string("w_347_cast_fp16")]; + tensor transpose_634_perm_0 = const()[name = string("transpose_634_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_transpose_x_1 = const()[name = string("attn_transpose_x_1"), val = bool(false)]; + bool attn_transpose_y_1 = const()[name = string("attn_transpose_y_1"), val = bool(true)]; + tensor transpose_634_cast_fp16 = transpose(perm = transpose_634_perm_0, x = reshape_319_cast_fp16)[name = string("transpose_0")]; + tensor attn_cast_fp16 = matmul(transpose_x = attn_transpose_x_1, transpose_y = attn_transpose_y_1, x = transpose_634_cast_fp16, y = w_347_cast_fp16)[name = string("attn_cast_fp16")]; + tensor var_19810 = const()[name = string("op_19810"), val = tensor([1, 2048, 1, 1])]; + tensor input_849_cast_fp16 = reshape(shape = var_19810, x = attn_cast_fp16)[name = string("input_849_cast_fp16")]; + string attn_output_pad_type_0 = const()[name = string("attn_output_pad_type_0"), val = string("valid")]; + tensor attn_output_strides_0 = const()[name = string("attn_output_strides_0"), val = tensor([1, 1])]; + tensor attn_output_pad_0 = const()[name = string("attn_output_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor attn_output_dilations_0 = const()[name = string("attn_output_dilations_0"), val = tensor([1, 1])]; + int32 attn_output_groups_0 = const()[name = string("attn_output_groups_0"), val = int32(1)]; + tensor attn_output_cast_fp16 = conv(dilations = attn_output_dilations_0, groups = attn_output_groups_0, pad = attn_output_pad_0, pad_type = attn_output_pad_type_0, strides = attn_output_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_849_cast_fp16)[name = string("attn_output_cast_fp16")]; + tensor x_cast_fp16 = add(x = x_601_cast_fp16, y = attn_output_cast_fp16)[name = string("x_cast_fp16")]; + tensor var_19824_cast_fp16 = mul(x = x_cast_fp16, y = x_cast_fp16)[name = string("op_19824_cast_fp16")]; + tensor variance_667_axes_0 = const()[name = string("variance_667_axes_0"), val = tensor([1])]; + bool variance_667_keep_dims_0 = const()[name = string("variance_667_keep_dims_0"), val = bool(true)]; + tensor variance_667_cast_fp16 = reduce_mean(axes = variance_667_axes_0, keep_dims = variance_667_keep_dims_0, x = var_19824_cast_fp16)[name = string("variance_667_cast_fp16")]; + fp16 var_19827_to_fp16 = const()[name = string("op_19827_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19828_cast_fp16 = add(x = variance_667_cast_fp16, y = var_19827_to_fp16)[name = string("op_19828_cast_fp16")]; + fp32 var_19829_epsilon_0 = const()[name = string("op_19829_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19829_cast_fp16 = rsqrt(epsilon = var_19829_epsilon_0, x = var_19828_cast_fp16)[name = string("op_19829_cast_fp16")]; + tensor var_19830_cast_fp16 = mul(x = x_cast_fp16, y = var_19829_cast_fp16)[name = string("op_19830_cast_fp16")]; + tensor input_851_cast_fp16 = mul(x = var_19830_cast_fp16, y = layers_4_post_attention_layernorm_weight_to_fp16)[name = string("input_851_cast_fp16")]; + string input_853_pad_type_0 = const()[name = string("input_853_pad_type_0"), val = string("valid")]; + tensor input_853_strides_0 = const()[name = string("input_853_strides_0"), val = tensor([1, 1])]; + tensor input_853_pad_0 = const()[name = string("input_853_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_853_dilations_0 = const()[name = string("input_853_dilations_0"), val = tensor([1, 1])]; + int32 input_853_groups_0 = const()[name = string("input_853_groups_0"), val = int32(1)]; + tensor input_853_cast_fp16 = conv(dilations = input_853_dilations_0, groups = input_853_groups_0, pad = input_853_pad_0, pad_type = input_853_pad_type_0, strides = input_853_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_851_cast_fp16)[name = string("input_853_cast_fp16")]; + tensor var_19838_cast_fp16 = silu(x = input_853_cast_fp16)[name = string("op_19838_cast_fp16")]; + string var_19844_pad_type_0 = const()[name = string("op_19844_pad_type_0"), val = string("valid")]; + tensor var_19844_strides_0 = const()[name = string("op_19844_strides_0"), val = tensor([1, 1])]; + tensor var_19844_pad_0 = const()[name = string("op_19844_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19844_dilations_0 = const()[name = string("op_19844_dilations_0"), val = tensor([1, 1])]; + int32 var_19844_groups_0 = const()[name = string("op_19844_groups_0"), val = int32(1)]; + tensor var_19844_cast_fp16 = conv(dilations = var_19844_dilations_0, groups = var_19844_groups_0, pad = var_19844_pad_0, pad_type = var_19844_pad_type_0, strides = var_19844_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_851_cast_fp16)[name = string("op_19844_cast_fp16")]; + tensor input_855_cast_fp16 = mul(x = var_19838_cast_fp16, y = var_19844_cast_fp16)[name = string("input_855_cast_fp16")]; + string h_pad_type_0 = const()[name = string("h_pad_type_0"), val = string("valid")]; + tensor h_strides_0 = const()[name = string("h_strides_0"), val = tensor([1, 1])]; + tensor h_pad_0 = const()[name = string("h_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor h_dilations_0 = const()[name = string("h_dilations_0"), val = tensor([1, 1])]; + int32 h_groups_0 = const()[name = string("h_groups_0"), val = int32(1)]; + tensor h_cast_fp16 = conv(dilations = h_dilations_0, groups = h_groups_0, pad = h_pad_0, pad_type = h_pad_type_0, strides = h_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_855_cast_fp16)[name = string("h_cast_fp16")]; + tensor inputs_cast_fp16 = add(x = x_cast_fp16, y = h_cast_fp16)[name = string("inputs_cast_fp16")]; + tensor inputs_sq_cast_fp16 = mul(x = inputs_cast_fp16, y = inputs_cast_fp16)[name = string("inputs_sq_cast_fp16")]; + tensor variance_axes_0 = const()[name = string("variance_axes_0"), val = tensor([1])]; + bool variance_keep_dims_0 = const()[name = string("variance_keep_dims_0"), val = bool(true)]; + tensor variance_cast_fp16 = reduce_mean(axes = variance_axes_0, keep_dims = variance_keep_dims_0, x = inputs_sq_cast_fp16)[name = string("variance_cast_fp16")]; + fp16 var_19865_to_fp16 = const()[name = string("op_19865_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19866_cast_fp16 = add(x = variance_cast_fp16, y = var_19865_to_fp16)[name = string("op_19866_cast_fp16")]; + fp32 var_19867_epsilon_0 = const()[name = string("op_19867_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19867_cast_fp16 = rsqrt(epsilon = var_19867_epsilon_0, x = var_19866_cast_fp16)[name = string("op_19867_cast_fp16")]; + tensor hidden_states_cast_fp16 = mul(x = inputs_cast_fp16, y = var_19867_cast_fp16)[name = string("hidden_states_cast_fp16")]; + tensor input_857_cast_fp16 = mul(x = w_41_to_fp16, y = hidden_states_cast_fp16)[name = string("input_857_cast_fp16")]; + string logits_57_pad_type_0 = const()[name = string("logits_57_pad_type_0"), val = string("valid")]; + tensor logits_57_strides_0 = const()[name = string("logits_57_strides_0"), val = tensor([1, 1])]; + tensor logits_57_pad_0 = const()[name = string("logits_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_57_dilations_0 = const()[name = string("logits_57_dilations_0"), val = tensor([1, 1])]; + int32 logits_57_groups_0 = const()[name = string("logits_57_groups_0"), val = int32(1)]; + tensor lm_heads_14_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108075776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110172992))))[name = string("lm_heads_14_weight_to_fp16_palettized")]; + tensor logits_57_cast_fp16 = conv(dilations = logits_57_dilations_0, groups = logits_57_groups_0, pad = logits_57_pad_0, pad_type = logits_57_pad_type_0, strides = logits_57_strides_0, weight = lm_heads_14_weight_to_fp16_palettized, x = input_857_cast_fp16)[name = string("logits_57_cast_fp16")]; + tensor var_19885 = const()[name = string("op_19885"), val = tensor([1, 2048])]; + tensor logits_cast_fp16 = reshape(shape = var_19885, x = logits_57_cast_fp16)[name = string("logits_cast_fp16")]; + tensor scaled_logits_cast_fp16 = real_div(x = logits_cast_fp16, y = temperature)[name = string("scaled_logits_cast_fp16")]; + int32 var_19895 = const()[name = string("op_19895"), val = int32(100)]; + int32 top_values_axis_0 = const()[name = string("top_values_axis_0"), val = int32(1)]; + bool top_values_ascending_0 = const()[name = string("top_values_ascending_0"), val = bool(false)]; + bool top_values_sort_0 = const()[name = string("top_values_sort_0"), val = bool(true)]; + bool top_values_return_indices_0 = const()[name = string("top_values_return_indices_0"), val = bool(true)]; + string top_values_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_cast_fp16_cast_uint16_0, tensor top_values_cast_fp16_cast_uint16_1 = topk(ascending = top_values_ascending_0, axis = top_values_axis_0, k = var_19895, output_indices_dtype = top_values_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_return_indices_0, sort = top_values_sort_0, x = scaled_logits_cast_fp16)[name = string("top_values_cast_fp16_cast_uint16")]; + tensor var_19901_cast_fp16 = mul(x = top_values_cast_fp16_cast_uint16_0, y = var_2433_cast_fp16_to_fp16)[name = string("op_19901_cast_fp16")]; + tensor var_19905_cast_fp16 = add(x = var_19901_cast_fp16, y = var_2438_cast_fp16)[name = string("op_19905_cast_fp16")]; + tensor reduce_min_14_axes_0 = const()[name = string("reduce_min_14_axes_0"), val = tensor([1])]; + bool reduce_min_14_keep_dims_0 = const()[name = string("reduce_min_14_keep_dims_0"), val = bool(true)]; + tensor reduce_min_14_cast_fp16 = reduce_min(axes = reduce_min_14_axes_0, keep_dims = reduce_min_14_keep_dims_0, x = var_19905_cast_fp16)[name = string("reduce_min_14_cast_fp16")]; + tensor var_19908_cast_fp16 = greater_equal(x = scaled_logits_cast_fp16, y = reduce_min_14_cast_fp16)[name = string("op_19908_cast_fp16")]; + fp16 var_19909_value_0_to_fp16 = const()[name = string("op_19909_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_19909_cast_fp16 = fill_like(ref_tensor = scaled_logits_cast_fp16, value = var_19909_value_0_to_fp16)[name = string("op_19909_cast_fp16")]; + tensor masked_logits_cast_fp16 = select(a = scaled_logits_cast_fp16, b = var_19909_cast_fp16, cond = var_19908_cast_fp16)[name = string("masked_logits_cast_fp16")]; + tensor var_19913_begin_0 = const()[name = string("op_19913_begin_0"), val = tensor([14, 0])]; + tensor var_19913_end_0 = const()[name = string("op_19913_end_0"), val = tensor([15, 2048])]; + tensor var_19913_end_mask_0 = const()[name = string("op_19913_end_mask_0"), val = tensor([false, true])]; + tensor var_19913_squeeze_mask_0 = const()[name = string("op_19913_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_19913_cast_fp16 = slice_by_index(begin = var_19913_begin_0, end = var_19913_end_0, end_mask = var_19913_end_mask_0, squeeze_mask = var_19913_squeeze_mask_0, x = gumbel)[name = string("op_19913_cast_fp16")]; + tensor var_19916 = const()[name = string("op_19916"), val = tensor([1, 2048])]; + tensor var_19917_cast_fp16 = reshape(shape = var_19916, x = var_19913_cast_fp16)[name = string("op_19917_cast_fp16")]; + tensor noisy_logits_cast_fp16 = add(x = masked_logits_cast_fp16, y = var_19917_cast_fp16)[name = string("noisy_logits_cast_fp16")]; + int32 code_29_axis_0 = const()[name = string("code_29_axis_0"), val = int32(1)]; + bool code_29_keep_dims_0 = const()[name = string("code_29_keep_dims_0"), val = bool(false)]; + string code_29_output_dtype_0 = const()[name = string("code_29_output_dtype_0"), val = string("int32")]; + tensor code_29_cast_fp16 = reduce_argmax(axis = code_29_axis_0, keep_dims = code_29_keep_dims_0, output_dtype = code_29_output_dtype_0, x = noisy_logits_cast_fp16)[name = string("code_29_cast_fp16")]; + int32 var_19928 = const()[name = string("op_19928"), val = int32(28672)]; + tensor input = add(x = code_29_cast_fp16, y = var_19928)[name = string("input")]; + int32 code_embed_57_axis_0 = const()[name = string("code_embed_57_axis_0"), val = int32(0)]; + int32 code_embed_57_batch_dims_0 = const()[name = string("code_embed_57_batch_dims_0"), val = int32(0)]; + bool code_embed_57_validate_indices_0 = const()[name = string("code_embed_57_validate_indices_0"), val = bool(false)]; + string input_to_uint16_dtype_0 = const()[name = string("input_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_to_uint16 = cast(dtype = input_to_uint16_dtype_0, x = input)[name = string("cast_0")]; + tensor code_embed_57_cast_fp16_cast_uint16 = gather(axis = code_embed_57_axis_0, batch_dims = code_embed_57_batch_dims_0, indices = input_to_uint16, validate_indices = code_embed_57_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_57_cast_fp16_cast_uint16")]; + tensor var_19932 = const()[name = string("op_19932"), val = tensor([1, 1024, 1, 1])]; + tensor code_embed_cast_fp16 = reshape(shape = var_19932, x = code_embed_57_cast_fp16_cast_uint16)[name = string("code_embed_cast_fp16")]; + tensor embed_sum = add(x = embed_sum_cast_fp16, y = code_embed_cast_fp16)[name = string("op_19938_cast_fp16")]; + int32 var_19941_axis_0 = const()[name = string("op_19941_axis_0"), val = int32(1)]; + tensor codes = stack(axis = var_19941_axis_0, values = (code_1_cast_fp16, code_3_cast_fp16, code_5_cast_fp16, code_7_cast_fp16, code_9_cast_fp16, code_11_cast_fp16, code_13_cast_fp16, code_15_cast_fp16, code_17_cast_fp16, code_19_cast_fp16, code_21_cast_fp16, code_23_cast_fp16, code_25_cast_fp16, code_27_cast_fp16, code_29_cast_fp16))[name = string("op_19941")]; + } -> (codes, embed_sum); + func stepped(tensor cache_length, tensor input_embeds, tensor key_cache, tensor key_padding_mask, tensor kv_cache_update_mask, tensor value_cache) { + int32 pos_cos_batch_dims_0 = const()[name = string("pos_cos_batch_dims_0"), val = int32(0)]; + bool pos_cos_validate_indices_0 = const()[name = string("pos_cos_validate_indices_0"), val = bool(false)]; + tensor position_embeddings_cos_weight_to_fp16 = const()[name = string("position_embeddings_cos_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64)))]; + string cache_length_to_int16_dtype_0 = const()[name = string("cache_length_to_int16_dtype_0"), val = string("int16")]; + string cast_111_dtype_0 = const()[name = string("cast_111_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; + tensor cache_length_to_int16 = cast(dtype = cache_length_to_int16_dtype_0, x = cache_length)[name = string("cast_5")]; + tensor cast_111 = cast(dtype = cast_111_dtype_0, x = cache_length_to_int16)[name = string("cast_4")]; + tensor greater_equal_0 = greater_equal(x = cast_111, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; + int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(16)]; + tensor add_0 = add(x = cast_111, y = slice_by_index_0)[name = string("add_0")]; + tensor select_0 = select(a = cast_111, b = add_0, cond = greater_equal_0)[name = string("select_0")]; + string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; + string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; + tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; + tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; + tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; + int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(16)]; + tensor add_0_1 = add(x = cast_0, y = slice_by_index_0_1)[name = string("add_0_1")]; + tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; + int32 pos_cos_cast_fp16_cast_uint16_cast_uint16_axis_0 = const()[name = string("pos_cos_cast_fp16_cast_uint16_cast_uint16_axis_0"), val = int32(0)]; + tensor pos_cos_cast_fp16_cast_uint16_cast_uint16 = gather(axis = pos_cos_cast_fp16_cast_uint16_cast_uint16_axis_0, batch_dims = pos_cos_batch_dims_0, indices = select_0_1, validate_indices = pos_cos_validate_indices_0, x = position_embeddings_cos_weight_to_fp16)[name = string("pos_cos_cast_fp16_cast_uint16_cast_uint16")]; + tensor obj_7_axes_0 = const()[name = string("obj_7_axes_0"), val = tensor([2])]; + tensor obj_7_cast_fp16 = expand_dims(axes = obj_7_axes_0, x = pos_cos_cast_fp16_cast_uint16_cast_uint16)[name = string("obj_7_cast_fp16")]; + int32 pos_sin_axis_0 = const()[name = string("pos_sin_axis_0"), val = int32(0)]; + int32 pos_sin_batch_dims_0 = const()[name = string("pos_sin_batch_dims_0"), val = int32(0)]; + bool pos_sin_validate_indices_0 = const()[name = string("pos_sin_validate_indices_0"), val = bool(false)]; + tensor position_embeddings_sin_weight_to_fp16 = const()[name = string("position_embeddings_sin_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4224)))]; + string cache_length_to_uint16_dtype_0 = const()[name = string("cache_length_to_uint16_dtype_0"), val = string("uint16")]; + tensor cache_length_to_uint16 = cast(dtype = cache_length_to_uint16_dtype_0, x = cache_length)[name = string("cast_1")]; + tensor pos_sin_cast_fp16_cast_uint16 = gather(axis = pos_sin_axis_0, batch_dims = pos_sin_batch_dims_0, indices = cache_length_to_uint16, validate_indices = pos_sin_validate_indices_0, x = position_embeddings_sin_weight_to_fp16)[name = string("pos_sin_cast_fp16_cast_uint16")]; + tensor obj_9_axes_0 = const()[name = string("obj_9_axes_0"), val = tensor([2])]; + tensor obj_9_cast_fp16 = expand_dims(axes = obj_9_axes_0, x = pos_sin_cast_fp16_cast_uint16)[name = string("obj_9_cast_fp16")]; + tensor tile_0 = const()[name = string("tile_0"), val = tensor([1024, 1024, 1024, 1024, 1024])]; + int32 var_84_axis_0 = const()[name = string("op_84_axis_0"), val = int32(1)]; + tensor var_84_cast_fp16_0, tensor var_84_cast_fp16_1, tensor var_84_cast_fp16_2, tensor var_84_cast_fp16_3, tensor var_84_cast_fp16_4 = split(axis = var_84_axis_0, split_sizes = tile_0, x = key_cache)[name = string("op_84_cast_fp16")]; + tensor tile_1 = const()[name = string("tile_1"), val = tensor([1024, 1024, 1024, 1024, 1024])]; + int32 var_92_axis_0 = const()[name = string("op_92_axis_0"), val = int32(1)]; + tensor var_92_cast_fp16_0, tensor var_92_cast_fp16_1, tensor var_92_cast_fp16_2, tensor var_92_cast_fp16_3, tensor var_92_cast_fp16_4 = split(axis = var_92_axis_0, split_sizes = tile_1, x = value_cache)[name = string("op_92_cast_fp16")]; + int32 var_99 = const()[name = string("op_99"), val = int32(3)]; + int32 var_109 = const()[name = string("op_109"), val = int32(-2)]; + int32 var_117 = const()[name = string("op_117"), val = int32(1)]; + tensor inputs_sq_1_cast_fp16 = mul(x = input_embeds, y = input_embeds)[name = string("inputs_sq_1_cast_fp16")]; + tensor variance_1_axes_0 = const()[name = string("variance_1_axes_0"), val = tensor([1])]; + bool variance_1_keep_dims_0 = const()[name = string("variance_1_keep_dims_0"), val = bool(true)]; + tensor variance_1_cast_fp16 = reduce_mean(axes = variance_1_axes_0, keep_dims = variance_1_keep_dims_0, x = inputs_sq_1_cast_fp16)[name = string("variance_1_cast_fp16")]; + fp16 var_129_to_fp16 = const()[name = string("op_129_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_130_cast_fp16 = add(x = variance_1_cast_fp16, y = var_129_to_fp16)[name = string("op_130_cast_fp16")]; + fp32 var_131_epsilon_0 = const()[name = string("op_131_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_131_cast_fp16 = rsqrt(epsilon = var_131_epsilon_0, x = var_130_cast_fp16)[name = string("op_131_cast_fp16")]; + tensor hidden_states_1_cast_fp16 = mul(x = input_embeds, y = var_131_cast_fp16)[name = string("hidden_states_1_cast_fp16")]; + tensor w_1_to_fp16 = const()[name = string("w_1_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8384)))]; + tensor obj_1_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_1_cast_fp16)[name = string("obj_1_cast_fp16")]; + string query_1_pad_type_0 = const()[name = string("query_1_pad_type_0"), val = string("valid")]; + tensor query_1_strides_0 = const()[name = string("query_1_strides_0"), val = tensor([1, 1])]; + tensor query_1_pad_0 = const()[name = string("query_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_1_dilations_0 = const()[name = string("query_1_dilations_0"), val = tensor([1, 1])]; + int32 query_1_groups_0 = const()[name = string("query_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(10496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2107712))))[name = string("layers_0_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor layers_0_self_attn_q_proj_bias_to_fp16 = const()[name = string("layers_0_self_attn_q_proj_bias_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2108288)))]; + tensor query_1_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_1_dilations_0, groups = query_1_groups_0, pad = query_1_pad_0, pad_type = query_1_pad_type_0, strides = query_1_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_1_cast_fp16)[name = string("query_1_cast_fp16")]; + string current_key_1_pad_type_0 = const()[name = string("current_key_1_pad_type_0"), val = string("valid")]; + tensor current_key_1_strides_0 = const()[name = string("current_key_1_strides_0"), val = tensor([1, 1])]; + tensor current_key_1_pad_0 = const()[name = string("current_key_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_1_dilations_0 = const()[name = string("current_key_1_dilations_0"), val = tensor([1, 1])]; + int32 current_key_1_groups_0 = const()[name = string("current_key_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2112448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3161088))))[name = string("layers_0_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_1_cast_fp16 = conv(dilations = current_key_1_dilations_0, groups = current_key_1_groups_0, pad = current_key_1_pad_0, pad_type = current_key_1_pad_type_0, strides = current_key_1_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_1_cast_fp16)[name = string("current_key_1_cast_fp16")]; + string current_value_1_pad_type_0 = const()[name = string("current_value_1_pad_type_0"), val = string("valid")]; + tensor current_value_1_strides_0 = const()[name = string("current_value_1_strides_0"), val = tensor([1, 1])]; + tensor current_value_1_pad_0 = const()[name = string("current_value_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_1_dilations_0 = const()[name = string("current_value_1_dilations_0"), val = tensor([1, 1])]; + int32 current_value_1_groups_0 = const()[name = string("current_value_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3161664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4210304))))[name = string("layers_0_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor layers_0_self_attn_v_proj_bias_to_fp16 = const()[name = string("layers_0_self_attn_v_proj_bias_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4210880)))]; + tensor current_value_1_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_1_dilations_0, groups = current_value_1_groups_0, pad = current_value_1_pad_0, pad_type = current_value_1_pad_type_0, strides = current_value_1_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_1_cast_fp16)[name = string("current_value_1_cast_fp16")]; + tensor var_168 = const()[name = string("op_168"), val = tensor([16, 128, 1, 1])]; + tensor inputs_1_cast_fp16 = reshape(shape = var_168, x = query_1_cast_fp16)[name = string("inputs_1_cast_fp16")]; + tensor inputs_sq_3_cast_fp16 = mul(x = inputs_1_cast_fp16, y = inputs_1_cast_fp16)[name = string("inputs_sq_3_cast_fp16")]; + tensor variance_3_axes_0 = const()[name = string("variance_3_axes_0"), val = tensor([1])]; + bool variance_3_keep_dims_0 = const()[name = string("variance_3_keep_dims_0"), val = bool(true)]; + tensor variance_3_cast_fp16 = reduce_mean(axes = variance_3_axes_0, keep_dims = variance_3_keep_dims_0, x = inputs_sq_3_cast_fp16)[name = string("variance_3_cast_fp16")]; + fp16 var_174_to_fp16 = const()[name = string("op_174_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_175_cast_fp16 = add(x = variance_3_cast_fp16, y = var_174_to_fp16)[name = string("op_175_cast_fp16")]; + fp32 var_176_epsilon_0 = const()[name = string("op_176_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_176_cast_fp16 = rsqrt(epsilon = var_176_epsilon_0, x = var_175_cast_fp16)[name = string("op_176_cast_fp16")]; + tensor hidden_states_3_cast_fp16 = mul(x = inputs_1_cast_fp16, y = var_176_cast_fp16)[name = string("hidden_states_3_cast_fp16")]; + tensor w_3_to_fp16 = const()[name = string("w_3_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4212992)))]; + tensor query_normed_1_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_3_cast_fp16)[name = string("query_normed_1_cast_fp16")]; + tensor var_184 = const()[name = string("op_184"), val = tensor([8, 128, 1, 1])]; + tensor inputs_3_cast_fp16 = reshape(shape = var_184, x = current_key_1_cast_fp16)[name = string("inputs_3_cast_fp16")]; + tensor inputs_sq_5_cast_fp16 = mul(x = inputs_3_cast_fp16, y = inputs_3_cast_fp16)[name = string("inputs_sq_5_cast_fp16")]; + tensor variance_5_axes_0 = const()[name = string("variance_5_axes_0"), val = tensor([1])]; + bool variance_5_keep_dims_0 = const()[name = string("variance_5_keep_dims_0"), val = bool(true)]; + tensor variance_5_cast_fp16 = reduce_mean(axes = variance_5_axes_0, keep_dims = variance_5_keep_dims_0, x = inputs_sq_5_cast_fp16)[name = string("variance_5_cast_fp16")]; + fp16 var_190_to_fp16 = const()[name = string("op_190_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_191_cast_fp16 = add(x = variance_5_cast_fp16, y = var_190_to_fp16)[name = string("op_191_cast_fp16")]; + fp32 var_192_epsilon_0 = const()[name = string("op_192_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_192_cast_fp16 = rsqrt(epsilon = var_192_epsilon_0, x = var_191_cast_fp16)[name = string("op_192_cast_fp16")]; + tensor hidden_states_5_cast_fp16 = mul(x = inputs_3_cast_fp16, y = var_192_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; + tensor w_5_to_fp16 = const()[name = string("w_5_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4213312)))]; + tensor current_key_normed_1_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_5_cast_fp16)[name = string("current_key_normed_1_cast_fp16")]; + tensor var_210 = const()[name = string("op_210"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_1_cast_fp16 = reshape(shape = var_210, x = query_normed_1_cast_fp16)[name = string("mh_q_1_cast_fp16")]; + tensor var_212 = const()[name = string("op_212"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_1_cast_fp16 = reshape(shape = var_212, x = current_key_normed_1_cast_fp16)[name = string("mh_k_1_cast_fp16")]; + tensor cos_1_axes_0 = const()[name = string("cos_1_axes_0"), val = tensor([1])]; + tensor cos_1_cast_fp16 = expand_dims(axes = cos_1_axes_0, x = obj_7_cast_fp16)[name = string("cos_1_cast_fp16")]; + tensor sin_1_axes_0 = const()[name = string("sin_1_axes_0"), val = tensor([1])]; + tensor sin_1_cast_fp16 = expand_dims(axes = sin_1_axes_0, x = obj_9_cast_fp16)[name = string("sin_1_cast_fp16")]; + tensor var_216_cast_fp16 = mul(x = mh_q_1_cast_fp16, y = cos_1_cast_fp16)[name = string("op_216_cast_fp16")]; + tensor var_221_begin_0 = const()[name = string("op_221_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_221_end_0 = const()[name = string("op_221_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_221_end_mask_0 = const()[name = string("op_221_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_221_cast_fp16 = slice_by_index(begin = var_221_begin_0, end = var_221_end_0, end_mask = var_221_end_mask_0, x = mh_q_1_cast_fp16)[name = string("op_221_cast_fp16")]; + tensor var_227_begin_0 = const()[name = string("op_227_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_227_end_0 = const()[name = string("op_227_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_227_end_mask_0 = const()[name = string("op_227_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_227_cast_fp16 = slice_by_index(begin = var_227_begin_0, end = var_227_end_0, end_mask = var_227_end_mask_0, x = mh_q_1_cast_fp16)[name = string("op_227_cast_fp16")]; + fp16 const_17_promoted_to_fp16 = const()[name = string("const_17_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_229_cast_fp16 = mul(x = var_227_cast_fp16, y = const_17_promoted_to_fp16)[name = string("op_229_cast_fp16")]; + bool var_231_interleave_0 = const()[name = string("op_231_interleave_0"), val = bool(false)]; + tensor var_231_cast_fp16 = concat(axis = var_109, interleave = var_231_interleave_0, values = (var_229_cast_fp16, var_221_cast_fp16))[name = string("op_231_cast_fp16")]; + tensor var_232_cast_fp16 = mul(x = var_231_cast_fp16, y = sin_1_cast_fp16)[name = string("op_232_cast_fp16")]; + tensor mh_q_3_cast_fp16 = add(x = var_216_cast_fp16, y = var_232_cast_fp16)[name = string("mh_q_3_cast_fp16")]; + tensor var_234_cast_fp16 = mul(x = mh_k_1_cast_fp16, y = cos_1_cast_fp16)[name = string("op_234_cast_fp16")]; + tensor var_239_begin_0 = const()[name = string("op_239_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_239_end_0 = const()[name = string("op_239_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_239_end_mask_0 = const()[name = string("op_239_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_239_cast_fp16 = slice_by_index(begin = var_239_begin_0, end = var_239_end_0, end_mask = var_239_end_mask_0, x = mh_k_1_cast_fp16)[name = string("op_239_cast_fp16")]; + tensor var_245_begin_0 = const()[name = string("op_245_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_245_end_0 = const()[name = string("op_245_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_245_end_mask_0 = const()[name = string("op_245_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_245_cast_fp16 = slice_by_index(begin = var_245_begin_0, end = var_245_end_0, end_mask = var_245_end_mask_0, x = mh_k_1_cast_fp16)[name = string("op_245_cast_fp16")]; + fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_247_cast_fp16 = mul(x = var_245_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_247_cast_fp16")]; + bool var_249_interleave_0 = const()[name = string("op_249_interleave_0"), val = bool(false)]; + tensor var_249_cast_fp16 = concat(axis = var_109, interleave = var_249_interleave_0, values = (var_247_cast_fp16, var_239_cast_fp16))[name = string("op_249_cast_fp16")]; + tensor var_250_cast_fp16 = mul(x = var_249_cast_fp16, y = sin_1_cast_fp16)[name = string("op_250_cast_fp16")]; + tensor mh_k_3_cast_fp16 = add(x = var_234_cast_fp16, y = var_250_cast_fp16)[name = string("mh_k_3_cast_fp16")]; + tensor var_254 = const()[name = string("op_254"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_3_cast_fp16 = reshape(shape = var_254, x = mh_k_3_cast_fp16)[name = string("current_key_3_cast_fp16")]; + tensor var_257_axes_0 = const()[name = string("op_257_axes_0"), val = tensor([1])]; + tensor var_257_cast_fp16 = expand_dims(axes = var_257_axes_0, x = kv_cache_update_mask)[name = string("op_257_cast_fp16")]; + tensor var_258_axes_0 = const()[name = string("op_258_axes_0"), val = tensor([2])]; + tensor var_258_cast_fp16 = expand_dims(axes = var_258_axes_0, x = var_257_cast_fp16)[name = string("op_258_cast_fp16")]; + fp16 var_110_to_fp16 = const()[name = string("op_110_to_fp16"), val = fp16(0x1p+0)]; + tensor var_260_cast_fp16 = sub(x = var_110_to_fp16, y = var_258_cast_fp16)[name = string("op_260_cast_fp16")]; + tensor var_261_cast_fp16 = mul(x = var_84_cast_fp16_0, y = var_260_cast_fp16)[name = string("op_261_cast_fp16")]; + tensor var_262_cast_fp16 = mul(x = current_key_3_cast_fp16, y = var_258_cast_fp16)[name = string("op_262_cast_fp16")]; + tensor key_3_cast_fp16 = add(x = var_261_cast_fp16, y = var_262_cast_fp16)[name = string("key_3_cast_fp16")]; + tensor var_265_cast_fp16 = mul(x = var_92_cast_fp16_0, y = var_260_cast_fp16)[name = string("op_265_cast_fp16")]; + tensor var_266_cast_fp16 = mul(x = current_value_1_cast_fp16, y = var_258_cast_fp16)[name = string("op_266_cast_fp16")]; + tensor value_1_cast_fp16 = add(x = var_265_cast_fp16, y = var_266_cast_fp16)[name = string("value_1_cast_fp16")]; + tensor var_270 = const()[name = string("op_270"), val = tensor([1, 8, 128, 16])]; + tensor key_heads_1_cast_fp16 = reshape(shape = var_270, x = key_3_cast_fp16)[name = string("key_heads_1_cast_fp16")]; + tensor var_272 = const()[name = string("op_272"), val = tensor([1, 8, 128, 16])]; + tensor value_heads_1_cast_fp16 = reshape(shape = var_272, x = value_1_cast_fp16)[name = string("value_heads_1_cast_fp16")]; + tensor var_275_begin_0 = const()[name = string("op_275_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_275_end_0 = const()[name = string("op_275_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_275_end_mask_0 = const()[name = string("op_275_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_275_cast_fp16 = slice_by_index(begin = var_275_begin_0, end = var_275_end_0, end_mask = var_275_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_275_cast_fp16")]; + tensor var_279_begin_0 = const()[name = string("op_279_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_279_end_0 = const()[name = string("op_279_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_279_end_mask_0 = const()[name = string("op_279_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_279_cast_fp16 = slice_by_index(begin = var_279_begin_0, end = var_279_end_0, end_mask = var_279_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_279_cast_fp16")]; + tensor var_291_begin_0 = const()[name = string("op_291_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_291_end_0 = const()[name = string("op_291_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_291_end_mask_0 = const()[name = string("op_291_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_291_cast_fp16 = slice_by_index(begin = var_291_begin_0, end = var_291_end_0, end_mask = var_291_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_291_cast_fp16")]; + tensor var_295_begin_0 = const()[name = string("op_295_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_295_end_0 = const()[name = string("op_295_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_295_end_mask_0 = const()[name = string("op_295_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_295_cast_fp16 = slice_by_index(begin = var_295_begin_0, end = var_295_end_0, end_mask = var_295_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_295_cast_fp16")]; + tensor var_307_begin_0 = const()[name = string("op_307_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_307_end_0 = const()[name = string("op_307_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_307_end_mask_0 = const()[name = string("op_307_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_307_cast_fp16 = slice_by_index(begin = var_307_begin_0, end = var_307_end_0, end_mask = var_307_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_307_cast_fp16")]; + tensor var_311_begin_0 = const()[name = string("op_311_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_311_end_0 = const()[name = string("op_311_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_311_end_mask_0 = const()[name = string("op_311_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_311_cast_fp16 = slice_by_index(begin = var_311_begin_0, end = var_311_end_0, end_mask = var_311_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_311_cast_fp16")]; + tensor var_323_begin_0 = const()[name = string("op_323_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_323_end_0 = const()[name = string("op_323_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_323_end_mask_0 = const()[name = string("op_323_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_323_cast_fp16 = slice_by_index(begin = var_323_begin_0, end = var_323_end_0, end_mask = var_323_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_323_cast_fp16")]; + tensor var_327_begin_0 = const()[name = string("op_327_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_327_end_0 = const()[name = string("op_327_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_327_end_mask_0 = const()[name = string("op_327_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_327_cast_fp16 = slice_by_index(begin = var_327_begin_0, end = var_327_end_0, end_mask = var_327_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_327_cast_fp16")]; + tensor var_339_begin_0 = const()[name = string("op_339_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_339_end_0 = const()[name = string("op_339_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_339_end_mask_0 = const()[name = string("op_339_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_339_cast_fp16 = slice_by_index(begin = var_339_begin_0, end = var_339_end_0, end_mask = var_339_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_339_cast_fp16")]; + tensor var_343_begin_0 = const()[name = string("op_343_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_343_end_0 = const()[name = string("op_343_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_343_end_mask_0 = const()[name = string("op_343_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_343_cast_fp16 = slice_by_index(begin = var_343_begin_0, end = var_343_end_0, end_mask = var_343_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_343_cast_fp16")]; + tensor var_355_begin_0 = const()[name = string("op_355_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_355_end_0 = const()[name = string("op_355_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_355_end_mask_0 = const()[name = string("op_355_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_355_cast_fp16 = slice_by_index(begin = var_355_begin_0, end = var_355_end_0, end_mask = var_355_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_355_cast_fp16")]; + tensor var_359_begin_0 = const()[name = string("op_359_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_359_end_0 = const()[name = string("op_359_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_359_end_mask_0 = const()[name = string("op_359_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_359_cast_fp16 = slice_by_index(begin = var_359_begin_0, end = var_359_end_0, end_mask = var_359_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_359_cast_fp16")]; + tensor var_371_begin_0 = const()[name = string("op_371_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_371_end_0 = const()[name = string("op_371_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_371_end_mask_0 = const()[name = string("op_371_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_371_cast_fp16 = slice_by_index(begin = var_371_begin_0, end = var_371_end_0, end_mask = var_371_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_371_cast_fp16")]; + tensor var_375_begin_0 = const()[name = string("op_375_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_375_end_0 = const()[name = string("op_375_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_375_end_mask_0 = const()[name = string("op_375_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_375_cast_fp16 = slice_by_index(begin = var_375_begin_0, end = var_375_end_0, end_mask = var_375_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_375_cast_fp16")]; + tensor var_387_begin_0 = const()[name = string("op_387_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_387_end_0 = const()[name = string("op_387_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_387_end_mask_0 = const()[name = string("op_387_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_387_cast_fp16 = slice_by_index(begin = var_387_begin_0, end = var_387_end_0, end_mask = var_387_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_387_cast_fp16")]; + tensor var_391_begin_0 = const()[name = string("op_391_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_391_end_0 = const()[name = string("op_391_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_391_end_mask_0 = const()[name = string("op_391_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_391_cast_fp16 = slice_by_index(begin = var_391_begin_0, end = var_391_end_0, end_mask = var_391_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_391_cast_fp16")]; + bool key_heads_3_interleave_0 = const()[name = string("key_heads_3_interleave_0"), val = bool(false)]; + tensor key_heads_3_cast_fp16 = concat(axis = var_117, interleave = key_heads_3_interleave_0, values = (var_275_cast_fp16, var_275_cast_fp16, var_291_cast_fp16, var_291_cast_fp16, var_307_cast_fp16, var_307_cast_fp16, var_323_cast_fp16, var_323_cast_fp16, var_339_cast_fp16, var_339_cast_fp16, var_355_cast_fp16, var_355_cast_fp16, var_371_cast_fp16, var_371_cast_fp16, var_387_cast_fp16, var_387_cast_fp16))[name = string("key_heads_3_cast_fp16")]; + bool value_heads_3_interleave_0 = const()[name = string("value_heads_3_interleave_0"), val = bool(false)]; + tensor value_heads_3_cast_fp16 = concat(axis = var_117, interleave = value_heads_3_interleave_0, values = (var_279_cast_fp16, var_279_cast_fp16, var_295_cast_fp16, var_295_cast_fp16, var_311_cast_fp16, var_311_cast_fp16, var_327_cast_fp16, var_327_cast_fp16, var_343_cast_fp16, var_343_cast_fp16, var_359_cast_fp16, var_359_cast_fp16, var_375_cast_fp16, var_375_cast_fp16, var_391_cast_fp16, var_391_cast_fp16))[name = string("value_heads_3_cast_fp16")]; + fp16 var_414_to_fp16 = const()[name = string("op_414_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_415_cast_fp16 = mul(x = mh_q_3_cast_fp16, y = var_414_to_fp16)[name = string("op_415_cast_fp16")]; + bool mh_w_1_transpose_x_0 = const()[name = string("mh_w_1_transpose_x_0"), val = bool(true)]; + bool mh_w_1_transpose_y_0 = const()[name = string("mh_w_1_transpose_y_0"), val = bool(false)]; + tensor mh_w_1_cast_fp16 = matmul(transpose_x = mh_w_1_transpose_x_0, transpose_y = mh_w_1_transpose_y_0, x = var_415_cast_fp16, y = key_heads_3_cast_fp16)[name = string("mh_w_1_cast_fp16")]; + tensor var_423_axes_0 = const()[name = string("op_423_axes_0"), val = tensor([1])]; + tensor var_423_cast_fp16 = expand_dims(axes = var_423_axes_0, x = key_padding_mask)[name = string("op_423_cast_fp16")]; + tensor var_424_axes_0 = const()[name = string("op_424_axes_0"), val = tensor([2])]; + tensor var_424_cast_fp16 = expand_dims(axes = var_424_axes_0, x = var_423_cast_fp16)[name = string("op_424_cast_fp16")]; + tensor mh_w_3_cast_fp16 = add(x = mh_w_1_cast_fp16, y = var_424_cast_fp16)[name = string("mh_w_3_cast_fp16")]; + tensor var_427_cast_fp16 = softmax(axis = var_99, x = mh_w_3_cast_fp16)[name = string("op_427_cast_fp16")]; + bool attn_1_transpose_x_0 = const()[name = string("attn_1_transpose_x_0"), val = bool(false)]; + bool attn_1_transpose_y_0 = const()[name = string("attn_1_transpose_y_0"), val = bool(true)]; + tensor attn_1_cast_fp16 = matmul(transpose_x = attn_1_transpose_x_0, transpose_y = attn_1_transpose_y_0, x = value_heads_3_cast_fp16, y = var_427_cast_fp16)[name = string("attn_1_cast_fp16")]; + tensor var_432 = const()[name = string("op_432"), val = tensor([1, -1, 1, 1])]; + tensor input_1_cast_fp16 = reshape(shape = var_432, x = attn_1_cast_fp16)[name = string("input_1_cast_fp16")]; + string obj_11_pad_type_0 = const()[name = string("obj_11_pad_type_0"), val = string("valid")]; + tensor obj_11_strides_0 = const()[name = string("obj_11_strides_0"), val = tensor([1, 1])]; + tensor obj_11_pad_0 = const()[name = string("obj_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_11_dilations_0 = const()[name = string("obj_11_dilations_0"), val = tensor([1, 1])]; + int32 obj_11_groups_0 = const()[name = string("obj_11_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4213632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6310848))))[name = string("layers_0_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_11_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_11_dilations_0, groups = obj_11_groups_0, pad = obj_11_pad_0, pad_type = obj_11_pad_type_0, strides = obj_11_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_1_cast_fp16)[name = string("obj_11_cast_fp16")]; + tensor inputs_5_cast_fp16 = add(x = input_embeds, y = obj_11_cast_fp16)[name = string("inputs_5_cast_fp16")]; + tensor inputs_sq_7_cast_fp16 = mul(x = inputs_5_cast_fp16, y = inputs_5_cast_fp16)[name = string("inputs_sq_7_cast_fp16")]; + tensor variance_7_axes_0 = const()[name = string("variance_7_axes_0"), val = tensor([1])]; + bool variance_7_keep_dims_0 = const()[name = string("variance_7_keep_dims_0"), val = bool(true)]; + tensor variance_7_cast_fp16 = reduce_mean(axes = variance_7_axes_0, keep_dims = variance_7_keep_dims_0, x = inputs_sq_7_cast_fp16)[name = string("variance_7_cast_fp16")]; + fp16 var_450_to_fp16 = const()[name = string("op_450_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_451_cast_fp16 = add(x = variance_7_cast_fp16, y = var_450_to_fp16)[name = string("op_451_cast_fp16")]; + fp32 var_452_epsilon_0 = const()[name = string("op_452_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_452_cast_fp16 = rsqrt(epsilon = var_452_epsilon_0, x = var_451_cast_fp16)[name = string("op_452_cast_fp16")]; + tensor hidden_states_7_cast_fp16 = mul(x = inputs_5_cast_fp16, y = var_452_cast_fp16)[name = string("hidden_states_7_cast_fp16")]; + tensor w_7_to_fp16 = const()[name = string("w_7_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6311424)))]; + tensor input_3_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_7_cast_fp16)[name = string("input_3_cast_fp16")]; + string input_5_pad_type_0 = const()[name = string("input_5_pad_type_0"), val = string("valid")]; + tensor input_5_strides_0 = const()[name = string("input_5_strides_0"), val = tensor([1, 1])]; + tensor input_5_pad_0 = const()[name = string("input_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_5_dilations_0 = const()[name = string("input_5_dilations_0"), val = tensor([1, 1])]; + int32 input_5_groups_0 = const()[name = string("input_5_groups_0"), val = int32(1)]; + tensor layers_0_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6313536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9459328))))[name = string("layers_0_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_5_cast_fp16 = conv(dilations = input_5_dilations_0, groups = input_5_groups_0, pad = input_5_pad_0, pad_type = input_5_pad_type_0, strides = input_5_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_3_cast_fp16)[name = string("input_5_cast_fp16")]; + tensor var_466_cast_fp16 = silu(x = input_5_cast_fp16)[name = string("op_466_cast_fp16")]; + string var_472_pad_type_0 = const()[name = string("op_472_pad_type_0"), val = string("valid")]; + tensor var_472_strides_0 = const()[name = string("op_472_strides_0"), val = tensor([1, 1])]; + tensor var_472_pad_0 = const()[name = string("op_472_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_472_dilations_0 = const()[name = string("op_472_dilations_0"), val = tensor([1, 1])]; + int32 var_472_groups_0 = const()[name = string("op_472_groups_0"), val = int32(1)]; + tensor layers_0_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(9459904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12605696))))[name = string("layers_0_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_472_cast_fp16 = conv(dilations = var_472_dilations_0, groups = var_472_groups_0, pad = var_472_pad_0, pad_type = var_472_pad_type_0, strides = var_472_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_3_cast_fp16)[name = string("op_472_cast_fp16")]; + tensor input_7_cast_fp16 = mul(x = var_466_cast_fp16, y = var_472_cast_fp16)[name = string("input_7_cast_fp16")]; + string hidden_states_9_pad_type_0 = const()[name = string("hidden_states_9_pad_type_0"), val = string("valid")]; + tensor hidden_states_9_strides_0 = const()[name = string("hidden_states_9_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_9_pad_0 = const()[name = string("hidden_states_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_9_dilations_0 = const()[name = string("hidden_states_9_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_9_groups_0 = const()[name = string("hidden_states_9_groups_0"), val = int32(1)]; + tensor layers_0_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12606272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15752064))))[name = string("layers_0_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_9_cast_fp16 = conv(dilations = hidden_states_9_dilations_0, groups = hidden_states_9_groups_0, pad = hidden_states_9_pad_0, pad_type = hidden_states_9_pad_type_0, strides = hidden_states_9_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; + tensor inputs_7_cast_fp16 = add(x = inputs_5_cast_fp16, y = hidden_states_9_cast_fp16)[name = string("inputs_7_cast_fp16")]; + int32 var_486 = const()[name = string("op_486"), val = int32(3)]; + int32 var_496 = const()[name = string("op_496"), val = int32(-2)]; + int32 var_504 = const()[name = string("op_504"), val = int32(1)]; + tensor inputs_sq_9_cast_fp16 = mul(x = inputs_7_cast_fp16, y = inputs_7_cast_fp16)[name = string("inputs_sq_9_cast_fp16")]; + tensor variance_9_axes_0 = const()[name = string("variance_9_axes_0"), val = tensor([1])]; + bool variance_9_keep_dims_0 = const()[name = string("variance_9_keep_dims_0"), val = bool(true)]; + tensor variance_9_cast_fp16 = reduce_mean(axes = variance_9_axes_0, keep_dims = variance_9_keep_dims_0, x = inputs_sq_9_cast_fp16)[name = string("variance_9_cast_fp16")]; + fp16 var_516_to_fp16 = const()[name = string("op_516_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_517_cast_fp16 = add(x = variance_9_cast_fp16, y = var_516_to_fp16)[name = string("op_517_cast_fp16")]; + fp32 var_518_epsilon_0 = const()[name = string("op_518_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_518_cast_fp16 = rsqrt(epsilon = var_518_epsilon_0, x = var_517_cast_fp16)[name = string("op_518_cast_fp16")]; + tensor hidden_states_11_cast_fp16 = mul(x = inputs_7_cast_fp16, y = var_518_cast_fp16)[name = string("hidden_states_11_cast_fp16")]; + tensor w_9_to_fp16 = const()[name = string("w_9_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15752640)))]; + tensor obj_13_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_11_cast_fp16)[name = string("obj_13_cast_fp16")]; + string query_7_pad_type_0 = const()[name = string("query_7_pad_type_0"), val = string("valid")]; + tensor query_7_strides_0 = const()[name = string("query_7_strides_0"), val = tensor([1, 1])]; + tensor query_7_pad_0 = const()[name = string("query_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_7_dilations_0 = const()[name = string("query_7_dilations_0"), val = tensor([1, 1])]; + int32 query_7_groups_0 = const()[name = string("query_7_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(15754752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17851968))))[name = string("layers_1_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor query_7_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_7_dilations_0, groups = query_7_groups_0, pad = query_7_pad_0, pad_type = query_7_pad_type_0, strides = query_7_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_13_cast_fp16)[name = string("query_7_cast_fp16")]; + string current_key_5_pad_type_0 = const()[name = string("current_key_5_pad_type_0"), val = string("valid")]; + tensor current_key_5_strides_0 = const()[name = string("current_key_5_strides_0"), val = tensor([1, 1])]; + tensor current_key_5_pad_0 = const()[name = string("current_key_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_5_dilations_0 = const()[name = string("current_key_5_dilations_0"), val = tensor([1, 1])]; + int32 current_key_5_groups_0 = const()[name = string("current_key_5_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17852544))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18901184))))[name = string("layers_1_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_5_cast_fp16 = conv(dilations = current_key_5_dilations_0, groups = current_key_5_groups_0, pad = current_key_5_pad_0, pad_type = current_key_5_pad_type_0, strides = current_key_5_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_13_cast_fp16)[name = string("current_key_5_cast_fp16")]; + string current_value_3_pad_type_0 = const()[name = string("current_value_3_pad_type_0"), val = string("valid")]; + tensor current_value_3_strides_0 = const()[name = string("current_value_3_strides_0"), val = tensor([1, 1])]; + tensor current_value_3_pad_0 = const()[name = string("current_value_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_3_dilations_0 = const()[name = string("current_value_3_dilations_0"), val = tensor([1, 1])]; + int32 current_value_3_groups_0 = const()[name = string("current_value_3_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18901760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19950400))))[name = string("layers_1_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor current_value_3_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_3_dilations_0, groups = current_value_3_groups_0, pad = current_value_3_pad_0, pad_type = current_value_3_pad_type_0, strides = current_value_3_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_13_cast_fp16)[name = string("current_value_3_cast_fp16")]; + tensor var_555 = const()[name = string("op_555"), val = tensor([16, 128, 1, 1])]; + tensor inputs_9_cast_fp16 = reshape(shape = var_555, x = query_7_cast_fp16)[name = string("inputs_9_cast_fp16")]; + tensor inputs_sq_11_cast_fp16 = mul(x = inputs_9_cast_fp16, y = inputs_9_cast_fp16)[name = string("inputs_sq_11_cast_fp16")]; + tensor variance_11_axes_0 = const()[name = string("variance_11_axes_0"), val = tensor([1])]; + bool variance_11_keep_dims_0 = const()[name = string("variance_11_keep_dims_0"), val = bool(true)]; + tensor variance_11_cast_fp16 = reduce_mean(axes = variance_11_axes_0, keep_dims = variance_11_keep_dims_0, x = inputs_sq_11_cast_fp16)[name = string("variance_11_cast_fp16")]; + fp16 var_561_to_fp16 = const()[name = string("op_561_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_562_cast_fp16 = add(x = variance_11_cast_fp16, y = var_561_to_fp16)[name = string("op_562_cast_fp16")]; + fp32 var_563_epsilon_0 = const()[name = string("op_563_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_563_cast_fp16 = rsqrt(epsilon = var_563_epsilon_0, x = var_562_cast_fp16)[name = string("op_563_cast_fp16")]; + tensor hidden_states_13_cast_fp16 = mul(x = inputs_9_cast_fp16, y = var_563_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; + tensor w_11_to_fp16 = const()[name = string("w_11_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19950976)))]; + tensor query_normed_3_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_13_cast_fp16)[name = string("query_normed_3_cast_fp16")]; + tensor var_571 = const()[name = string("op_571"), val = tensor([8, 128, 1, 1])]; + tensor inputs_11_cast_fp16 = reshape(shape = var_571, x = current_key_5_cast_fp16)[name = string("inputs_11_cast_fp16")]; + tensor inputs_sq_13_cast_fp16 = mul(x = inputs_11_cast_fp16, y = inputs_11_cast_fp16)[name = string("inputs_sq_13_cast_fp16")]; + tensor variance_13_axes_0 = const()[name = string("variance_13_axes_0"), val = tensor([1])]; + bool variance_13_keep_dims_0 = const()[name = string("variance_13_keep_dims_0"), val = bool(true)]; + tensor variance_13_cast_fp16 = reduce_mean(axes = variance_13_axes_0, keep_dims = variance_13_keep_dims_0, x = inputs_sq_13_cast_fp16)[name = string("variance_13_cast_fp16")]; + fp16 var_577_to_fp16 = const()[name = string("op_577_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_578_cast_fp16 = add(x = variance_13_cast_fp16, y = var_577_to_fp16)[name = string("op_578_cast_fp16")]; + fp32 var_579_epsilon_0 = const()[name = string("op_579_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_579_cast_fp16 = rsqrt(epsilon = var_579_epsilon_0, x = var_578_cast_fp16)[name = string("op_579_cast_fp16")]; + tensor hidden_states_15_cast_fp16 = mul(x = inputs_11_cast_fp16, y = var_579_cast_fp16)[name = string("hidden_states_15_cast_fp16")]; + tensor w_13_to_fp16 = const()[name = string("w_13_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19951296)))]; + tensor current_key_normed_3_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_15_cast_fp16)[name = string("current_key_normed_3_cast_fp16")]; + tensor var_597 = const()[name = string("op_597"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_7_cast_fp16 = reshape(shape = var_597, x = query_normed_3_cast_fp16)[name = string("mh_q_7_cast_fp16")]; + tensor var_599 = const()[name = string("op_599"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_5_cast_fp16 = reshape(shape = var_599, x = current_key_normed_3_cast_fp16)[name = string("mh_k_5_cast_fp16")]; + tensor var_603_cast_fp16 = mul(x = mh_q_7_cast_fp16, y = cos_1_cast_fp16)[name = string("op_603_cast_fp16")]; + tensor var_608_begin_0 = const()[name = string("op_608_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_608_end_0 = const()[name = string("op_608_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_608_end_mask_0 = const()[name = string("op_608_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_608_cast_fp16 = slice_by_index(begin = var_608_begin_0, end = var_608_end_0, end_mask = var_608_end_mask_0, x = mh_q_7_cast_fp16)[name = string("op_608_cast_fp16")]; + tensor var_614_begin_0 = const()[name = string("op_614_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_614_end_0 = const()[name = string("op_614_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_614_end_mask_0 = const()[name = string("op_614_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_614_cast_fp16 = slice_by_index(begin = var_614_begin_0, end = var_614_end_0, end_mask = var_614_end_mask_0, x = mh_q_7_cast_fp16)[name = string("op_614_cast_fp16")]; + fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_616_cast_fp16 = mul(x = var_614_cast_fp16, y = const_40_promoted_to_fp16)[name = string("op_616_cast_fp16")]; + bool var_618_interleave_0 = const()[name = string("op_618_interleave_0"), val = bool(false)]; + tensor var_618_cast_fp16 = concat(axis = var_496, interleave = var_618_interleave_0, values = (var_616_cast_fp16, var_608_cast_fp16))[name = string("op_618_cast_fp16")]; + tensor var_619_cast_fp16 = mul(x = var_618_cast_fp16, y = sin_1_cast_fp16)[name = string("op_619_cast_fp16")]; + tensor mh_q_9_cast_fp16 = add(x = var_603_cast_fp16, y = var_619_cast_fp16)[name = string("mh_q_9_cast_fp16")]; + tensor var_621_cast_fp16 = mul(x = mh_k_5_cast_fp16, y = cos_1_cast_fp16)[name = string("op_621_cast_fp16")]; + tensor var_626_begin_0 = const()[name = string("op_626_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_626_end_0 = const()[name = string("op_626_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_626_end_mask_0 = const()[name = string("op_626_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_626_cast_fp16 = slice_by_index(begin = var_626_begin_0, end = var_626_end_0, end_mask = var_626_end_mask_0, x = mh_k_5_cast_fp16)[name = string("op_626_cast_fp16")]; + tensor var_632_begin_0 = const()[name = string("op_632_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_632_end_0 = const()[name = string("op_632_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_632_end_mask_0 = const()[name = string("op_632_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_632_cast_fp16 = slice_by_index(begin = var_632_begin_0, end = var_632_end_0, end_mask = var_632_end_mask_0, x = mh_k_5_cast_fp16)[name = string("op_632_cast_fp16")]; + fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_634_cast_fp16 = mul(x = var_632_cast_fp16, y = const_43_promoted_to_fp16)[name = string("op_634_cast_fp16")]; + bool var_636_interleave_0 = const()[name = string("op_636_interleave_0"), val = bool(false)]; + tensor var_636_cast_fp16 = concat(axis = var_496, interleave = var_636_interleave_0, values = (var_634_cast_fp16, var_626_cast_fp16))[name = string("op_636_cast_fp16")]; + tensor var_637_cast_fp16 = mul(x = var_636_cast_fp16, y = sin_1_cast_fp16)[name = string("op_637_cast_fp16")]; + tensor mh_k_7_cast_fp16 = add(x = var_621_cast_fp16, y = var_637_cast_fp16)[name = string("mh_k_7_cast_fp16")]; + tensor var_641 = const()[name = string("op_641"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_7_cast_fp16 = reshape(shape = var_641, x = mh_k_7_cast_fp16)[name = string("current_key_7_cast_fp16")]; + tensor var_648_cast_fp16 = mul(x = var_84_cast_fp16_1, y = var_260_cast_fp16)[name = string("op_648_cast_fp16")]; + tensor var_649_cast_fp16 = mul(x = current_key_7_cast_fp16, y = var_258_cast_fp16)[name = string("op_649_cast_fp16")]; + tensor key_9_cast_fp16 = add(x = var_648_cast_fp16, y = var_649_cast_fp16)[name = string("key_9_cast_fp16")]; + tensor var_652_cast_fp16 = mul(x = var_92_cast_fp16_1, y = var_260_cast_fp16)[name = string("op_652_cast_fp16")]; + tensor var_653_cast_fp16 = mul(x = current_value_3_cast_fp16, y = var_258_cast_fp16)[name = string("op_653_cast_fp16")]; + tensor value_5_cast_fp16 = add(x = var_652_cast_fp16, y = var_653_cast_fp16)[name = string("value_5_cast_fp16")]; + tensor var_657 = const()[name = string("op_657"), val = tensor([1, 8, 128, 16])]; + tensor key_heads_5_cast_fp16 = reshape(shape = var_657, x = key_9_cast_fp16)[name = string("key_heads_5_cast_fp16")]; + tensor var_659 = const()[name = string("op_659"), val = tensor([1, 8, 128, 16])]; + tensor value_heads_5_cast_fp16 = reshape(shape = var_659, x = value_5_cast_fp16)[name = string("value_heads_5_cast_fp16")]; + tensor var_662_begin_0 = const()[name = string("op_662_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_662_end_0 = const()[name = string("op_662_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_662_end_mask_0 = const()[name = string("op_662_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_662_cast_fp16 = slice_by_index(begin = var_662_begin_0, end = var_662_end_0, end_mask = var_662_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_662_cast_fp16")]; + tensor var_666_begin_0 = const()[name = string("op_666_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_666_end_0 = const()[name = string("op_666_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_666_end_mask_0 = const()[name = string("op_666_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_666_cast_fp16 = slice_by_index(begin = var_666_begin_0, end = var_666_end_0, end_mask = var_666_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_666_cast_fp16")]; + tensor var_678_begin_0 = const()[name = string("op_678_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_678_end_0 = const()[name = string("op_678_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_678_end_mask_0 = const()[name = string("op_678_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_678_cast_fp16 = slice_by_index(begin = var_678_begin_0, end = var_678_end_0, end_mask = var_678_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_678_cast_fp16")]; + tensor var_682_begin_0 = const()[name = string("op_682_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_682_end_0 = const()[name = string("op_682_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_682_end_mask_0 = const()[name = string("op_682_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_682_cast_fp16 = slice_by_index(begin = var_682_begin_0, end = var_682_end_0, end_mask = var_682_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_682_cast_fp16")]; + tensor var_694_begin_0 = const()[name = string("op_694_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_694_end_0 = const()[name = string("op_694_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_694_end_mask_0 = const()[name = string("op_694_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_694_cast_fp16 = slice_by_index(begin = var_694_begin_0, end = var_694_end_0, end_mask = var_694_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_694_cast_fp16")]; + tensor var_698_begin_0 = const()[name = string("op_698_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_698_end_0 = const()[name = string("op_698_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_698_end_mask_0 = const()[name = string("op_698_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_698_cast_fp16 = slice_by_index(begin = var_698_begin_0, end = var_698_end_0, end_mask = var_698_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_698_cast_fp16")]; + tensor var_710_begin_0 = const()[name = string("op_710_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_710_end_0 = const()[name = string("op_710_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_710_end_mask_0 = const()[name = string("op_710_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_710_cast_fp16 = slice_by_index(begin = var_710_begin_0, end = var_710_end_0, end_mask = var_710_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_710_cast_fp16")]; + tensor var_714_begin_0 = const()[name = string("op_714_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_714_end_0 = const()[name = string("op_714_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_714_end_mask_0 = const()[name = string("op_714_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_714_cast_fp16 = slice_by_index(begin = var_714_begin_0, end = var_714_end_0, end_mask = var_714_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_714_cast_fp16")]; + tensor var_726_begin_0 = const()[name = string("op_726_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_726_end_0 = const()[name = string("op_726_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_726_end_mask_0 = const()[name = string("op_726_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_726_cast_fp16 = slice_by_index(begin = var_726_begin_0, end = var_726_end_0, end_mask = var_726_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_726_cast_fp16")]; + tensor var_730_begin_0 = const()[name = string("op_730_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_730_end_0 = const()[name = string("op_730_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_730_end_mask_0 = const()[name = string("op_730_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_730_cast_fp16 = slice_by_index(begin = var_730_begin_0, end = var_730_end_0, end_mask = var_730_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_730_cast_fp16")]; + tensor var_742_begin_0 = const()[name = string("op_742_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_742_end_0 = const()[name = string("op_742_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_742_end_mask_0 = const()[name = string("op_742_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_742_cast_fp16 = slice_by_index(begin = var_742_begin_0, end = var_742_end_0, end_mask = var_742_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_742_cast_fp16")]; + tensor var_746_begin_0 = const()[name = string("op_746_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_746_end_0 = const()[name = string("op_746_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_746_end_mask_0 = const()[name = string("op_746_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_746_cast_fp16 = slice_by_index(begin = var_746_begin_0, end = var_746_end_0, end_mask = var_746_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_746_cast_fp16")]; + tensor var_758_begin_0 = const()[name = string("op_758_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_758_end_0 = const()[name = string("op_758_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_758_end_mask_0 = const()[name = string("op_758_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_758_cast_fp16 = slice_by_index(begin = var_758_begin_0, end = var_758_end_0, end_mask = var_758_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_758_cast_fp16")]; + tensor var_762_begin_0 = const()[name = string("op_762_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_762_end_0 = const()[name = string("op_762_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_762_end_mask_0 = const()[name = string("op_762_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_762_cast_fp16 = slice_by_index(begin = var_762_begin_0, end = var_762_end_0, end_mask = var_762_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_762_cast_fp16")]; + tensor var_774_begin_0 = const()[name = string("op_774_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_774_end_0 = const()[name = string("op_774_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_774_end_mask_0 = const()[name = string("op_774_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_774_cast_fp16 = slice_by_index(begin = var_774_begin_0, end = var_774_end_0, end_mask = var_774_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_774_cast_fp16")]; + tensor var_778_begin_0 = const()[name = string("op_778_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_778_end_0 = const()[name = string("op_778_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_778_end_mask_0 = const()[name = string("op_778_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_778_cast_fp16 = slice_by_index(begin = var_778_begin_0, end = var_778_end_0, end_mask = var_778_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_778_cast_fp16")]; + bool key_heads_7_interleave_0 = const()[name = string("key_heads_7_interleave_0"), val = bool(false)]; + tensor key_heads_7_cast_fp16 = concat(axis = var_504, interleave = key_heads_7_interleave_0, values = (var_662_cast_fp16, var_662_cast_fp16, var_678_cast_fp16, var_678_cast_fp16, var_694_cast_fp16, var_694_cast_fp16, var_710_cast_fp16, var_710_cast_fp16, var_726_cast_fp16, var_726_cast_fp16, var_742_cast_fp16, var_742_cast_fp16, var_758_cast_fp16, var_758_cast_fp16, var_774_cast_fp16, var_774_cast_fp16))[name = string("key_heads_7_cast_fp16")]; + bool value_heads_7_interleave_0 = const()[name = string("value_heads_7_interleave_0"), val = bool(false)]; + tensor value_heads_7_cast_fp16 = concat(axis = var_504, interleave = value_heads_7_interleave_0, values = (var_666_cast_fp16, var_666_cast_fp16, var_682_cast_fp16, var_682_cast_fp16, var_698_cast_fp16, var_698_cast_fp16, var_714_cast_fp16, var_714_cast_fp16, var_730_cast_fp16, var_730_cast_fp16, var_746_cast_fp16, var_746_cast_fp16, var_762_cast_fp16, var_762_cast_fp16, var_778_cast_fp16, var_778_cast_fp16))[name = string("value_heads_7_cast_fp16")]; + fp16 var_801_to_fp16 = const()[name = string("op_801_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_802_cast_fp16 = mul(x = mh_q_9_cast_fp16, y = var_801_to_fp16)[name = string("op_802_cast_fp16")]; + bool mh_w_5_transpose_x_0 = const()[name = string("mh_w_5_transpose_x_0"), val = bool(true)]; + bool mh_w_5_transpose_y_0 = const()[name = string("mh_w_5_transpose_y_0"), val = bool(false)]; + tensor mh_w_5_cast_fp16 = matmul(transpose_x = mh_w_5_transpose_x_0, transpose_y = mh_w_5_transpose_y_0, x = var_802_cast_fp16, y = key_heads_7_cast_fp16)[name = string("mh_w_5_cast_fp16")]; + tensor mh_w_7_cast_fp16 = add(x = mh_w_5_cast_fp16, y = var_424_cast_fp16)[name = string("mh_w_7_cast_fp16")]; + tensor var_814_cast_fp16 = softmax(axis = var_486, x = mh_w_7_cast_fp16)[name = string("op_814_cast_fp16")]; + bool attn_3_transpose_x_0 = const()[name = string("attn_3_transpose_x_0"), val = bool(false)]; + bool attn_3_transpose_y_0 = const()[name = string("attn_3_transpose_y_0"), val = bool(true)]; + tensor attn_3_cast_fp16 = matmul(transpose_x = attn_3_transpose_x_0, transpose_y = attn_3_transpose_y_0, x = value_heads_7_cast_fp16, y = var_814_cast_fp16)[name = string("attn_3_cast_fp16")]; + tensor var_819 = const()[name = string("op_819"), val = tensor([1, -1, 1, 1])]; + tensor input_9_cast_fp16 = reshape(shape = var_819, x = attn_3_cast_fp16)[name = string("input_9_cast_fp16")]; + string obj_19_pad_type_0 = const()[name = string("obj_19_pad_type_0"), val = string("valid")]; + tensor obj_19_strides_0 = const()[name = string("obj_19_strides_0"), val = tensor([1, 1])]; + tensor obj_19_pad_0 = const()[name = string("obj_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_19_dilations_0 = const()[name = string("obj_19_dilations_0"), val = tensor([1, 1])]; + int32 obj_19_groups_0 = const()[name = string("obj_19_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19951616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22048832))))[name = string("layers_1_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_19_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_19_dilations_0, groups = obj_19_groups_0, pad = obj_19_pad_0, pad_type = obj_19_pad_type_0, strides = obj_19_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_9_cast_fp16)[name = string("obj_19_cast_fp16")]; + tensor inputs_13_cast_fp16 = add(x = inputs_7_cast_fp16, y = obj_19_cast_fp16)[name = string("inputs_13_cast_fp16")]; + tensor inputs_sq_15_cast_fp16 = mul(x = inputs_13_cast_fp16, y = inputs_13_cast_fp16)[name = string("inputs_sq_15_cast_fp16")]; + tensor variance_15_axes_0 = const()[name = string("variance_15_axes_0"), val = tensor([1])]; + bool variance_15_keep_dims_0 = const()[name = string("variance_15_keep_dims_0"), val = bool(true)]; + tensor variance_15_cast_fp16 = reduce_mean(axes = variance_15_axes_0, keep_dims = variance_15_keep_dims_0, x = inputs_sq_15_cast_fp16)[name = string("variance_15_cast_fp16")]; + fp16 var_837_to_fp16 = const()[name = string("op_837_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_838_cast_fp16 = add(x = variance_15_cast_fp16, y = var_837_to_fp16)[name = string("op_838_cast_fp16")]; + fp32 var_839_epsilon_0 = const()[name = string("op_839_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_839_cast_fp16 = rsqrt(epsilon = var_839_epsilon_0, x = var_838_cast_fp16)[name = string("op_839_cast_fp16")]; + tensor hidden_states_17_cast_fp16 = mul(x = inputs_13_cast_fp16, y = var_839_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; + tensor w_15_to_fp16 = const()[name = string("w_15_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22049408)))]; + tensor input_11_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_17_cast_fp16)[name = string("input_11_cast_fp16")]; + string input_13_pad_type_0 = const()[name = string("input_13_pad_type_0"), val = string("valid")]; + tensor input_13_strides_0 = const()[name = string("input_13_strides_0"), val = tensor([1, 1])]; + tensor input_13_pad_0 = const()[name = string("input_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_13_dilations_0 = const()[name = string("input_13_dilations_0"), val = tensor([1, 1])]; + int32 input_13_groups_0 = const()[name = string("input_13_groups_0"), val = int32(1)]; + tensor layers_1_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22051520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25197312))))[name = string("layers_1_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_13_cast_fp16 = conv(dilations = input_13_dilations_0, groups = input_13_groups_0, pad = input_13_pad_0, pad_type = input_13_pad_type_0, strides = input_13_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_11_cast_fp16)[name = string("input_13_cast_fp16")]; + tensor var_853_cast_fp16 = silu(x = input_13_cast_fp16)[name = string("op_853_cast_fp16")]; + string var_859_pad_type_0 = const()[name = string("op_859_pad_type_0"), val = string("valid")]; + tensor var_859_strides_0 = const()[name = string("op_859_strides_0"), val = tensor([1, 1])]; + tensor var_859_pad_0 = const()[name = string("op_859_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_859_dilations_0 = const()[name = string("op_859_dilations_0"), val = tensor([1, 1])]; + int32 var_859_groups_0 = const()[name = string("op_859_groups_0"), val = int32(1)]; + tensor layers_1_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25197888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28343680))))[name = string("layers_1_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_859_cast_fp16 = conv(dilations = var_859_dilations_0, groups = var_859_groups_0, pad = var_859_pad_0, pad_type = var_859_pad_type_0, strides = var_859_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_11_cast_fp16)[name = string("op_859_cast_fp16")]; + tensor input_15_cast_fp16 = mul(x = var_853_cast_fp16, y = var_859_cast_fp16)[name = string("input_15_cast_fp16")]; + string hidden_states_19_pad_type_0 = const()[name = string("hidden_states_19_pad_type_0"), val = string("valid")]; + tensor hidden_states_19_strides_0 = const()[name = string("hidden_states_19_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_19_pad_0 = const()[name = string("hidden_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_19_dilations_0 = const()[name = string("hidden_states_19_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_19_groups_0 = const()[name = string("hidden_states_19_groups_0"), val = int32(1)]; + tensor layers_1_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28344256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31490048))))[name = string("layers_1_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_19_cast_fp16 = conv(dilations = hidden_states_19_dilations_0, groups = hidden_states_19_groups_0, pad = hidden_states_19_pad_0, pad_type = hidden_states_19_pad_type_0, strides = hidden_states_19_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_15_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; + tensor inputs_15_cast_fp16 = add(x = inputs_13_cast_fp16, y = hidden_states_19_cast_fp16)[name = string("inputs_15_cast_fp16")]; + int32 var_873 = const()[name = string("op_873"), val = int32(3)]; + int32 var_883 = const()[name = string("op_883"), val = int32(-2)]; + int32 var_891 = const()[name = string("op_891"), val = int32(1)]; + tensor inputs_sq_17_cast_fp16 = mul(x = inputs_15_cast_fp16, y = inputs_15_cast_fp16)[name = string("inputs_sq_17_cast_fp16")]; + tensor variance_17_axes_0 = const()[name = string("variance_17_axes_0"), val = tensor([1])]; + bool variance_17_keep_dims_0 = const()[name = string("variance_17_keep_dims_0"), val = bool(true)]; + tensor variance_17_cast_fp16 = reduce_mean(axes = variance_17_axes_0, keep_dims = variance_17_keep_dims_0, x = inputs_sq_17_cast_fp16)[name = string("variance_17_cast_fp16")]; + fp16 var_903_to_fp16 = const()[name = string("op_903_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_904_cast_fp16 = add(x = variance_17_cast_fp16, y = var_903_to_fp16)[name = string("op_904_cast_fp16")]; + fp32 var_905_epsilon_0 = const()[name = string("op_905_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_905_cast_fp16 = rsqrt(epsilon = var_905_epsilon_0, x = var_904_cast_fp16)[name = string("op_905_cast_fp16")]; + tensor hidden_states_21_cast_fp16 = mul(x = inputs_15_cast_fp16, y = var_905_cast_fp16)[name = string("hidden_states_21_cast_fp16")]; + tensor w_17_to_fp16 = const()[name = string("w_17_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31490624)))]; + tensor obj_21_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_21_cast_fp16)[name = string("obj_21_cast_fp16")]; + string query_13_pad_type_0 = const()[name = string("query_13_pad_type_0"), val = string("valid")]; + tensor query_13_strides_0 = const()[name = string("query_13_strides_0"), val = tensor([1, 1])]; + tensor query_13_pad_0 = const()[name = string("query_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_13_dilations_0 = const()[name = string("query_13_dilations_0"), val = tensor([1, 1])]; + int32 query_13_groups_0 = const()[name = string("query_13_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31492736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33589952))))[name = string("layers_2_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor query_13_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_13_dilations_0, groups = query_13_groups_0, pad = query_13_pad_0, pad_type = query_13_pad_type_0, strides = query_13_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_21_cast_fp16)[name = string("query_13_cast_fp16")]; + string current_key_9_pad_type_0 = const()[name = string("current_key_9_pad_type_0"), val = string("valid")]; + tensor current_key_9_strides_0 = const()[name = string("current_key_9_strides_0"), val = tensor([1, 1])]; + tensor current_key_9_pad_0 = const()[name = string("current_key_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_9_dilations_0 = const()[name = string("current_key_9_dilations_0"), val = tensor([1, 1])]; + int32 current_key_9_groups_0 = const()[name = string("current_key_9_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33590528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34639168))))[name = string("layers_2_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_9_cast_fp16 = conv(dilations = current_key_9_dilations_0, groups = current_key_9_groups_0, pad = current_key_9_pad_0, pad_type = current_key_9_pad_type_0, strides = current_key_9_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_21_cast_fp16)[name = string("current_key_9_cast_fp16")]; + string current_value_5_pad_type_0 = const()[name = string("current_value_5_pad_type_0"), val = string("valid")]; + tensor current_value_5_strides_0 = const()[name = string("current_value_5_strides_0"), val = tensor([1, 1])]; + tensor current_value_5_pad_0 = const()[name = string("current_value_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_5_dilations_0 = const()[name = string("current_value_5_dilations_0"), val = tensor([1, 1])]; + int32 current_value_5_groups_0 = const()[name = string("current_value_5_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34639744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35688384))))[name = string("layers_2_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor current_value_5_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_5_dilations_0, groups = current_value_5_groups_0, pad = current_value_5_pad_0, pad_type = current_value_5_pad_type_0, strides = current_value_5_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_21_cast_fp16)[name = string("current_value_5_cast_fp16")]; + tensor var_942 = const()[name = string("op_942"), val = tensor([16, 128, 1, 1])]; + tensor inputs_17_cast_fp16 = reshape(shape = var_942, x = query_13_cast_fp16)[name = string("inputs_17_cast_fp16")]; + tensor inputs_sq_19_cast_fp16 = mul(x = inputs_17_cast_fp16, y = inputs_17_cast_fp16)[name = string("inputs_sq_19_cast_fp16")]; + tensor variance_19_axes_0 = const()[name = string("variance_19_axes_0"), val = tensor([1])]; + bool variance_19_keep_dims_0 = const()[name = string("variance_19_keep_dims_0"), val = bool(true)]; + tensor variance_19_cast_fp16 = reduce_mean(axes = variance_19_axes_0, keep_dims = variance_19_keep_dims_0, x = inputs_sq_19_cast_fp16)[name = string("variance_19_cast_fp16")]; + fp16 var_948_to_fp16 = const()[name = string("op_948_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_949_cast_fp16 = add(x = variance_19_cast_fp16, y = var_948_to_fp16)[name = string("op_949_cast_fp16")]; + fp32 var_950_epsilon_0 = const()[name = string("op_950_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_950_cast_fp16 = rsqrt(epsilon = var_950_epsilon_0, x = var_949_cast_fp16)[name = string("op_950_cast_fp16")]; + tensor hidden_states_23_cast_fp16 = mul(x = inputs_17_cast_fp16, y = var_950_cast_fp16)[name = string("hidden_states_23_cast_fp16")]; + tensor w_19_to_fp16 = const()[name = string("w_19_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35688960)))]; + tensor query_normed_5_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_23_cast_fp16)[name = string("query_normed_5_cast_fp16")]; + tensor var_958 = const()[name = string("op_958"), val = tensor([8, 128, 1, 1])]; + tensor inputs_19_cast_fp16 = reshape(shape = var_958, x = current_key_9_cast_fp16)[name = string("inputs_19_cast_fp16")]; + tensor inputs_sq_21_cast_fp16 = mul(x = inputs_19_cast_fp16, y = inputs_19_cast_fp16)[name = string("inputs_sq_21_cast_fp16")]; + tensor variance_21_axes_0 = const()[name = string("variance_21_axes_0"), val = tensor([1])]; + bool variance_21_keep_dims_0 = const()[name = string("variance_21_keep_dims_0"), val = bool(true)]; + tensor variance_21_cast_fp16 = reduce_mean(axes = variance_21_axes_0, keep_dims = variance_21_keep_dims_0, x = inputs_sq_21_cast_fp16)[name = string("variance_21_cast_fp16")]; + fp16 var_964_to_fp16 = const()[name = string("op_964_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_965_cast_fp16 = add(x = variance_21_cast_fp16, y = var_964_to_fp16)[name = string("op_965_cast_fp16")]; + fp32 var_966_epsilon_0 = const()[name = string("op_966_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_966_cast_fp16 = rsqrt(epsilon = var_966_epsilon_0, x = var_965_cast_fp16)[name = string("op_966_cast_fp16")]; + tensor hidden_states_25_cast_fp16 = mul(x = inputs_19_cast_fp16, y = var_966_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; + tensor w_21_to_fp16 = const()[name = string("w_21_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35689280)))]; + tensor current_key_normed_5_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_25_cast_fp16)[name = string("current_key_normed_5_cast_fp16")]; + tensor var_984 = const()[name = string("op_984"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_13_cast_fp16 = reshape(shape = var_984, x = query_normed_5_cast_fp16)[name = string("mh_q_13_cast_fp16")]; + tensor var_986 = const()[name = string("op_986"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_9_cast_fp16 = reshape(shape = var_986, x = current_key_normed_5_cast_fp16)[name = string("mh_k_9_cast_fp16")]; + tensor var_990_cast_fp16 = mul(x = mh_q_13_cast_fp16, y = cos_1_cast_fp16)[name = string("op_990_cast_fp16")]; + tensor var_995_begin_0 = const()[name = string("op_995_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_995_end_0 = const()[name = string("op_995_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_995_end_mask_0 = const()[name = string("op_995_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_995_cast_fp16 = slice_by_index(begin = var_995_begin_0, end = var_995_end_0, end_mask = var_995_end_mask_0, x = mh_q_13_cast_fp16)[name = string("op_995_cast_fp16")]; + tensor var_1001_begin_0 = const()[name = string("op_1001_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1001_end_0 = const()[name = string("op_1001_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_1001_end_mask_0 = const()[name = string("op_1001_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1001_cast_fp16 = slice_by_index(begin = var_1001_begin_0, end = var_1001_end_0, end_mask = var_1001_end_mask_0, x = mh_q_13_cast_fp16)[name = string("op_1001_cast_fp16")]; + fp16 const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1003_cast_fp16 = mul(x = var_1001_cast_fp16, y = const_63_promoted_to_fp16)[name = string("op_1003_cast_fp16")]; + bool var_1005_interleave_0 = const()[name = string("op_1005_interleave_0"), val = bool(false)]; + tensor var_1005_cast_fp16 = concat(axis = var_883, interleave = var_1005_interleave_0, values = (var_1003_cast_fp16, var_995_cast_fp16))[name = string("op_1005_cast_fp16")]; + tensor var_1006_cast_fp16 = mul(x = var_1005_cast_fp16, y = sin_1_cast_fp16)[name = string("op_1006_cast_fp16")]; + tensor mh_q_15_cast_fp16 = add(x = var_990_cast_fp16, y = var_1006_cast_fp16)[name = string("mh_q_15_cast_fp16")]; + tensor var_1008_cast_fp16 = mul(x = mh_k_9_cast_fp16, y = cos_1_cast_fp16)[name = string("op_1008_cast_fp16")]; + tensor var_1013_begin_0 = const()[name = string("op_1013_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1013_end_0 = const()[name = string("op_1013_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_1013_end_mask_0 = const()[name = string("op_1013_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1013_cast_fp16 = slice_by_index(begin = var_1013_begin_0, end = var_1013_end_0, end_mask = var_1013_end_mask_0, x = mh_k_9_cast_fp16)[name = string("op_1013_cast_fp16")]; + tensor var_1019_begin_0 = const()[name = string("op_1019_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1019_end_0 = const()[name = string("op_1019_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_1019_end_mask_0 = const()[name = string("op_1019_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1019_cast_fp16 = slice_by_index(begin = var_1019_begin_0, end = var_1019_end_0, end_mask = var_1019_end_mask_0, x = mh_k_9_cast_fp16)[name = string("op_1019_cast_fp16")]; + fp16 const_66_promoted_to_fp16 = const()[name = string("const_66_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1021_cast_fp16 = mul(x = var_1019_cast_fp16, y = const_66_promoted_to_fp16)[name = string("op_1021_cast_fp16")]; + bool var_1023_interleave_0 = const()[name = string("op_1023_interleave_0"), val = bool(false)]; + tensor var_1023_cast_fp16 = concat(axis = var_883, interleave = var_1023_interleave_0, values = (var_1021_cast_fp16, var_1013_cast_fp16))[name = string("op_1023_cast_fp16")]; + tensor var_1024_cast_fp16 = mul(x = var_1023_cast_fp16, y = sin_1_cast_fp16)[name = string("op_1024_cast_fp16")]; + tensor mh_k_11_cast_fp16 = add(x = var_1008_cast_fp16, y = var_1024_cast_fp16)[name = string("mh_k_11_cast_fp16")]; + tensor var_1028 = const()[name = string("op_1028"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_11_cast_fp16 = reshape(shape = var_1028, x = mh_k_11_cast_fp16)[name = string("current_key_11_cast_fp16")]; + tensor var_1035_cast_fp16 = mul(x = var_84_cast_fp16_2, y = var_260_cast_fp16)[name = string("op_1035_cast_fp16")]; + tensor var_1036_cast_fp16 = mul(x = current_key_11_cast_fp16, y = var_258_cast_fp16)[name = string("op_1036_cast_fp16")]; + tensor key_15_cast_fp16 = add(x = var_1035_cast_fp16, y = var_1036_cast_fp16)[name = string("key_15_cast_fp16")]; + tensor var_1039_cast_fp16 = mul(x = var_92_cast_fp16_2, y = var_260_cast_fp16)[name = string("op_1039_cast_fp16")]; + tensor var_1040_cast_fp16 = mul(x = current_value_5_cast_fp16, y = var_258_cast_fp16)[name = string("op_1040_cast_fp16")]; + tensor value_9_cast_fp16 = add(x = var_1039_cast_fp16, y = var_1040_cast_fp16)[name = string("value_9_cast_fp16")]; + tensor var_1044 = const()[name = string("op_1044"), val = tensor([1, 8, 128, 16])]; + tensor key_heads_9_cast_fp16 = reshape(shape = var_1044, x = key_15_cast_fp16)[name = string("key_heads_9_cast_fp16")]; + tensor var_1046 = const()[name = string("op_1046"), val = tensor([1, 8, 128, 16])]; + tensor value_heads_9_cast_fp16 = reshape(shape = var_1046, x = value_9_cast_fp16)[name = string("value_heads_9_cast_fp16")]; + tensor var_1049_begin_0 = const()[name = string("op_1049_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1049_end_0 = const()[name = string("op_1049_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1049_end_mask_0 = const()[name = string("op_1049_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1049_cast_fp16 = slice_by_index(begin = var_1049_begin_0, end = var_1049_end_0, end_mask = var_1049_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1049_cast_fp16")]; + tensor var_1053_begin_0 = const()[name = string("op_1053_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1053_end_0 = const()[name = string("op_1053_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1053_end_mask_0 = const()[name = string("op_1053_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1053_cast_fp16 = slice_by_index(begin = var_1053_begin_0, end = var_1053_end_0, end_mask = var_1053_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1053_cast_fp16")]; + tensor var_1065_begin_0 = const()[name = string("op_1065_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_1065_end_0 = const()[name = string("op_1065_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_1065_end_mask_0 = const()[name = string("op_1065_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1065_cast_fp16 = slice_by_index(begin = var_1065_begin_0, end = var_1065_end_0, end_mask = var_1065_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1065_cast_fp16")]; + tensor var_1069_begin_0 = const()[name = string("op_1069_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_1069_end_0 = const()[name = string("op_1069_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_1069_end_mask_0 = const()[name = string("op_1069_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1069_cast_fp16 = slice_by_index(begin = var_1069_begin_0, end = var_1069_end_0, end_mask = var_1069_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1069_cast_fp16")]; + tensor var_1081_begin_0 = const()[name = string("op_1081_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_1081_end_0 = const()[name = string("op_1081_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_1081_end_mask_0 = const()[name = string("op_1081_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1081_cast_fp16 = slice_by_index(begin = var_1081_begin_0, end = var_1081_end_0, end_mask = var_1081_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1081_cast_fp16")]; + tensor var_1085_begin_0 = const()[name = string("op_1085_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_1085_end_0 = const()[name = string("op_1085_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_1085_end_mask_0 = const()[name = string("op_1085_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1085_cast_fp16 = slice_by_index(begin = var_1085_begin_0, end = var_1085_end_0, end_mask = var_1085_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1085_cast_fp16")]; + tensor var_1097_begin_0 = const()[name = string("op_1097_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_1097_end_0 = const()[name = string("op_1097_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_1097_end_mask_0 = const()[name = string("op_1097_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1097_cast_fp16 = slice_by_index(begin = var_1097_begin_0, end = var_1097_end_0, end_mask = var_1097_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1097_cast_fp16")]; + tensor var_1101_begin_0 = const()[name = string("op_1101_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_1101_end_0 = const()[name = string("op_1101_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_1101_end_mask_0 = const()[name = string("op_1101_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1101_cast_fp16 = slice_by_index(begin = var_1101_begin_0, end = var_1101_end_0, end_mask = var_1101_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1101_cast_fp16")]; + tensor var_1113_begin_0 = const()[name = string("op_1113_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_1113_end_0 = const()[name = string("op_1113_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_1113_end_mask_0 = const()[name = string("op_1113_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1113_cast_fp16 = slice_by_index(begin = var_1113_begin_0, end = var_1113_end_0, end_mask = var_1113_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1113_cast_fp16")]; + tensor var_1117_begin_0 = const()[name = string("op_1117_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_1117_end_0 = const()[name = string("op_1117_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_1117_end_mask_0 = const()[name = string("op_1117_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1117_cast_fp16 = slice_by_index(begin = var_1117_begin_0, end = var_1117_end_0, end_mask = var_1117_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1117_cast_fp16")]; + tensor var_1129_begin_0 = const()[name = string("op_1129_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_1129_end_0 = const()[name = string("op_1129_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_1129_end_mask_0 = const()[name = string("op_1129_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1129_cast_fp16 = slice_by_index(begin = var_1129_begin_0, end = var_1129_end_0, end_mask = var_1129_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1129_cast_fp16")]; + tensor var_1133_begin_0 = const()[name = string("op_1133_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_1133_end_0 = const()[name = string("op_1133_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_1133_end_mask_0 = const()[name = string("op_1133_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1133_cast_fp16 = slice_by_index(begin = var_1133_begin_0, end = var_1133_end_0, end_mask = var_1133_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1133_cast_fp16")]; + tensor var_1145_begin_0 = const()[name = string("op_1145_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_1145_end_0 = const()[name = string("op_1145_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_1145_end_mask_0 = const()[name = string("op_1145_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1145_cast_fp16 = slice_by_index(begin = var_1145_begin_0, end = var_1145_end_0, end_mask = var_1145_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1145_cast_fp16")]; + tensor var_1149_begin_0 = const()[name = string("op_1149_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_1149_end_0 = const()[name = string("op_1149_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_1149_end_mask_0 = const()[name = string("op_1149_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1149_cast_fp16 = slice_by_index(begin = var_1149_begin_0, end = var_1149_end_0, end_mask = var_1149_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1149_cast_fp16")]; + tensor var_1161_begin_0 = const()[name = string("op_1161_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_1161_end_0 = const()[name = string("op_1161_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1161_end_mask_0 = const()[name = string("op_1161_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1161_cast_fp16 = slice_by_index(begin = var_1161_begin_0, end = var_1161_end_0, end_mask = var_1161_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1161_cast_fp16")]; + tensor var_1165_begin_0 = const()[name = string("op_1165_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_1165_end_0 = const()[name = string("op_1165_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1165_end_mask_0 = const()[name = string("op_1165_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1165_cast_fp16 = slice_by_index(begin = var_1165_begin_0, end = var_1165_end_0, end_mask = var_1165_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1165_cast_fp16")]; + bool key_heads_11_interleave_0 = const()[name = string("key_heads_11_interleave_0"), val = bool(false)]; + tensor key_heads_11_cast_fp16 = concat(axis = var_891, interleave = key_heads_11_interleave_0, values = (var_1049_cast_fp16, var_1049_cast_fp16, var_1065_cast_fp16, var_1065_cast_fp16, var_1081_cast_fp16, var_1081_cast_fp16, var_1097_cast_fp16, var_1097_cast_fp16, var_1113_cast_fp16, var_1113_cast_fp16, var_1129_cast_fp16, var_1129_cast_fp16, var_1145_cast_fp16, var_1145_cast_fp16, var_1161_cast_fp16, var_1161_cast_fp16))[name = string("key_heads_11_cast_fp16")]; + bool value_heads_11_interleave_0 = const()[name = string("value_heads_11_interleave_0"), val = bool(false)]; + tensor value_heads_11_cast_fp16 = concat(axis = var_891, interleave = value_heads_11_interleave_0, values = (var_1053_cast_fp16, var_1053_cast_fp16, var_1069_cast_fp16, var_1069_cast_fp16, var_1085_cast_fp16, var_1085_cast_fp16, var_1101_cast_fp16, var_1101_cast_fp16, var_1117_cast_fp16, var_1117_cast_fp16, var_1133_cast_fp16, var_1133_cast_fp16, var_1149_cast_fp16, var_1149_cast_fp16, var_1165_cast_fp16, var_1165_cast_fp16))[name = string("value_heads_11_cast_fp16")]; + fp16 var_1188_to_fp16 = const()[name = string("op_1188_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_1189_cast_fp16 = mul(x = mh_q_15_cast_fp16, y = var_1188_to_fp16)[name = string("op_1189_cast_fp16")]; + bool mh_w_9_transpose_x_0 = const()[name = string("mh_w_9_transpose_x_0"), val = bool(true)]; + bool mh_w_9_transpose_y_0 = const()[name = string("mh_w_9_transpose_y_0"), val = bool(false)]; + tensor mh_w_9_cast_fp16 = matmul(transpose_x = mh_w_9_transpose_x_0, transpose_y = mh_w_9_transpose_y_0, x = var_1189_cast_fp16, y = key_heads_11_cast_fp16)[name = string("mh_w_9_cast_fp16")]; + tensor mh_w_11_cast_fp16 = add(x = mh_w_9_cast_fp16, y = var_424_cast_fp16)[name = string("mh_w_11_cast_fp16")]; + tensor var_1201_cast_fp16 = softmax(axis = var_873, x = mh_w_11_cast_fp16)[name = string("op_1201_cast_fp16")]; + bool attn_5_transpose_x_0 = const()[name = string("attn_5_transpose_x_0"), val = bool(false)]; + bool attn_5_transpose_y_0 = const()[name = string("attn_5_transpose_y_0"), val = bool(true)]; + tensor attn_5_cast_fp16 = matmul(transpose_x = attn_5_transpose_x_0, transpose_y = attn_5_transpose_y_0, x = value_heads_11_cast_fp16, y = var_1201_cast_fp16)[name = string("attn_5_cast_fp16")]; + tensor var_1206 = const()[name = string("op_1206"), val = tensor([1, -1, 1, 1])]; + tensor input_17_cast_fp16 = reshape(shape = var_1206, x = attn_5_cast_fp16)[name = string("input_17_cast_fp16")]; + string obj_27_pad_type_0 = const()[name = string("obj_27_pad_type_0"), val = string("valid")]; + tensor obj_27_strides_0 = const()[name = string("obj_27_strides_0"), val = tensor([1, 1])]; + tensor obj_27_pad_0 = const()[name = string("obj_27_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_27_dilations_0 = const()[name = string("obj_27_dilations_0"), val = tensor([1, 1])]; + int32 obj_27_groups_0 = const()[name = string("obj_27_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35689600))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37786816))))[name = string("layers_2_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_27_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_27_dilations_0, groups = obj_27_groups_0, pad = obj_27_pad_0, pad_type = obj_27_pad_type_0, strides = obj_27_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_17_cast_fp16)[name = string("obj_27_cast_fp16")]; + tensor inputs_21_cast_fp16 = add(x = inputs_15_cast_fp16, y = obj_27_cast_fp16)[name = string("inputs_21_cast_fp16")]; + tensor inputs_sq_23_cast_fp16 = mul(x = inputs_21_cast_fp16, y = inputs_21_cast_fp16)[name = string("inputs_sq_23_cast_fp16")]; + tensor variance_23_axes_0 = const()[name = string("variance_23_axes_0"), val = tensor([1])]; + bool variance_23_keep_dims_0 = const()[name = string("variance_23_keep_dims_0"), val = bool(true)]; + tensor variance_23_cast_fp16 = reduce_mean(axes = variance_23_axes_0, keep_dims = variance_23_keep_dims_0, x = inputs_sq_23_cast_fp16)[name = string("variance_23_cast_fp16")]; + fp16 var_1224_to_fp16 = const()[name = string("op_1224_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1225_cast_fp16 = add(x = variance_23_cast_fp16, y = var_1224_to_fp16)[name = string("op_1225_cast_fp16")]; + fp32 var_1226_epsilon_0 = const()[name = string("op_1226_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1226_cast_fp16 = rsqrt(epsilon = var_1226_epsilon_0, x = var_1225_cast_fp16)[name = string("op_1226_cast_fp16")]; + tensor hidden_states_27_cast_fp16 = mul(x = inputs_21_cast_fp16, y = var_1226_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; + tensor w_23_to_fp16 = const()[name = string("w_23_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37787392)))]; + tensor input_19_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_27_cast_fp16)[name = string("input_19_cast_fp16")]; + string input_21_pad_type_0 = const()[name = string("input_21_pad_type_0"), val = string("valid")]; + tensor input_21_strides_0 = const()[name = string("input_21_strides_0"), val = tensor([1, 1])]; + tensor input_21_pad_0 = const()[name = string("input_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_21_dilations_0 = const()[name = string("input_21_dilations_0"), val = tensor([1, 1])]; + int32 input_21_groups_0 = const()[name = string("input_21_groups_0"), val = int32(1)]; + tensor layers_2_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37789504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40935296))))[name = string("layers_2_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_21_cast_fp16 = conv(dilations = input_21_dilations_0, groups = input_21_groups_0, pad = input_21_pad_0, pad_type = input_21_pad_type_0, strides = input_21_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_19_cast_fp16)[name = string("input_21_cast_fp16")]; + tensor var_1240_cast_fp16 = silu(x = input_21_cast_fp16)[name = string("op_1240_cast_fp16")]; + string var_1246_pad_type_0 = const()[name = string("op_1246_pad_type_0"), val = string("valid")]; + tensor var_1246_strides_0 = const()[name = string("op_1246_strides_0"), val = tensor([1, 1])]; + tensor var_1246_pad_0 = const()[name = string("op_1246_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1246_dilations_0 = const()[name = string("op_1246_dilations_0"), val = tensor([1, 1])]; + int32 var_1246_groups_0 = const()[name = string("op_1246_groups_0"), val = int32(1)]; + tensor layers_2_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(40935872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44081664))))[name = string("layers_2_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_1246_cast_fp16 = conv(dilations = var_1246_dilations_0, groups = var_1246_groups_0, pad = var_1246_pad_0, pad_type = var_1246_pad_type_0, strides = var_1246_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_19_cast_fp16)[name = string("op_1246_cast_fp16")]; + tensor input_23_cast_fp16 = mul(x = var_1240_cast_fp16, y = var_1246_cast_fp16)[name = string("input_23_cast_fp16")]; + string hidden_states_29_pad_type_0 = const()[name = string("hidden_states_29_pad_type_0"), val = string("valid")]; + tensor hidden_states_29_strides_0 = const()[name = string("hidden_states_29_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_29_pad_0 = const()[name = string("hidden_states_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_29_dilations_0 = const()[name = string("hidden_states_29_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_29_groups_0 = const()[name = string("hidden_states_29_groups_0"), val = int32(1)]; + tensor layers_2_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44082240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47228032))))[name = string("layers_2_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_29_cast_fp16 = conv(dilations = hidden_states_29_dilations_0, groups = hidden_states_29_groups_0, pad = hidden_states_29_pad_0, pad_type = hidden_states_29_pad_type_0, strides = hidden_states_29_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_23_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; + tensor inputs_23_cast_fp16 = add(x = inputs_21_cast_fp16, y = hidden_states_29_cast_fp16)[name = string("inputs_23_cast_fp16")]; + int32 var_1260 = const()[name = string("op_1260"), val = int32(3)]; + int32 var_1270 = const()[name = string("op_1270"), val = int32(-2)]; + int32 var_1278 = const()[name = string("op_1278"), val = int32(1)]; + tensor inputs_sq_25_cast_fp16 = mul(x = inputs_23_cast_fp16, y = inputs_23_cast_fp16)[name = string("inputs_sq_25_cast_fp16")]; + tensor variance_25_axes_0 = const()[name = string("variance_25_axes_0"), val = tensor([1])]; + bool variance_25_keep_dims_0 = const()[name = string("variance_25_keep_dims_0"), val = bool(true)]; + tensor variance_25_cast_fp16 = reduce_mean(axes = variance_25_axes_0, keep_dims = variance_25_keep_dims_0, x = inputs_sq_25_cast_fp16)[name = string("variance_25_cast_fp16")]; + fp16 var_1290_to_fp16 = const()[name = string("op_1290_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1291_cast_fp16 = add(x = variance_25_cast_fp16, y = var_1290_to_fp16)[name = string("op_1291_cast_fp16")]; + fp32 var_1292_epsilon_0 = const()[name = string("op_1292_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1292_cast_fp16 = rsqrt(epsilon = var_1292_epsilon_0, x = var_1291_cast_fp16)[name = string("op_1292_cast_fp16")]; + tensor hidden_states_31_cast_fp16 = mul(x = inputs_23_cast_fp16, y = var_1292_cast_fp16)[name = string("hidden_states_31_cast_fp16")]; + tensor w_25_to_fp16 = const()[name = string("w_25_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47228608)))]; + tensor obj_29_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_31_cast_fp16)[name = string("obj_29_cast_fp16")]; + string query_19_pad_type_0 = const()[name = string("query_19_pad_type_0"), val = string("valid")]; + tensor query_19_strides_0 = const()[name = string("query_19_strides_0"), val = tensor([1, 1])]; + tensor query_19_pad_0 = const()[name = string("query_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_19_dilations_0 = const()[name = string("query_19_dilations_0"), val = tensor([1, 1])]; + int32 query_19_groups_0 = const()[name = string("query_19_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47230720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49327936))))[name = string("layers_3_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor query_19_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_19_dilations_0, groups = query_19_groups_0, pad = query_19_pad_0, pad_type = query_19_pad_type_0, strides = query_19_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_29_cast_fp16)[name = string("query_19_cast_fp16")]; + string current_key_13_pad_type_0 = const()[name = string("current_key_13_pad_type_0"), val = string("valid")]; + tensor current_key_13_strides_0 = const()[name = string("current_key_13_strides_0"), val = tensor([1, 1])]; + tensor current_key_13_pad_0 = const()[name = string("current_key_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_13_dilations_0 = const()[name = string("current_key_13_dilations_0"), val = tensor([1, 1])]; + int32 current_key_13_groups_0 = const()[name = string("current_key_13_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49328512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50377152))))[name = string("layers_3_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_13_cast_fp16 = conv(dilations = current_key_13_dilations_0, groups = current_key_13_groups_0, pad = current_key_13_pad_0, pad_type = current_key_13_pad_type_0, strides = current_key_13_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_29_cast_fp16)[name = string("current_key_13_cast_fp16")]; + string current_value_7_pad_type_0 = const()[name = string("current_value_7_pad_type_0"), val = string("valid")]; + tensor current_value_7_strides_0 = const()[name = string("current_value_7_strides_0"), val = tensor([1, 1])]; + tensor current_value_7_pad_0 = const()[name = string("current_value_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_7_dilations_0 = const()[name = string("current_value_7_dilations_0"), val = tensor([1, 1])]; + int32 current_value_7_groups_0 = const()[name = string("current_value_7_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50377728))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51426368))))[name = string("layers_3_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor current_value_7_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_7_dilations_0, groups = current_value_7_groups_0, pad = current_value_7_pad_0, pad_type = current_value_7_pad_type_0, strides = current_value_7_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_29_cast_fp16)[name = string("current_value_7_cast_fp16")]; + tensor var_1329 = const()[name = string("op_1329"), val = tensor([16, 128, 1, 1])]; + tensor inputs_25_cast_fp16 = reshape(shape = var_1329, x = query_19_cast_fp16)[name = string("inputs_25_cast_fp16")]; + tensor inputs_sq_27_cast_fp16 = mul(x = inputs_25_cast_fp16, y = inputs_25_cast_fp16)[name = string("inputs_sq_27_cast_fp16")]; + tensor variance_27_axes_0 = const()[name = string("variance_27_axes_0"), val = tensor([1])]; + bool variance_27_keep_dims_0 = const()[name = string("variance_27_keep_dims_0"), val = bool(true)]; + tensor variance_27_cast_fp16 = reduce_mean(axes = variance_27_axes_0, keep_dims = variance_27_keep_dims_0, x = inputs_sq_27_cast_fp16)[name = string("variance_27_cast_fp16")]; + fp16 var_1335_to_fp16 = const()[name = string("op_1335_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1336_cast_fp16 = add(x = variance_27_cast_fp16, y = var_1335_to_fp16)[name = string("op_1336_cast_fp16")]; + fp32 var_1337_epsilon_0 = const()[name = string("op_1337_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1337_cast_fp16 = rsqrt(epsilon = var_1337_epsilon_0, x = var_1336_cast_fp16)[name = string("op_1337_cast_fp16")]; + tensor hidden_states_33_cast_fp16 = mul(x = inputs_25_cast_fp16, y = var_1337_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; + tensor w_27_to_fp16 = const()[name = string("w_27_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51426944)))]; + tensor query_normed_7_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_33_cast_fp16)[name = string("query_normed_7_cast_fp16")]; + tensor var_1345 = const()[name = string("op_1345"), val = tensor([8, 128, 1, 1])]; + tensor inputs_27_cast_fp16 = reshape(shape = var_1345, x = current_key_13_cast_fp16)[name = string("inputs_27_cast_fp16")]; + tensor inputs_sq_29_cast_fp16 = mul(x = inputs_27_cast_fp16, y = inputs_27_cast_fp16)[name = string("inputs_sq_29_cast_fp16")]; + tensor variance_29_axes_0 = const()[name = string("variance_29_axes_0"), val = tensor([1])]; + bool variance_29_keep_dims_0 = const()[name = string("variance_29_keep_dims_0"), val = bool(true)]; + tensor variance_29_cast_fp16 = reduce_mean(axes = variance_29_axes_0, keep_dims = variance_29_keep_dims_0, x = inputs_sq_29_cast_fp16)[name = string("variance_29_cast_fp16")]; + fp16 var_1351_to_fp16 = const()[name = string("op_1351_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1352_cast_fp16 = add(x = variance_29_cast_fp16, y = var_1351_to_fp16)[name = string("op_1352_cast_fp16")]; + fp32 var_1353_epsilon_0 = const()[name = string("op_1353_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1353_cast_fp16 = rsqrt(epsilon = var_1353_epsilon_0, x = var_1352_cast_fp16)[name = string("op_1353_cast_fp16")]; + tensor hidden_states_35_cast_fp16 = mul(x = inputs_27_cast_fp16, y = var_1353_cast_fp16)[name = string("hidden_states_35_cast_fp16")]; + tensor w_29_to_fp16 = const()[name = string("w_29_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51427264)))]; + tensor current_key_normed_7_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_35_cast_fp16)[name = string("current_key_normed_7_cast_fp16")]; + tensor var_1371 = const()[name = string("op_1371"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_19_cast_fp16 = reshape(shape = var_1371, x = query_normed_7_cast_fp16)[name = string("mh_q_19_cast_fp16")]; + tensor var_1373 = const()[name = string("op_1373"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_13_cast_fp16 = reshape(shape = var_1373, x = current_key_normed_7_cast_fp16)[name = string("mh_k_13_cast_fp16")]; + tensor var_1377_cast_fp16 = mul(x = mh_q_19_cast_fp16, y = cos_1_cast_fp16)[name = string("op_1377_cast_fp16")]; + tensor var_1382_begin_0 = const()[name = string("op_1382_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1382_end_0 = const()[name = string("op_1382_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_1382_end_mask_0 = const()[name = string("op_1382_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1382_cast_fp16 = slice_by_index(begin = var_1382_begin_0, end = var_1382_end_0, end_mask = var_1382_end_mask_0, x = mh_q_19_cast_fp16)[name = string("op_1382_cast_fp16")]; + tensor var_1388_begin_0 = const()[name = string("op_1388_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1388_end_0 = const()[name = string("op_1388_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_1388_end_mask_0 = const()[name = string("op_1388_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1388_cast_fp16 = slice_by_index(begin = var_1388_begin_0, end = var_1388_end_0, end_mask = var_1388_end_mask_0, x = mh_q_19_cast_fp16)[name = string("op_1388_cast_fp16")]; + fp16 const_86_promoted_to_fp16 = const()[name = string("const_86_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1390_cast_fp16 = mul(x = var_1388_cast_fp16, y = const_86_promoted_to_fp16)[name = string("op_1390_cast_fp16")]; + bool var_1392_interleave_0 = const()[name = string("op_1392_interleave_0"), val = bool(false)]; + tensor var_1392_cast_fp16 = concat(axis = var_1270, interleave = var_1392_interleave_0, values = (var_1390_cast_fp16, var_1382_cast_fp16))[name = string("op_1392_cast_fp16")]; + tensor var_1393_cast_fp16 = mul(x = var_1392_cast_fp16, y = sin_1_cast_fp16)[name = string("op_1393_cast_fp16")]; + tensor mh_q_21_cast_fp16 = add(x = var_1377_cast_fp16, y = var_1393_cast_fp16)[name = string("mh_q_21_cast_fp16")]; + tensor var_1395_cast_fp16 = mul(x = mh_k_13_cast_fp16, y = cos_1_cast_fp16)[name = string("op_1395_cast_fp16")]; + tensor var_1400_begin_0 = const()[name = string("op_1400_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1400_end_0 = const()[name = string("op_1400_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_1400_end_mask_0 = const()[name = string("op_1400_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1400_cast_fp16 = slice_by_index(begin = var_1400_begin_0, end = var_1400_end_0, end_mask = var_1400_end_mask_0, x = mh_k_13_cast_fp16)[name = string("op_1400_cast_fp16")]; + tensor var_1406_begin_0 = const()[name = string("op_1406_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1406_end_0 = const()[name = string("op_1406_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_1406_end_mask_0 = const()[name = string("op_1406_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1406_cast_fp16 = slice_by_index(begin = var_1406_begin_0, end = var_1406_end_0, end_mask = var_1406_end_mask_0, x = mh_k_13_cast_fp16)[name = string("op_1406_cast_fp16")]; + fp16 const_89_promoted_to_fp16 = const()[name = string("const_89_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1408_cast_fp16 = mul(x = var_1406_cast_fp16, y = const_89_promoted_to_fp16)[name = string("op_1408_cast_fp16")]; + bool var_1410_interleave_0 = const()[name = string("op_1410_interleave_0"), val = bool(false)]; + tensor var_1410_cast_fp16 = concat(axis = var_1270, interleave = var_1410_interleave_0, values = (var_1408_cast_fp16, var_1400_cast_fp16))[name = string("op_1410_cast_fp16")]; + tensor var_1411_cast_fp16 = mul(x = var_1410_cast_fp16, y = sin_1_cast_fp16)[name = string("op_1411_cast_fp16")]; + tensor mh_k_15_cast_fp16 = add(x = var_1395_cast_fp16, y = var_1411_cast_fp16)[name = string("mh_k_15_cast_fp16")]; + tensor var_1415 = const()[name = string("op_1415"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_15_cast_fp16 = reshape(shape = var_1415, x = mh_k_15_cast_fp16)[name = string("current_key_15_cast_fp16")]; + tensor var_1422_cast_fp16 = mul(x = var_84_cast_fp16_3, y = var_260_cast_fp16)[name = string("op_1422_cast_fp16")]; + tensor var_1423_cast_fp16 = mul(x = current_key_15_cast_fp16, y = var_258_cast_fp16)[name = string("op_1423_cast_fp16")]; + tensor key_21_cast_fp16 = add(x = var_1422_cast_fp16, y = var_1423_cast_fp16)[name = string("key_21_cast_fp16")]; + tensor var_1426_cast_fp16 = mul(x = var_92_cast_fp16_3, y = var_260_cast_fp16)[name = string("op_1426_cast_fp16")]; + tensor var_1427_cast_fp16 = mul(x = current_value_7_cast_fp16, y = var_258_cast_fp16)[name = string("op_1427_cast_fp16")]; + tensor value_13_cast_fp16 = add(x = var_1426_cast_fp16, y = var_1427_cast_fp16)[name = string("value_13_cast_fp16")]; + tensor var_1431 = const()[name = string("op_1431"), val = tensor([1, 8, 128, 16])]; + tensor key_heads_13_cast_fp16 = reshape(shape = var_1431, x = key_21_cast_fp16)[name = string("key_heads_13_cast_fp16")]; + tensor var_1433 = const()[name = string("op_1433"), val = tensor([1, 8, 128, 16])]; + tensor value_heads_13_cast_fp16 = reshape(shape = var_1433, x = value_13_cast_fp16)[name = string("value_heads_13_cast_fp16")]; + tensor var_1436_begin_0 = const()[name = string("op_1436_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1436_end_0 = const()[name = string("op_1436_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1436_end_mask_0 = const()[name = string("op_1436_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1436_cast_fp16 = slice_by_index(begin = var_1436_begin_0, end = var_1436_end_0, end_mask = var_1436_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1436_cast_fp16")]; + tensor var_1440_begin_0 = const()[name = string("op_1440_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1440_end_0 = const()[name = string("op_1440_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1440_end_mask_0 = const()[name = string("op_1440_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1440_cast_fp16 = slice_by_index(begin = var_1440_begin_0, end = var_1440_end_0, end_mask = var_1440_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1440_cast_fp16")]; + tensor var_1452_begin_0 = const()[name = string("op_1452_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_1452_end_0 = const()[name = string("op_1452_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_1452_end_mask_0 = const()[name = string("op_1452_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1452_cast_fp16 = slice_by_index(begin = var_1452_begin_0, end = var_1452_end_0, end_mask = var_1452_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1452_cast_fp16")]; + tensor var_1456_begin_0 = const()[name = string("op_1456_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_1456_end_0 = const()[name = string("op_1456_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_1456_end_mask_0 = const()[name = string("op_1456_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1456_cast_fp16 = slice_by_index(begin = var_1456_begin_0, end = var_1456_end_0, end_mask = var_1456_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1456_cast_fp16")]; + tensor var_1468_begin_0 = const()[name = string("op_1468_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_1468_end_0 = const()[name = string("op_1468_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_1468_end_mask_0 = const()[name = string("op_1468_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1468_cast_fp16 = slice_by_index(begin = var_1468_begin_0, end = var_1468_end_0, end_mask = var_1468_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1468_cast_fp16")]; + tensor var_1472_begin_0 = const()[name = string("op_1472_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_1472_end_0 = const()[name = string("op_1472_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_1472_end_mask_0 = const()[name = string("op_1472_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1472_cast_fp16 = slice_by_index(begin = var_1472_begin_0, end = var_1472_end_0, end_mask = var_1472_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1472_cast_fp16")]; + tensor var_1484_begin_0 = const()[name = string("op_1484_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_1484_end_0 = const()[name = string("op_1484_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_1484_end_mask_0 = const()[name = string("op_1484_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1484_cast_fp16 = slice_by_index(begin = var_1484_begin_0, end = var_1484_end_0, end_mask = var_1484_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1484_cast_fp16")]; + tensor var_1488_begin_0 = const()[name = string("op_1488_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_1488_end_0 = const()[name = string("op_1488_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_1488_end_mask_0 = const()[name = string("op_1488_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1488_cast_fp16 = slice_by_index(begin = var_1488_begin_0, end = var_1488_end_0, end_mask = var_1488_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1488_cast_fp16")]; + tensor var_1500_begin_0 = const()[name = string("op_1500_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_1500_end_0 = const()[name = string("op_1500_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_1500_end_mask_0 = const()[name = string("op_1500_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1500_cast_fp16 = slice_by_index(begin = var_1500_begin_0, end = var_1500_end_0, end_mask = var_1500_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1500_cast_fp16")]; + tensor var_1504_begin_0 = const()[name = string("op_1504_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_1504_end_0 = const()[name = string("op_1504_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_1504_end_mask_0 = const()[name = string("op_1504_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1504_cast_fp16 = slice_by_index(begin = var_1504_begin_0, end = var_1504_end_0, end_mask = var_1504_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1504_cast_fp16")]; + tensor var_1516_begin_0 = const()[name = string("op_1516_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_1516_end_0 = const()[name = string("op_1516_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_1516_end_mask_0 = const()[name = string("op_1516_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1516_cast_fp16 = slice_by_index(begin = var_1516_begin_0, end = var_1516_end_0, end_mask = var_1516_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1516_cast_fp16")]; + tensor var_1520_begin_0 = const()[name = string("op_1520_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_1520_end_0 = const()[name = string("op_1520_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_1520_end_mask_0 = const()[name = string("op_1520_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1520_cast_fp16 = slice_by_index(begin = var_1520_begin_0, end = var_1520_end_0, end_mask = var_1520_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1520_cast_fp16")]; + tensor var_1532_begin_0 = const()[name = string("op_1532_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_1532_end_0 = const()[name = string("op_1532_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_1532_end_mask_0 = const()[name = string("op_1532_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1532_cast_fp16 = slice_by_index(begin = var_1532_begin_0, end = var_1532_end_0, end_mask = var_1532_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1532_cast_fp16")]; + tensor var_1536_begin_0 = const()[name = string("op_1536_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_1536_end_0 = const()[name = string("op_1536_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_1536_end_mask_0 = const()[name = string("op_1536_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1536_cast_fp16 = slice_by_index(begin = var_1536_begin_0, end = var_1536_end_0, end_mask = var_1536_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1536_cast_fp16")]; + tensor var_1548_begin_0 = const()[name = string("op_1548_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_1548_end_0 = const()[name = string("op_1548_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1548_end_mask_0 = const()[name = string("op_1548_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1548_cast_fp16 = slice_by_index(begin = var_1548_begin_0, end = var_1548_end_0, end_mask = var_1548_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1548_cast_fp16")]; + tensor var_1552_begin_0 = const()[name = string("op_1552_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_1552_end_0 = const()[name = string("op_1552_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1552_end_mask_0 = const()[name = string("op_1552_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1552_cast_fp16 = slice_by_index(begin = var_1552_begin_0, end = var_1552_end_0, end_mask = var_1552_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1552_cast_fp16")]; + bool key_heads_15_interleave_0 = const()[name = string("key_heads_15_interleave_0"), val = bool(false)]; + tensor key_heads_15_cast_fp16 = concat(axis = var_1278, interleave = key_heads_15_interleave_0, values = (var_1436_cast_fp16, var_1436_cast_fp16, var_1452_cast_fp16, var_1452_cast_fp16, var_1468_cast_fp16, var_1468_cast_fp16, var_1484_cast_fp16, var_1484_cast_fp16, var_1500_cast_fp16, var_1500_cast_fp16, var_1516_cast_fp16, var_1516_cast_fp16, var_1532_cast_fp16, var_1532_cast_fp16, var_1548_cast_fp16, var_1548_cast_fp16))[name = string("key_heads_15_cast_fp16")]; + bool value_heads_15_interleave_0 = const()[name = string("value_heads_15_interleave_0"), val = bool(false)]; + tensor value_heads_15_cast_fp16 = concat(axis = var_1278, interleave = value_heads_15_interleave_0, values = (var_1440_cast_fp16, var_1440_cast_fp16, var_1456_cast_fp16, var_1456_cast_fp16, var_1472_cast_fp16, var_1472_cast_fp16, var_1488_cast_fp16, var_1488_cast_fp16, var_1504_cast_fp16, var_1504_cast_fp16, var_1520_cast_fp16, var_1520_cast_fp16, var_1536_cast_fp16, var_1536_cast_fp16, var_1552_cast_fp16, var_1552_cast_fp16))[name = string("value_heads_15_cast_fp16")]; + fp16 var_1575_to_fp16 = const()[name = string("op_1575_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_1576_cast_fp16 = mul(x = mh_q_21_cast_fp16, y = var_1575_to_fp16)[name = string("op_1576_cast_fp16")]; + bool mh_w_13_transpose_x_0 = const()[name = string("mh_w_13_transpose_x_0"), val = bool(true)]; + bool mh_w_13_transpose_y_0 = const()[name = string("mh_w_13_transpose_y_0"), val = bool(false)]; + tensor mh_w_13_cast_fp16 = matmul(transpose_x = mh_w_13_transpose_x_0, transpose_y = mh_w_13_transpose_y_0, x = var_1576_cast_fp16, y = key_heads_15_cast_fp16)[name = string("mh_w_13_cast_fp16")]; + tensor mh_w_15_cast_fp16 = add(x = mh_w_13_cast_fp16, y = var_424_cast_fp16)[name = string("mh_w_15_cast_fp16")]; + tensor var_1588_cast_fp16 = softmax(axis = var_1260, x = mh_w_15_cast_fp16)[name = string("op_1588_cast_fp16")]; + bool attn_7_transpose_x_0 = const()[name = string("attn_7_transpose_x_0"), val = bool(false)]; + bool attn_7_transpose_y_0 = const()[name = string("attn_7_transpose_y_0"), val = bool(true)]; + tensor attn_7_cast_fp16 = matmul(transpose_x = attn_7_transpose_x_0, transpose_y = attn_7_transpose_y_0, x = value_heads_15_cast_fp16, y = var_1588_cast_fp16)[name = string("attn_7_cast_fp16")]; + tensor var_1593 = const()[name = string("op_1593"), val = tensor([1, -1, 1, 1])]; + tensor input_25_cast_fp16 = reshape(shape = var_1593, x = attn_7_cast_fp16)[name = string("input_25_cast_fp16")]; + string obj_35_pad_type_0 = const()[name = string("obj_35_pad_type_0"), val = string("valid")]; + tensor obj_35_strides_0 = const()[name = string("obj_35_strides_0"), val = tensor([1, 1])]; + tensor obj_35_pad_0 = const()[name = string("obj_35_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_35_dilations_0 = const()[name = string("obj_35_dilations_0"), val = tensor([1, 1])]; + int32 obj_35_groups_0 = const()[name = string("obj_35_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51427584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53524800))))[name = string("layers_3_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_35_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_35_dilations_0, groups = obj_35_groups_0, pad = obj_35_pad_0, pad_type = obj_35_pad_type_0, strides = obj_35_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_25_cast_fp16)[name = string("obj_35_cast_fp16")]; + tensor inputs_29_cast_fp16 = add(x = inputs_23_cast_fp16, y = obj_35_cast_fp16)[name = string("inputs_29_cast_fp16")]; + tensor inputs_sq_31_cast_fp16 = mul(x = inputs_29_cast_fp16, y = inputs_29_cast_fp16)[name = string("inputs_sq_31_cast_fp16")]; + tensor variance_31_axes_0 = const()[name = string("variance_31_axes_0"), val = tensor([1])]; + bool variance_31_keep_dims_0 = const()[name = string("variance_31_keep_dims_0"), val = bool(true)]; + tensor variance_31_cast_fp16 = reduce_mean(axes = variance_31_axes_0, keep_dims = variance_31_keep_dims_0, x = inputs_sq_31_cast_fp16)[name = string("variance_31_cast_fp16")]; + fp16 var_1611_to_fp16 = const()[name = string("op_1611_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1612_cast_fp16 = add(x = variance_31_cast_fp16, y = var_1611_to_fp16)[name = string("op_1612_cast_fp16")]; + fp32 var_1613_epsilon_0 = const()[name = string("op_1613_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1613_cast_fp16 = rsqrt(epsilon = var_1613_epsilon_0, x = var_1612_cast_fp16)[name = string("op_1613_cast_fp16")]; + tensor hidden_states_37_cast_fp16 = mul(x = inputs_29_cast_fp16, y = var_1613_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; + tensor w_31_to_fp16 = const()[name = string("w_31_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53525376)))]; + tensor input_27_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_37_cast_fp16)[name = string("input_27_cast_fp16")]; + string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; + tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; + tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; + int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; + tensor layers_3_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53527488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(56673280))))[name = string("layers_3_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_29_cast_fp16 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_27_cast_fp16)[name = string("input_29_cast_fp16")]; + tensor var_1627_cast_fp16 = silu(x = input_29_cast_fp16)[name = string("op_1627_cast_fp16")]; + string var_1633_pad_type_0 = const()[name = string("op_1633_pad_type_0"), val = string("valid")]; + tensor var_1633_strides_0 = const()[name = string("op_1633_strides_0"), val = tensor([1, 1])]; + tensor var_1633_pad_0 = const()[name = string("op_1633_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1633_dilations_0 = const()[name = string("op_1633_dilations_0"), val = tensor([1, 1])]; + int32 var_1633_groups_0 = const()[name = string("op_1633_groups_0"), val = int32(1)]; + tensor layers_3_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(56673856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59819648))))[name = string("layers_3_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_1633_cast_fp16 = conv(dilations = var_1633_dilations_0, groups = var_1633_groups_0, pad = var_1633_pad_0, pad_type = var_1633_pad_type_0, strides = var_1633_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_27_cast_fp16)[name = string("op_1633_cast_fp16")]; + tensor input_31_cast_fp16 = mul(x = var_1627_cast_fp16, y = var_1633_cast_fp16)[name = string("input_31_cast_fp16")]; + string hidden_states_39_pad_type_0 = const()[name = string("hidden_states_39_pad_type_0"), val = string("valid")]; + tensor hidden_states_39_strides_0 = const()[name = string("hidden_states_39_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_39_pad_0 = const()[name = string("hidden_states_39_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_39_dilations_0 = const()[name = string("hidden_states_39_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_39_groups_0 = const()[name = string("hidden_states_39_groups_0"), val = int32(1)]; + tensor layers_3_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59820224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62966016))))[name = string("layers_3_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_39_cast_fp16 = conv(dilations = hidden_states_39_dilations_0, groups = hidden_states_39_groups_0, pad = hidden_states_39_pad_0, pad_type = hidden_states_39_pad_type_0, strides = hidden_states_39_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_31_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; + tensor inputs_31_cast_fp16 = add(x = inputs_29_cast_fp16, y = hidden_states_39_cast_fp16)[name = string("inputs_31_cast_fp16")]; + int32 var_1647 = const()[name = string("op_1647"), val = int32(3)]; + int32 var_1657 = const()[name = string("op_1657"), val = int32(-2)]; + int32 var_1665 = const()[name = string("op_1665"), val = int32(1)]; + tensor inputs_sq_33_cast_fp16 = mul(x = inputs_31_cast_fp16, y = inputs_31_cast_fp16)[name = string("inputs_sq_33_cast_fp16")]; + tensor variance_33_axes_0 = const()[name = string("variance_33_axes_0"), val = tensor([1])]; + bool variance_33_keep_dims_0 = const()[name = string("variance_33_keep_dims_0"), val = bool(true)]; + tensor variance_33_cast_fp16 = reduce_mean(axes = variance_33_axes_0, keep_dims = variance_33_keep_dims_0, x = inputs_sq_33_cast_fp16)[name = string("variance_33_cast_fp16")]; + fp16 var_1677_to_fp16 = const()[name = string("op_1677_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1678_cast_fp16 = add(x = variance_33_cast_fp16, y = var_1677_to_fp16)[name = string("op_1678_cast_fp16")]; + fp32 var_1679_epsilon_0 = const()[name = string("op_1679_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1679_cast_fp16 = rsqrt(epsilon = var_1679_epsilon_0, x = var_1678_cast_fp16)[name = string("op_1679_cast_fp16")]; + tensor hidden_states_41_cast_fp16 = mul(x = inputs_31_cast_fp16, y = var_1679_cast_fp16)[name = string("hidden_states_41_cast_fp16")]; + tensor w_33_to_fp16 = const()[name = string("w_33_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62966592)))]; + tensor obj_37_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_41_cast_fp16)[name = string("obj_37_cast_fp16")]; + string query_25_pad_type_0 = const()[name = string("query_25_pad_type_0"), val = string("valid")]; + tensor query_25_strides_0 = const()[name = string("query_25_strides_0"), val = tensor([1, 1])]; + tensor query_25_pad_0 = const()[name = string("query_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_25_dilations_0 = const()[name = string("query_25_dilations_0"), val = tensor([1, 1])]; + int32 query_25_groups_0 = const()[name = string("query_25_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62968704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65065920))))[name = string("layers_4_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor query_25_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_25_dilations_0, groups = query_25_groups_0, pad = query_25_pad_0, pad_type = query_25_pad_type_0, strides = query_25_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_37_cast_fp16)[name = string("query_25_cast_fp16")]; + string current_key_17_pad_type_0 = const()[name = string("current_key_17_pad_type_0"), val = string("valid")]; + tensor current_key_17_strides_0 = const()[name = string("current_key_17_strides_0"), val = tensor([1, 1])]; + tensor current_key_17_pad_0 = const()[name = string("current_key_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_17_dilations_0 = const()[name = string("current_key_17_dilations_0"), val = tensor([1, 1])]; + int32 current_key_17_groups_0 = const()[name = string("current_key_17_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65066496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(66115136))))[name = string("layers_4_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_17_cast_fp16 = conv(dilations = current_key_17_dilations_0, groups = current_key_17_groups_0, pad = current_key_17_pad_0, pad_type = current_key_17_pad_type_0, strides = current_key_17_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_37_cast_fp16)[name = string("current_key_17_cast_fp16")]; + string current_value_pad_type_0 = const()[name = string("current_value_pad_type_0"), val = string("valid")]; + tensor current_value_strides_0 = const()[name = string("current_value_strides_0"), val = tensor([1, 1])]; + tensor current_value_pad_0 = const()[name = string("current_value_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_dilations_0 = const()[name = string("current_value_dilations_0"), val = tensor([1, 1])]; + int32 current_value_groups_0 = const()[name = string("current_value_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(66115712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67164352))))[name = string("layers_4_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor current_value_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_dilations_0, groups = current_value_groups_0, pad = current_value_pad_0, pad_type = current_value_pad_type_0, strides = current_value_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_37_cast_fp16)[name = string("current_value_cast_fp16")]; + tensor var_1716 = const()[name = string("op_1716"), val = tensor([16, 128, 1, 1])]; + tensor inputs_33_cast_fp16 = reshape(shape = var_1716, x = query_25_cast_fp16)[name = string("inputs_33_cast_fp16")]; + tensor inputs_sq_35_cast_fp16 = mul(x = inputs_33_cast_fp16, y = inputs_33_cast_fp16)[name = string("inputs_sq_35_cast_fp16")]; + tensor variance_35_axes_0 = const()[name = string("variance_35_axes_0"), val = tensor([1])]; + bool variance_35_keep_dims_0 = const()[name = string("variance_35_keep_dims_0"), val = bool(true)]; + tensor variance_35_cast_fp16 = reduce_mean(axes = variance_35_axes_0, keep_dims = variance_35_keep_dims_0, x = inputs_sq_35_cast_fp16)[name = string("variance_35_cast_fp16")]; + fp16 var_1722_to_fp16 = const()[name = string("op_1722_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1723_cast_fp16 = add(x = variance_35_cast_fp16, y = var_1722_to_fp16)[name = string("op_1723_cast_fp16")]; + fp32 var_1724_epsilon_0 = const()[name = string("op_1724_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1724_cast_fp16 = rsqrt(epsilon = var_1724_epsilon_0, x = var_1723_cast_fp16)[name = string("op_1724_cast_fp16")]; + tensor hidden_states_43_cast_fp16 = mul(x = inputs_33_cast_fp16, y = var_1724_cast_fp16)[name = string("hidden_states_43_cast_fp16")]; + tensor w_35_to_fp16 = const()[name = string("w_35_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67164928)))]; + tensor query_normed_cast_fp16 = mul(x = w_35_to_fp16, y = hidden_states_43_cast_fp16)[name = string("query_normed_cast_fp16")]; + tensor var_1732 = const()[name = string("op_1732"), val = tensor([8, 128, 1, 1])]; + tensor inputs_35_cast_fp16 = reshape(shape = var_1732, x = current_key_17_cast_fp16)[name = string("inputs_35_cast_fp16")]; + tensor inputs_sq_37_cast_fp16 = mul(x = inputs_35_cast_fp16, y = inputs_35_cast_fp16)[name = string("inputs_sq_37_cast_fp16")]; + tensor variance_37_axes_0 = const()[name = string("variance_37_axes_0"), val = tensor([1])]; + bool variance_37_keep_dims_0 = const()[name = string("variance_37_keep_dims_0"), val = bool(true)]; + tensor variance_37_cast_fp16 = reduce_mean(axes = variance_37_axes_0, keep_dims = variance_37_keep_dims_0, x = inputs_sq_37_cast_fp16)[name = string("variance_37_cast_fp16")]; + fp16 var_1738_to_fp16 = const()[name = string("op_1738_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1739_cast_fp16 = add(x = variance_37_cast_fp16, y = var_1738_to_fp16)[name = string("op_1739_cast_fp16")]; + fp32 var_1740_epsilon_0 = const()[name = string("op_1740_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1740_cast_fp16 = rsqrt(epsilon = var_1740_epsilon_0, x = var_1739_cast_fp16)[name = string("op_1740_cast_fp16")]; + tensor hidden_states_45_cast_fp16 = mul(x = inputs_35_cast_fp16, y = var_1740_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; + tensor w_37_to_fp16 = const()[name = string("w_37_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67165248)))]; + tensor current_key_normed_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_45_cast_fp16)[name = string("current_key_normed_cast_fp16")]; + tensor var_1758 = const()[name = string("op_1758"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_25_cast_fp16 = reshape(shape = var_1758, x = query_normed_cast_fp16)[name = string("mh_q_25_cast_fp16")]; + tensor var_1760 = const()[name = string("op_1760"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_17_cast_fp16 = reshape(shape = var_1760, x = current_key_normed_cast_fp16)[name = string("mh_k_17_cast_fp16")]; + tensor var_1764_cast_fp16 = mul(x = mh_q_25_cast_fp16, y = cos_1_cast_fp16)[name = string("op_1764_cast_fp16")]; + tensor var_1769_begin_0 = const()[name = string("op_1769_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1769_end_0 = const()[name = string("op_1769_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_1769_end_mask_0 = const()[name = string("op_1769_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1769_cast_fp16 = slice_by_index(begin = var_1769_begin_0, end = var_1769_end_0, end_mask = var_1769_end_mask_0, x = mh_q_25_cast_fp16)[name = string("op_1769_cast_fp16")]; + tensor var_1775_begin_0 = const()[name = string("op_1775_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1775_end_0 = const()[name = string("op_1775_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_1775_end_mask_0 = const()[name = string("op_1775_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1775_cast_fp16 = slice_by_index(begin = var_1775_begin_0, end = var_1775_end_0, end_mask = var_1775_end_mask_0, x = mh_q_25_cast_fp16)[name = string("op_1775_cast_fp16")]; + fp16 const_109_promoted_to_fp16 = const()[name = string("const_109_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1777_cast_fp16 = mul(x = var_1775_cast_fp16, y = const_109_promoted_to_fp16)[name = string("op_1777_cast_fp16")]; + bool var_1779_interleave_0 = const()[name = string("op_1779_interleave_0"), val = bool(false)]; + tensor var_1779_cast_fp16 = concat(axis = var_1657, interleave = var_1779_interleave_0, values = (var_1777_cast_fp16, var_1769_cast_fp16))[name = string("op_1779_cast_fp16")]; + tensor var_1780_cast_fp16 = mul(x = var_1779_cast_fp16, y = sin_1_cast_fp16)[name = string("op_1780_cast_fp16")]; + tensor mh_q_27_cast_fp16 = add(x = var_1764_cast_fp16, y = var_1780_cast_fp16)[name = string("mh_q_27_cast_fp16")]; + tensor var_1782_cast_fp16 = mul(x = mh_k_17_cast_fp16, y = cos_1_cast_fp16)[name = string("op_1782_cast_fp16")]; + tensor var_1787_begin_0 = const()[name = string("op_1787_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1787_end_0 = const()[name = string("op_1787_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_1787_end_mask_0 = const()[name = string("op_1787_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1787_cast_fp16 = slice_by_index(begin = var_1787_begin_0, end = var_1787_end_0, end_mask = var_1787_end_mask_0, x = mh_k_17_cast_fp16)[name = string("op_1787_cast_fp16")]; + tensor var_1793_begin_0 = const()[name = string("op_1793_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1793_end_0 = const()[name = string("op_1793_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_1793_end_mask_0 = const()[name = string("op_1793_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1793_cast_fp16 = slice_by_index(begin = var_1793_begin_0, end = var_1793_end_0, end_mask = var_1793_end_mask_0, x = mh_k_17_cast_fp16)[name = string("op_1793_cast_fp16")]; + fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1795_cast_fp16 = mul(x = var_1793_cast_fp16, y = const_112_promoted_to_fp16)[name = string("op_1795_cast_fp16")]; + bool var_1797_interleave_0 = const()[name = string("op_1797_interleave_0"), val = bool(false)]; + tensor var_1797_cast_fp16 = concat(axis = var_1657, interleave = var_1797_interleave_0, values = (var_1795_cast_fp16, var_1787_cast_fp16))[name = string("op_1797_cast_fp16")]; + tensor var_1798_cast_fp16 = mul(x = var_1797_cast_fp16, y = sin_1_cast_fp16)[name = string("op_1798_cast_fp16")]; + tensor mh_k_cast_fp16 = add(x = var_1782_cast_fp16, y = var_1798_cast_fp16)[name = string("mh_k_cast_fp16")]; + tensor var_1802 = const()[name = string("op_1802"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_cast_fp16 = reshape(shape = var_1802, x = mh_k_cast_fp16)[name = string("current_key_cast_fp16")]; + tensor var_1809_cast_fp16 = mul(x = var_84_cast_fp16_4, y = var_260_cast_fp16)[name = string("op_1809_cast_fp16")]; + tensor var_1810_cast_fp16 = mul(x = current_key_cast_fp16, y = var_258_cast_fp16)[name = string("op_1810_cast_fp16")]; + tensor key_27_cast_fp16 = add(x = var_1809_cast_fp16, y = var_1810_cast_fp16)[name = string("key_27_cast_fp16")]; + tensor var_1813_cast_fp16 = mul(x = var_92_cast_fp16_4, y = var_260_cast_fp16)[name = string("op_1813_cast_fp16")]; + tensor var_1814_cast_fp16 = mul(x = current_value_cast_fp16, y = var_258_cast_fp16)[name = string("op_1814_cast_fp16")]; + tensor value_17_cast_fp16 = add(x = var_1813_cast_fp16, y = var_1814_cast_fp16)[name = string("value_17_cast_fp16")]; + tensor var_1818 = const()[name = string("op_1818"), val = tensor([1, 8, 128, 16])]; + tensor key_heads_17_cast_fp16 = reshape(shape = var_1818, x = key_27_cast_fp16)[name = string("key_heads_17_cast_fp16")]; + tensor var_1820 = const()[name = string("op_1820"), val = tensor([1, 8, 128, 16])]; + tensor value_heads_17_cast_fp16 = reshape(shape = var_1820, x = value_17_cast_fp16)[name = string("value_heads_17_cast_fp16")]; + tensor var_1823_begin_0 = const()[name = string("op_1823_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1823_end_0 = const()[name = string("op_1823_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1823_end_mask_0 = const()[name = string("op_1823_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1823_cast_fp16 = slice_by_index(begin = var_1823_begin_0, end = var_1823_end_0, end_mask = var_1823_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1823_cast_fp16")]; + tensor var_1827_begin_0 = const()[name = string("op_1827_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1827_end_0 = const()[name = string("op_1827_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1827_end_mask_0 = const()[name = string("op_1827_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1827_cast_fp16 = slice_by_index(begin = var_1827_begin_0, end = var_1827_end_0, end_mask = var_1827_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1827_cast_fp16")]; + tensor var_1839_begin_0 = const()[name = string("op_1839_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_1839_end_0 = const()[name = string("op_1839_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_1839_end_mask_0 = const()[name = string("op_1839_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1839_cast_fp16 = slice_by_index(begin = var_1839_begin_0, end = var_1839_end_0, end_mask = var_1839_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1839_cast_fp16")]; + tensor var_1843_begin_0 = const()[name = string("op_1843_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_1843_end_0 = const()[name = string("op_1843_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_1843_end_mask_0 = const()[name = string("op_1843_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1843_cast_fp16 = slice_by_index(begin = var_1843_begin_0, end = var_1843_end_0, end_mask = var_1843_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1843_cast_fp16")]; + tensor var_1855_begin_0 = const()[name = string("op_1855_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_1855_end_0 = const()[name = string("op_1855_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_1855_end_mask_0 = const()[name = string("op_1855_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1855_cast_fp16 = slice_by_index(begin = var_1855_begin_0, end = var_1855_end_0, end_mask = var_1855_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1855_cast_fp16")]; + tensor var_1859_begin_0 = const()[name = string("op_1859_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_1859_end_0 = const()[name = string("op_1859_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_1859_end_mask_0 = const()[name = string("op_1859_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1859_cast_fp16 = slice_by_index(begin = var_1859_begin_0, end = var_1859_end_0, end_mask = var_1859_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1859_cast_fp16")]; + tensor var_1871_begin_0 = const()[name = string("op_1871_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_1871_end_0 = const()[name = string("op_1871_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_1871_end_mask_0 = const()[name = string("op_1871_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1871_cast_fp16 = slice_by_index(begin = var_1871_begin_0, end = var_1871_end_0, end_mask = var_1871_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1871_cast_fp16")]; + tensor var_1875_begin_0 = const()[name = string("op_1875_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_1875_end_0 = const()[name = string("op_1875_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_1875_end_mask_0 = const()[name = string("op_1875_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1875_cast_fp16 = slice_by_index(begin = var_1875_begin_0, end = var_1875_end_0, end_mask = var_1875_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1875_cast_fp16")]; + tensor var_1887_begin_0 = const()[name = string("op_1887_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_1887_end_0 = const()[name = string("op_1887_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_1887_end_mask_0 = const()[name = string("op_1887_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1887_cast_fp16 = slice_by_index(begin = var_1887_begin_0, end = var_1887_end_0, end_mask = var_1887_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1887_cast_fp16")]; + tensor var_1891_begin_0 = const()[name = string("op_1891_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_1891_end_0 = const()[name = string("op_1891_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_1891_end_mask_0 = const()[name = string("op_1891_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1891_cast_fp16 = slice_by_index(begin = var_1891_begin_0, end = var_1891_end_0, end_mask = var_1891_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1891_cast_fp16")]; + tensor var_1903_begin_0 = const()[name = string("op_1903_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_1903_end_0 = const()[name = string("op_1903_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_1903_end_mask_0 = const()[name = string("op_1903_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1903_cast_fp16 = slice_by_index(begin = var_1903_begin_0, end = var_1903_end_0, end_mask = var_1903_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1903_cast_fp16")]; + tensor var_1907_begin_0 = const()[name = string("op_1907_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_1907_end_0 = const()[name = string("op_1907_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_1907_end_mask_0 = const()[name = string("op_1907_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1907_cast_fp16 = slice_by_index(begin = var_1907_begin_0, end = var_1907_end_0, end_mask = var_1907_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1907_cast_fp16")]; + tensor var_1919_begin_0 = const()[name = string("op_1919_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_1919_end_0 = const()[name = string("op_1919_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_1919_end_mask_0 = const()[name = string("op_1919_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1919_cast_fp16 = slice_by_index(begin = var_1919_begin_0, end = var_1919_end_0, end_mask = var_1919_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1919_cast_fp16")]; + tensor var_1923_begin_0 = const()[name = string("op_1923_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_1923_end_0 = const()[name = string("op_1923_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_1923_end_mask_0 = const()[name = string("op_1923_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1923_cast_fp16 = slice_by_index(begin = var_1923_begin_0, end = var_1923_end_0, end_mask = var_1923_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1923_cast_fp16")]; + tensor var_1935_begin_0 = const()[name = string("op_1935_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_1935_end_0 = const()[name = string("op_1935_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1935_end_mask_0 = const()[name = string("op_1935_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1935_cast_fp16 = slice_by_index(begin = var_1935_begin_0, end = var_1935_end_0, end_mask = var_1935_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1935_cast_fp16")]; + tensor var_1939_begin_0 = const()[name = string("op_1939_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_1939_end_0 = const()[name = string("op_1939_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1939_end_mask_0 = const()[name = string("op_1939_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1939_cast_fp16 = slice_by_index(begin = var_1939_begin_0, end = var_1939_end_0, end_mask = var_1939_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1939_cast_fp16")]; + bool key_heads_interleave_0 = const()[name = string("key_heads_interleave_0"), val = bool(false)]; + tensor key_heads_cast_fp16 = concat(axis = var_1665, interleave = key_heads_interleave_0, values = (var_1823_cast_fp16, var_1823_cast_fp16, var_1839_cast_fp16, var_1839_cast_fp16, var_1855_cast_fp16, var_1855_cast_fp16, var_1871_cast_fp16, var_1871_cast_fp16, var_1887_cast_fp16, var_1887_cast_fp16, var_1903_cast_fp16, var_1903_cast_fp16, var_1919_cast_fp16, var_1919_cast_fp16, var_1935_cast_fp16, var_1935_cast_fp16))[name = string("key_heads_cast_fp16")]; + bool value_heads_interleave_0 = const()[name = string("value_heads_interleave_0"), val = bool(false)]; + tensor value_heads_cast_fp16 = concat(axis = var_1665, interleave = value_heads_interleave_0, values = (var_1827_cast_fp16, var_1827_cast_fp16, var_1843_cast_fp16, var_1843_cast_fp16, var_1859_cast_fp16, var_1859_cast_fp16, var_1875_cast_fp16, var_1875_cast_fp16, var_1891_cast_fp16, var_1891_cast_fp16, var_1907_cast_fp16, var_1907_cast_fp16, var_1923_cast_fp16, var_1923_cast_fp16, var_1939_cast_fp16, var_1939_cast_fp16))[name = string("value_heads_cast_fp16")]; + fp16 var_1962_to_fp16 = const()[name = string("op_1962_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_1963_cast_fp16 = mul(x = mh_q_27_cast_fp16, y = var_1962_to_fp16)[name = string("op_1963_cast_fp16")]; + bool mh_w_17_transpose_x_0 = const()[name = string("mh_w_17_transpose_x_0"), val = bool(true)]; + bool mh_w_17_transpose_y_0 = const()[name = string("mh_w_17_transpose_y_0"), val = bool(false)]; + tensor mh_w_17_cast_fp16 = matmul(transpose_x = mh_w_17_transpose_x_0, transpose_y = mh_w_17_transpose_y_0, x = var_1963_cast_fp16, y = key_heads_cast_fp16)[name = string("mh_w_17_cast_fp16")]; + tensor mh_w_cast_fp16 = add(x = mh_w_17_cast_fp16, y = var_424_cast_fp16)[name = string("mh_w_cast_fp16")]; + tensor var_1975_cast_fp16 = softmax(axis = var_1647, x = mh_w_cast_fp16)[name = string("op_1975_cast_fp16")]; + bool attn_transpose_x_0 = const()[name = string("attn_transpose_x_0"), val = bool(false)]; + bool attn_transpose_y_0 = const()[name = string("attn_transpose_y_0"), val = bool(true)]; + tensor attn_cast_fp16 = matmul(transpose_x = attn_transpose_x_0, transpose_y = attn_transpose_y_0, x = value_heads_cast_fp16, y = var_1975_cast_fp16)[name = string("attn_cast_fp16")]; + tensor var_1980 = const()[name = string("op_1980"), val = tensor([1, -1, 1, 1])]; + tensor input_33_cast_fp16 = reshape(shape = var_1980, x = attn_cast_fp16)[name = string("input_33_cast_fp16")]; + string obj_pad_type_0 = const()[name = string("obj_pad_type_0"), val = string("valid")]; + tensor obj_strides_0 = const()[name = string("obj_strides_0"), val = tensor([1, 1])]; + tensor obj_pad_0 = const()[name = string("obj_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_dilations_0 = const()[name = string("obj_dilations_0"), val = tensor([1, 1])]; + int32 obj_groups_0 = const()[name = string("obj_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67165568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69262784))))[name = string("layers_4_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_dilations_0, groups = obj_groups_0, pad = obj_pad_0, pad_type = obj_pad_type_0, strides = obj_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_33_cast_fp16)[name = string("obj_cast_fp16")]; + tensor inputs_37_cast_fp16 = add(x = inputs_31_cast_fp16, y = obj_cast_fp16)[name = string("inputs_37_cast_fp16")]; + tensor inputs_sq_39_cast_fp16 = mul(x = inputs_37_cast_fp16, y = inputs_37_cast_fp16)[name = string("inputs_sq_39_cast_fp16")]; + tensor variance_39_axes_0 = const()[name = string("variance_39_axes_0"), val = tensor([1])]; + bool variance_39_keep_dims_0 = const()[name = string("variance_39_keep_dims_0"), val = bool(true)]; + tensor variance_39_cast_fp16 = reduce_mean(axes = variance_39_axes_0, keep_dims = variance_39_keep_dims_0, x = inputs_sq_39_cast_fp16)[name = string("variance_39_cast_fp16")]; + fp16 var_1998_to_fp16 = const()[name = string("op_1998_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1999_cast_fp16 = add(x = variance_39_cast_fp16, y = var_1998_to_fp16)[name = string("op_1999_cast_fp16")]; + fp32 var_2000_epsilon_0 = const()[name = string("op_2000_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2000_cast_fp16 = rsqrt(epsilon = var_2000_epsilon_0, x = var_1999_cast_fp16)[name = string("op_2000_cast_fp16")]; + tensor hidden_states_47_cast_fp16 = mul(x = inputs_37_cast_fp16, y = var_2000_cast_fp16)[name = string("hidden_states_47_cast_fp16")]; + tensor w_39_to_fp16 = const()[name = string("w_39_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69263360)))]; + tensor input_35_cast_fp16 = mul(x = w_39_to_fp16, y = hidden_states_47_cast_fp16)[name = string("input_35_cast_fp16")]; + string input_37_pad_type_0 = const()[name = string("input_37_pad_type_0"), val = string("valid")]; + tensor input_37_strides_0 = const()[name = string("input_37_strides_0"), val = tensor([1, 1])]; + tensor input_37_pad_0 = const()[name = string("input_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_37_dilations_0 = const()[name = string("input_37_dilations_0"), val = tensor([1, 1])]; + int32 input_37_groups_0 = const()[name = string("input_37_groups_0"), val = int32(1)]; + tensor layers_4_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69265472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72411264))))[name = string("layers_4_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_37_cast_fp16 = conv(dilations = input_37_dilations_0, groups = input_37_groups_0, pad = input_37_pad_0, pad_type = input_37_pad_type_0, strides = input_37_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_35_cast_fp16)[name = string("input_37_cast_fp16")]; + tensor var_2014_cast_fp16 = silu(x = input_37_cast_fp16)[name = string("op_2014_cast_fp16")]; + string var_2020_pad_type_0 = const()[name = string("op_2020_pad_type_0"), val = string("valid")]; + tensor var_2020_strides_0 = const()[name = string("op_2020_strides_0"), val = tensor([1, 1])]; + tensor var_2020_pad_0 = const()[name = string("op_2020_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2020_dilations_0 = const()[name = string("op_2020_dilations_0"), val = tensor([1, 1])]; + int32 var_2020_groups_0 = const()[name = string("op_2020_groups_0"), val = int32(1)]; + tensor layers_4_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72411840))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75557632))))[name = string("layers_4_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_2020_cast_fp16 = conv(dilations = var_2020_dilations_0, groups = var_2020_groups_0, pad = var_2020_pad_0, pad_type = var_2020_pad_type_0, strides = var_2020_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_35_cast_fp16)[name = string("op_2020_cast_fp16")]; + tensor input_39_cast_fp16 = mul(x = var_2014_cast_fp16, y = var_2020_cast_fp16)[name = string("input_39_cast_fp16")]; + string hidden_states_49_pad_type_0 = const()[name = string("hidden_states_49_pad_type_0"), val = string("valid")]; + tensor hidden_states_49_strides_0 = const()[name = string("hidden_states_49_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_49_pad_0 = const()[name = string("hidden_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_49_dilations_0 = const()[name = string("hidden_states_49_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_49_groups_0 = const()[name = string("hidden_states_49_groups_0"), val = int32(1)]; + tensor layers_4_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75558208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(78704000))))[name = string("layers_4_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_49_cast_fp16 = conv(dilations = hidden_states_49_dilations_0, groups = hidden_states_49_groups_0, pad = hidden_states_49_pad_0, pad_type = hidden_states_49_pad_type_0, strides = hidden_states_49_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_39_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; + tensor inputs_cast_fp16 = add(x = inputs_37_cast_fp16, y = hidden_states_49_cast_fp16)[name = string("inputs_cast_fp16")]; + tensor inputs_sq_cast_fp16 = mul(x = inputs_cast_fp16, y = inputs_cast_fp16)[name = string("inputs_sq_cast_fp16")]; + tensor variance_axes_0 = const()[name = string("variance_axes_0"), val = tensor([1])]; + bool variance_keep_dims_0 = const()[name = string("variance_keep_dims_0"), val = bool(true)]; + tensor variance_cast_fp16 = reduce_mean(axes = variance_axes_0, keep_dims = variance_keep_dims_0, x = inputs_sq_cast_fp16)[name = string("variance_cast_fp16")]; + fp16 var_2041_to_fp16 = const()[name = string("op_2041_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2042_cast_fp16 = add(x = variance_cast_fp16, y = var_2041_to_fp16)[name = string("op_2042_cast_fp16")]; + fp32 var_2043_epsilon_0 = const()[name = string("op_2043_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2043_cast_fp16 = rsqrt(epsilon = var_2043_epsilon_0, x = var_2042_cast_fp16)[name = string("op_2043_cast_fp16")]; + tensor hidden_states_cast_fp16 = mul(x = inputs_cast_fp16, y = var_2043_cast_fp16)[name = string("hidden_states_cast_fp16")]; + tensor w_to_fp16 = const()[name = string("w_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(78704576)))]; + tensor input_cast_fp16 = mul(x = w_to_fp16, y = hidden_states_cast_fp16)[name = string("input_cast_fp16")]; + string logits_1_pad_type_0 = const()[name = string("logits_1_pad_type_0"), val = string("valid")]; + tensor logits_1_strides_0 = const()[name = string("logits_1_strides_0"), val = tensor([1, 1])]; + tensor logits_1_pad_0 = const()[name = string("logits_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_1_dilations_0 = const()[name = string("logits_1_dilations_0"), val = tensor([1, 1])]; + int32 logits_1_groups_0 = const()[name = string("logits_1_groups_0"), val = int32(1)]; + tensor lm_heads_0_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(78706688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80803904))))[name = string("lm_heads_0_weight_to_fp16_palettized")]; + tensor logits_1_cast_fp16 = conv(dilations = logits_1_dilations_0, groups = logits_1_groups_0, pad = logits_1_pad_0, pad_type = logits_1_pad_type_0, strides = logits_1_strides_0, weight = lm_heads_0_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_1_cast_fp16")]; + tensor var_2060_axes_0 = const()[name = string("op_2060_axes_0"), val = tensor([3])]; + tensor var_2060_cast_fp16 = squeeze(axes = var_2060_axes_0, x = logits_1_cast_fp16)[name = string("op_2060_cast_fp16")]; + string logits_3_pad_type_0 = const()[name = string("logits_3_pad_type_0"), val = string("valid")]; + tensor logits_3_strides_0 = const()[name = string("logits_3_strides_0"), val = tensor([1, 1])]; + tensor logits_3_pad_0 = const()[name = string("logits_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_3_dilations_0 = const()[name = string("logits_3_dilations_0"), val = tensor([1, 1])]; + int32 logits_3_groups_0 = const()[name = string("logits_3_groups_0"), val = int32(1)]; + tensor lm_heads_1_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80804480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82901696))))[name = string("lm_heads_1_weight_to_fp16_palettized")]; + tensor logits_3_cast_fp16 = conv(dilations = logits_3_dilations_0, groups = logits_3_groups_0, pad = logits_3_pad_0, pad_type = logits_3_pad_type_0, strides = logits_3_strides_0, weight = lm_heads_1_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_3_cast_fp16")]; + tensor var_2076_axes_0 = const()[name = string("op_2076_axes_0"), val = tensor([3])]; + tensor var_2076_cast_fp16 = squeeze(axes = var_2076_axes_0, x = logits_3_cast_fp16)[name = string("op_2076_cast_fp16")]; + string logits_5_pad_type_0 = const()[name = string("logits_5_pad_type_0"), val = string("valid")]; + tensor logits_5_strides_0 = const()[name = string("logits_5_strides_0"), val = tensor([1, 1])]; + tensor logits_5_pad_0 = const()[name = string("logits_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_5_dilations_0 = const()[name = string("logits_5_dilations_0"), val = tensor([1, 1])]; + int32 logits_5_groups_0 = const()[name = string("logits_5_groups_0"), val = int32(1)]; + tensor lm_heads_2_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82902272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(84999488))))[name = string("lm_heads_2_weight_to_fp16_palettized")]; + tensor logits_5_cast_fp16 = conv(dilations = logits_5_dilations_0, groups = logits_5_groups_0, pad = logits_5_pad_0, pad_type = logits_5_pad_type_0, strides = logits_5_strides_0, weight = lm_heads_2_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_5_cast_fp16")]; + tensor var_2092_axes_0 = const()[name = string("op_2092_axes_0"), val = tensor([3])]; + tensor var_2092_cast_fp16 = squeeze(axes = var_2092_axes_0, x = logits_5_cast_fp16)[name = string("op_2092_cast_fp16")]; + string logits_7_pad_type_0 = const()[name = string("logits_7_pad_type_0"), val = string("valid")]; + tensor logits_7_strides_0 = const()[name = string("logits_7_strides_0"), val = tensor([1, 1])]; + tensor logits_7_pad_0 = const()[name = string("logits_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_7_dilations_0 = const()[name = string("logits_7_dilations_0"), val = tensor([1, 1])]; + int32 logits_7_groups_0 = const()[name = string("logits_7_groups_0"), val = int32(1)]; + tensor lm_heads_3_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85000064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87097280))))[name = string("lm_heads_3_weight_to_fp16_palettized")]; + tensor logits_7_cast_fp16 = conv(dilations = logits_7_dilations_0, groups = logits_7_groups_0, pad = logits_7_pad_0, pad_type = logits_7_pad_type_0, strides = logits_7_strides_0, weight = lm_heads_3_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_7_cast_fp16")]; + tensor var_2108_axes_0 = const()[name = string("op_2108_axes_0"), val = tensor([3])]; + tensor var_2108_cast_fp16 = squeeze(axes = var_2108_axes_0, x = logits_7_cast_fp16)[name = string("op_2108_cast_fp16")]; + string logits_9_pad_type_0 = const()[name = string("logits_9_pad_type_0"), val = string("valid")]; + tensor logits_9_strides_0 = const()[name = string("logits_9_strides_0"), val = tensor([1, 1])]; + tensor logits_9_pad_0 = const()[name = string("logits_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_9_dilations_0 = const()[name = string("logits_9_dilations_0"), val = tensor([1, 1])]; + int32 logits_9_groups_0 = const()[name = string("logits_9_groups_0"), val = int32(1)]; + tensor lm_heads_4_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87097856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89195072))))[name = string("lm_heads_4_weight_to_fp16_palettized")]; + tensor logits_9_cast_fp16 = conv(dilations = logits_9_dilations_0, groups = logits_9_groups_0, pad = logits_9_pad_0, pad_type = logits_9_pad_type_0, strides = logits_9_strides_0, weight = lm_heads_4_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_9_cast_fp16")]; + tensor var_2124_axes_0 = const()[name = string("op_2124_axes_0"), val = tensor([3])]; + tensor var_2124_cast_fp16 = squeeze(axes = var_2124_axes_0, x = logits_9_cast_fp16)[name = string("op_2124_cast_fp16")]; + string logits_11_pad_type_0 = const()[name = string("logits_11_pad_type_0"), val = string("valid")]; + tensor logits_11_strides_0 = const()[name = string("logits_11_strides_0"), val = tensor([1, 1])]; + tensor logits_11_pad_0 = const()[name = string("logits_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_11_dilations_0 = const()[name = string("logits_11_dilations_0"), val = tensor([1, 1])]; + int32 logits_11_groups_0 = const()[name = string("logits_11_groups_0"), val = int32(1)]; + tensor lm_heads_5_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89195648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91292864))))[name = string("lm_heads_5_weight_to_fp16_palettized")]; + tensor logits_11_cast_fp16 = conv(dilations = logits_11_dilations_0, groups = logits_11_groups_0, pad = logits_11_pad_0, pad_type = logits_11_pad_type_0, strides = logits_11_strides_0, weight = lm_heads_5_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_11_cast_fp16")]; + tensor var_2140_axes_0 = const()[name = string("op_2140_axes_0"), val = tensor([3])]; + tensor var_2140_cast_fp16 = squeeze(axes = var_2140_axes_0, x = logits_11_cast_fp16)[name = string("op_2140_cast_fp16")]; + string logits_13_pad_type_0 = const()[name = string("logits_13_pad_type_0"), val = string("valid")]; + tensor logits_13_strides_0 = const()[name = string("logits_13_strides_0"), val = tensor([1, 1])]; + tensor logits_13_pad_0 = const()[name = string("logits_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_13_dilations_0 = const()[name = string("logits_13_dilations_0"), val = tensor([1, 1])]; + int32 logits_13_groups_0 = const()[name = string("logits_13_groups_0"), val = int32(1)]; + tensor lm_heads_6_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91293440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93390656))))[name = string("lm_heads_6_weight_to_fp16_palettized")]; + tensor logits_13_cast_fp16 = conv(dilations = logits_13_dilations_0, groups = logits_13_groups_0, pad = logits_13_pad_0, pad_type = logits_13_pad_type_0, strides = logits_13_strides_0, weight = lm_heads_6_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_13_cast_fp16")]; + tensor var_2156_axes_0 = const()[name = string("op_2156_axes_0"), val = tensor([3])]; + tensor var_2156_cast_fp16 = squeeze(axes = var_2156_axes_0, x = logits_13_cast_fp16)[name = string("op_2156_cast_fp16")]; + string logits_15_pad_type_0 = const()[name = string("logits_15_pad_type_0"), val = string("valid")]; + tensor logits_15_strides_0 = const()[name = string("logits_15_strides_0"), val = tensor([1, 1])]; + tensor logits_15_pad_0 = const()[name = string("logits_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_15_dilations_0 = const()[name = string("logits_15_dilations_0"), val = tensor([1, 1])]; + int32 logits_15_groups_0 = const()[name = string("logits_15_groups_0"), val = int32(1)]; + tensor lm_heads_7_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93391232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95488448))))[name = string("lm_heads_7_weight_to_fp16_palettized")]; + tensor logits_15_cast_fp16 = conv(dilations = logits_15_dilations_0, groups = logits_15_groups_0, pad = logits_15_pad_0, pad_type = logits_15_pad_type_0, strides = logits_15_strides_0, weight = lm_heads_7_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_15_cast_fp16")]; + tensor var_2172_axes_0 = const()[name = string("op_2172_axes_0"), val = tensor([3])]; + tensor var_2172_cast_fp16 = squeeze(axes = var_2172_axes_0, x = logits_15_cast_fp16)[name = string("op_2172_cast_fp16")]; + string logits_17_pad_type_0 = const()[name = string("logits_17_pad_type_0"), val = string("valid")]; + tensor logits_17_strides_0 = const()[name = string("logits_17_strides_0"), val = tensor([1, 1])]; + tensor logits_17_pad_0 = const()[name = string("logits_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_17_dilations_0 = const()[name = string("logits_17_dilations_0"), val = tensor([1, 1])]; + int32 logits_17_groups_0 = const()[name = string("logits_17_groups_0"), val = int32(1)]; + tensor lm_heads_8_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95489024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(97586240))))[name = string("lm_heads_8_weight_to_fp16_palettized")]; + tensor logits_17_cast_fp16 = conv(dilations = logits_17_dilations_0, groups = logits_17_groups_0, pad = logits_17_pad_0, pad_type = logits_17_pad_type_0, strides = logits_17_strides_0, weight = lm_heads_8_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_17_cast_fp16")]; + tensor var_2188_axes_0 = const()[name = string("op_2188_axes_0"), val = tensor([3])]; + tensor var_2188_cast_fp16 = squeeze(axes = var_2188_axes_0, x = logits_17_cast_fp16)[name = string("op_2188_cast_fp16")]; + string logits_19_pad_type_0 = const()[name = string("logits_19_pad_type_0"), val = string("valid")]; + tensor logits_19_strides_0 = const()[name = string("logits_19_strides_0"), val = tensor([1, 1])]; + tensor logits_19_pad_0 = const()[name = string("logits_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_19_dilations_0 = const()[name = string("logits_19_dilations_0"), val = tensor([1, 1])]; + int32 logits_19_groups_0 = const()[name = string("logits_19_groups_0"), val = int32(1)]; + tensor lm_heads_9_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(97586816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(99684032))))[name = string("lm_heads_9_weight_to_fp16_palettized")]; + tensor logits_19_cast_fp16 = conv(dilations = logits_19_dilations_0, groups = logits_19_groups_0, pad = logits_19_pad_0, pad_type = logits_19_pad_type_0, strides = logits_19_strides_0, weight = lm_heads_9_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_19_cast_fp16")]; + tensor var_2204_axes_0 = const()[name = string("op_2204_axes_0"), val = tensor([3])]; + tensor var_2204_cast_fp16 = squeeze(axes = var_2204_axes_0, x = logits_19_cast_fp16)[name = string("op_2204_cast_fp16")]; + string logits_21_pad_type_0 = const()[name = string("logits_21_pad_type_0"), val = string("valid")]; + tensor logits_21_strides_0 = const()[name = string("logits_21_strides_0"), val = tensor([1, 1])]; + tensor logits_21_pad_0 = const()[name = string("logits_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_21_dilations_0 = const()[name = string("logits_21_dilations_0"), val = tensor([1, 1])]; + int32 logits_21_groups_0 = const()[name = string("logits_21_groups_0"), val = int32(1)]; + tensor lm_heads_10_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(99684608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101781824))))[name = string("lm_heads_10_weight_to_fp16_palettized")]; + tensor logits_21_cast_fp16 = conv(dilations = logits_21_dilations_0, groups = logits_21_groups_0, pad = logits_21_pad_0, pad_type = logits_21_pad_type_0, strides = logits_21_strides_0, weight = lm_heads_10_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_21_cast_fp16")]; + tensor var_2220_axes_0 = const()[name = string("op_2220_axes_0"), val = tensor([3])]; + tensor var_2220_cast_fp16 = squeeze(axes = var_2220_axes_0, x = logits_21_cast_fp16)[name = string("op_2220_cast_fp16")]; + string logits_23_pad_type_0 = const()[name = string("logits_23_pad_type_0"), val = string("valid")]; + tensor logits_23_strides_0 = const()[name = string("logits_23_strides_0"), val = tensor([1, 1])]; + tensor logits_23_pad_0 = const()[name = string("logits_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_23_dilations_0 = const()[name = string("logits_23_dilations_0"), val = tensor([1, 1])]; + int32 logits_23_groups_0 = const()[name = string("logits_23_groups_0"), val = int32(1)]; + tensor lm_heads_11_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101782400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103879616))))[name = string("lm_heads_11_weight_to_fp16_palettized")]; + tensor logits_23_cast_fp16 = conv(dilations = logits_23_dilations_0, groups = logits_23_groups_0, pad = logits_23_pad_0, pad_type = logits_23_pad_type_0, strides = logits_23_strides_0, weight = lm_heads_11_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_23_cast_fp16")]; + tensor var_2236_axes_0 = const()[name = string("op_2236_axes_0"), val = tensor([3])]; + tensor var_2236_cast_fp16 = squeeze(axes = var_2236_axes_0, x = logits_23_cast_fp16)[name = string("op_2236_cast_fp16")]; + string logits_25_pad_type_0 = const()[name = string("logits_25_pad_type_0"), val = string("valid")]; + tensor logits_25_strides_0 = const()[name = string("logits_25_strides_0"), val = tensor([1, 1])]; + tensor logits_25_pad_0 = const()[name = string("logits_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_25_dilations_0 = const()[name = string("logits_25_dilations_0"), val = tensor([1, 1])]; + int32 logits_25_groups_0 = const()[name = string("logits_25_groups_0"), val = int32(1)]; + tensor lm_heads_12_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103880192))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105977408))))[name = string("lm_heads_12_weight_to_fp16_palettized")]; + tensor logits_25_cast_fp16 = conv(dilations = logits_25_dilations_0, groups = logits_25_groups_0, pad = logits_25_pad_0, pad_type = logits_25_pad_type_0, strides = logits_25_strides_0, weight = lm_heads_12_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_25_cast_fp16")]; + tensor var_2252_axes_0 = const()[name = string("op_2252_axes_0"), val = tensor([3])]; + tensor var_2252_cast_fp16 = squeeze(axes = var_2252_axes_0, x = logits_25_cast_fp16)[name = string("op_2252_cast_fp16")]; + string logits_27_pad_type_0 = const()[name = string("logits_27_pad_type_0"), val = string("valid")]; + tensor logits_27_strides_0 = const()[name = string("logits_27_strides_0"), val = tensor([1, 1])]; + tensor logits_27_pad_0 = const()[name = string("logits_27_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_27_dilations_0 = const()[name = string("logits_27_dilations_0"), val = tensor([1, 1])]; + int32 logits_27_groups_0 = const()[name = string("logits_27_groups_0"), val = int32(1)]; + tensor lm_heads_13_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105977984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108075200))))[name = string("lm_heads_13_weight_to_fp16_palettized")]; + tensor logits_27_cast_fp16 = conv(dilations = logits_27_dilations_0, groups = logits_27_groups_0, pad = logits_27_pad_0, pad_type = logits_27_pad_type_0, strides = logits_27_strides_0, weight = lm_heads_13_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_27_cast_fp16")]; + tensor var_2268_axes_0 = const()[name = string("op_2268_axes_0"), val = tensor([3])]; + tensor var_2268_cast_fp16 = squeeze(axes = var_2268_axes_0, x = logits_27_cast_fp16)[name = string("op_2268_cast_fp16")]; + string logits_29_pad_type_0 = const()[name = string("logits_29_pad_type_0"), val = string("valid")]; + tensor logits_29_strides_0 = const()[name = string("logits_29_strides_0"), val = tensor([1, 1])]; + tensor logits_29_pad_0 = const()[name = string("logits_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_29_dilations_0 = const()[name = string("logits_29_dilations_0"), val = tensor([1, 1])]; + int32 logits_29_groups_0 = const()[name = string("logits_29_groups_0"), val = int32(1)]; + tensor lm_heads_14_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108075776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110172992))))[name = string("lm_heads_14_weight_to_fp16_palettized")]; + tensor logits_29_cast_fp16 = conv(dilations = logits_29_dilations_0, groups = logits_29_groups_0, pad = logits_29_pad_0, pad_type = logits_29_pad_type_0, strides = logits_29_strides_0, weight = lm_heads_14_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_29_cast_fp16")]; + tensor var_2284_axes_0 = const()[name = string("op_2284_axes_0"), val = tensor([3])]; + tensor var_2284_cast_fp16 = squeeze(axes = var_2284_axes_0, x = logits_29_cast_fp16)[name = string("op_2284_cast_fp16")]; + bool var_2290_interleave_0 = const()[name = string("op_2290_interleave_0"), val = bool(false)]; + int32 const_119 = const()[name = string("const_119"), val = int32(2)]; + tensor var_2290_cast_fp16 = concat(axis = const_119, interleave = var_2290_interleave_0, values = (var_2060_cast_fp16, var_2076_cast_fp16, var_2092_cast_fp16, var_2108_cast_fp16, var_2124_cast_fp16, var_2140_cast_fp16, var_2156_cast_fp16, var_2172_cast_fp16, var_2188_cast_fp16, var_2204_cast_fp16, var_2220_cast_fp16, var_2236_cast_fp16, var_2252_cast_fp16, var_2268_cast_fp16, var_2284_cast_fp16))[name = string("op_2290_cast_fp16")]; + int32 var_2292 = const()[name = string("op_2292"), val = int32(1)]; + bool var_2293_interleave_0 = const()[name = string("op_2293_interleave_0"), val = bool(false)]; + tensor key_cache_updates = concat(axis = var_2292, interleave = var_2293_interleave_0, values = (current_key_3_cast_fp16, current_key_7_cast_fp16, current_key_11_cast_fp16, current_key_15_cast_fp16, current_key_cast_fp16))[name = string("op_2293_cast_fp16")]; + int32 var_2295 = const()[name = string("op_2295"), val = int32(1)]; + bool var_2296_interleave_0 = const()[name = string("op_2296_interleave_0"), val = bool(false)]; + tensor value_cache_updates = concat(axis = var_2295, interleave = var_2296_interleave_0, values = (current_value_1_cast_fp16, current_value_3_cast_fp16, current_value_5_cast_fp16, current_value_7_cast_fp16, current_value_cast_fp16))[name = string("op_2296_cast_fp16")]; + tensor transpose_0_perm_0 = const()[name = string("transpose_0_perm_0"), val = tensor([0, 2, 1])]; + tensor all_logits = transpose(perm = transpose_0_perm_0, x = var_2290_cast_fp16)[name = string("transpose_0")]; + } -> (all_logits, key_cache_updates, value_cache_updates); +} \ No newline at end of file diff --git a/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/weights/weight.bin b/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/weights/weight.bin new file mode 100644 index 0000000000000000000000000000000000000000..005c8b5551ae883d31889435ab0becfb47638665 --- /dev/null +++ b/qwen3_tts/multi_code_decoder/12hz-0.6b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/weights/weight.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bd25b7ef15bf16b72f01f49f04e861cad3885de39cebce1cd1fae420f248e8c0 +size 141647584 diff --git a/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/analytics/coremldata.bin b/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/analytics/coremldata.bin new file mode 100644 index 0000000000000000000000000000000000000000..8060479338e272f82563cc4f40d47c2313b9e369 --- /dev/null +++ b/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/analytics/coremldata.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3cff719e21343a92fc7eedfd5da71c3f4eed536b7d0c6fd3c5129d509052ede5 +size 243 diff --git a/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/coremldata.bin b/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/coremldata.bin new file mode 100644 index 0000000000000000000000000000000000000000..1dd60e76e5250d37e31077f3a29551e070867581 --- /dev/null +++ b/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/coremldata.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ae22d36504bd24d589c7dec4dc9cf2f622fbfb44bbc120ff234253f32917646b +size 624 diff --git a/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/metadata.json b/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..0635f965de7274eb92aa51301680667caa9e94f5 --- /dev/null +++ b/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/metadata.json @@ -0,0 +1,386 @@ +[ + { + "metadataOutputVersion" : "3.0", + "storagePrecision" : "Mixed (Float16, Palettized (8 bits), UInt8)", + "outputSchema" : [ + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 15 × 2048)", + "shortDescription" : "", + "shape" : "[1, 15, 2048]", + "name" : "all_logits", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 1]", + "name" : "key_cache_updates", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 1]", + "name" : "value_cache_updates", + "type" : "MultiArray" + } + ], + "modelParameters" : [ + + ], + "specificationVersion" : 9, + "functions" : [ + { + "inputSchema" : [ + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 2048 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 2048, 1, 1]", + "name" : "input_embeds", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Int32", + "formattedType" : "MultiArray (Int32 1)", + "shortDescription" : "", + "shape" : "[1]", + "name" : "cache_length", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 16]", + "name" : "key_cache", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 16]", + "name" : "value_cache", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 16]", + "name" : "kv_cache_update_mask", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 16]", + "name" : "key_padding_mask", + "type" : "MultiArray" + } + ], + "computePrecision" : "Mixed (Float16, Float32, Int16, Int32, UInt16)", + "storagePrecision" : "Mixed (Float16, Palettized (8 bits), UInt8)", + "stateSchema" : [ + + ], + "outputSchema" : [ + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 15 × 2048)", + "shortDescription" : "", + "shape" : "[1, 15, 2048]", + "name" : "all_logits", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 1]", + "name" : "key_cache_updates", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 1]", + "name" : "value_cache_updates", + "type" : "MultiArray" + } + ], + "name" : "stepped", + "mlProgramOperationTypeHistogram" : { + "Ios18.expandDims" : 8, + "Ios18.softmax" : 5, + "Ios18.mul" : 123, + "Ios18.matmul" : 10, + "Ios18.rsqrt" : 21, + "Ios16.reduceMean" : 21, + "Split" : 2, + "Ios18.greaterEqual" : 2, + "Select" : 2, + "Ios18.gather" : 2, + "Ios18.add" : 58, + "Ios18.reshape" : 40, + "Ios18.constexprLutToDense" : 51, + "Ios18.conv" : 51, + "Ios18.concat" : 23, + "Ios18.cast" : 5, + "Ios18.sub" : 1, + "Ios18.silu" : 5, + "Ios18.transpose" : 1, + "Ios18.sliceByIndex" : 100, + "Ios18.squeeze" : 15 + } + }, + { + "inputSchema" : [ + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 2048 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 2048, 1, 1]", + "name" : "code0_hidden_states", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 2048 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 2048, 1, 1]", + "name" : "code0_embed", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 15 × 2048)", + "shortDescription" : "", + "shape" : "[15, 2048]", + "name" : "gumbel", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1)", + "shortDescription" : "", + "shape" : "[1]", + "name" : "temperature", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Int32", + "formattedType" : "MultiArray (Int32 1)", + "shortDescription" : "", + "shape" : "[1]", + "name" : "top_k", + "type" : "MultiArray" + } + ], + "computePrecision" : "Mixed (Float16, Float32, Int32, UInt16)", + "storagePrecision" : "Mixed (Float16, Palettized (8 bits), UInt8)", + "stateSchema" : [ + + ], + "outputSchema" : [ + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Int32", + "formattedType" : "MultiArray (Int32 1 × 15)", + "shortDescription" : "", + "shape" : "[1, 15]", + "name" : "codes", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 2048 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 2048, 1, 1]", + "name" : "embed_sum", + "type" : "MultiArray" + } + ], + "name" : "fused", + "mlProgramOperationTypeHistogram" : { + "Ios18.softmax" : 79, + "Ios18.mul" : 1951, + "Ios18.matmul" : 158, + "Ios18.rsqrt" : 333, + "Ios16.reduceMean" : 333, + "Ios18.realDiv" : 15, + "Ios18.greaterEqual" : 15, + "Select" : 15, + "Ios16.reduceMin" : 15, + "Ios18.add" : 932, + "Tile" : 158, + "Ios18.reduceArgmax" : 15, + "Ios16.fillLike" : 15, + "Ios18.reshape" : 996, + "Ios18.gather" : 15, + "Ios18.constexprLutToDense" : 52, + "Ios18.conv" : 586, + "Ios18.concat" : 189, + "Ios18.topk" : 15, + "Ios18.sub" : 1, + "Ios18.transpose" : 474, + "Ios18.silu" : 79, + "Ios18.cast" : 17, + "Ios18.less" : 1, + "Stack" : 1, + "Ios18.sliceByIndex" : 483 + } + } + ], + "mlProgramOperationTypeHistogram" : { + "Ios18.expandDims" : 8, + "Ios18.softmax" : 5, + "Ios18.mul" : 123, + "Ios18.matmul" : 10, + "Ios18.rsqrt" : 21, + "Ios16.reduceMean" : 21, + "Split" : 2, + "Ios18.greaterEqual" : 2, + "Select" : 2, + "Ios18.gather" : 2, + "Ios18.add" : 58, + "Ios18.reshape" : 40, + "Ios18.constexprLutToDense" : 51, + "Ios18.conv" : 51, + "Ios18.concat" : 23, + "Ios18.cast" : 5, + "Ios18.sub" : 1, + "Ios18.silu" : 5, + "Ios18.transpose" : 1, + "Ios18.sliceByIndex" : 100, + "Ios18.squeeze" : 15 + }, + "isUpdatable" : "0", + "stateSchema" : [ + + ], + "availability" : { + "macOS" : "15.0", + "tvOS" : "18.0", + "visionOS" : "2.0", + "watchOS" : "11.0", + "iOS" : "18.0", + "macCatalyst" : "18.0" + }, + "computePrecision" : "Mixed (Float16, Float32, Int16, Int32, UInt16)", + "modelType" : { + "name" : "MLModelType_mlProgram" + }, + "inputSchema" : [ + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 2048 × 1 × 1)", + "shortDescription" : "", + "shape" : "[1, 2048, 1, 1]", + "name" : "input_embeds", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Int32", + "formattedType" : "MultiArray (Int32 1)", + "shortDescription" : "", + "shape" : "[1]", + "name" : "cache_length", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 16]", + "name" : "key_cache", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 5120, 1, 16]", + "name" : "value_cache", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 16]", + "name" : "kv_cache_update_mask", + "type" : "MultiArray" + }, + { + "hasShapeFlexibility" : "0", + "isOptional" : "0", + "dataType" : "Float16", + "formattedType" : "MultiArray (Float16 1 × 16)", + "shortDescription" : "", + "shape" : "[1, 16]", + "name" : "key_padding_mask", + "type" : "MultiArray" + } + ], + "defaultFunctionName" : "stepped", + "generatedClassName" : "MultiCodeDecoder", + "userDefinedMetadata" : { + + }, + "method" : "predict" + } +] \ No newline at end of file diff --git a/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/model.mil b/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/model.mil new file mode 100644 index 0000000000000000000000000000000000000000..4b303f5e14b9819c8d4416f3ff465c9aaff79c8e --- /dev/null +++ b/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/model.mil @@ -0,0 +1,17227 @@ +program(1.3) +[buildInfo = dict({{"coremlc-component-MIL", "3520.4.1"}, {"coremlc-version", "3520.5.1"}})] +{ + func fused(tensor code0_embed, tensor code0_hidden_states, tensor gumbel, tensor temperature, tensor top_k) { + string inputs_1_pad_type_0 = const()[name = string("inputs_1_pad_type_0"), val = string("valid")]; + tensor inputs_1_strides_0 = const()[name = string("inputs_1_strides_0"), val = tensor([1, 1])]; + tensor inputs_1_pad_0 = const()[name = string("inputs_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_1_dilations_0 = const()[name = string("inputs_1_dilations_0"), val = tensor([1, 1])]; + int32 inputs_1_groups_0 = const()[name = string("inputs_1_groups_0"), val = int32(1)]; + tensor input_projection_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2097280))))[name = string("input_projection_weight_to_fp16_palettized")]; + tensor input_projection_bias_to_fp16 = const()[name = string("input_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2097856)))]; + tensor inputs_1_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_1_dilations_0, groups = inputs_1_groups_0, pad = inputs_1_pad_0, pad_type = inputs_1_pad_type_0, strides = inputs_1_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code0_hidden_states)[name = string("inputs_1_cast_fp16")]; + int32 var_175 = const()[name = string("op_175"), val = int32(3)]; + int32 var_185 = const()[name = string("op_185"), val = int32(-2)]; + tensor inputs_sq_1_cast_fp16 = mul(x = inputs_1_cast_fp16, y = inputs_1_cast_fp16)[name = string("inputs_sq_1_cast_fp16")]; + tensor variance_1_axes_0 = const()[name = string("variance_1_axes_0"), val = tensor([1])]; + bool variance_1_keep_dims_0 = const()[name = string("variance_1_keep_dims_0"), val = bool(true)]; + tensor variance_1_cast_fp16 = reduce_mean(axes = variance_1_axes_0, keep_dims = variance_1_keep_dims_0, x = inputs_sq_1_cast_fp16)[name = string("variance_1_cast_fp16")]; + fp16 var_199_to_fp16 = const()[name = string("op_199_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_200_cast_fp16 = add(x = variance_1_cast_fp16, y = var_199_to_fp16)[name = string("op_200_cast_fp16")]; + fp32 var_201_epsilon_0 = const()[name = string("op_201_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_201_cast_fp16 = rsqrt(epsilon = var_201_epsilon_0, x = var_200_cast_fp16)[name = string("op_201_cast_fp16")]; + tensor hidden_states_1_cast_fp16 = mul(x = inputs_1_cast_fp16, y = var_201_cast_fp16)[name = string("hidden_states_1_cast_fp16")]; + tensor w_1_to_fp16 = const()[name = string("w_1_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2108288)))]; + tensor obj_1_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_1_cast_fp16)[name = string("obj_1_cast_fp16")]; + string query_1_pad_type_0 = const()[name = string("query_1_pad_type_0"), val = string("valid")]; + tensor query_1_strides_0 = const()[name = string("query_1_strides_0"), val = tensor([1, 1])]; + tensor query_1_pad_0 = const()[name = string("query_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_1_dilations_0 = const()[name = string("query_1_dilations_0"), val = tensor([1, 1])]; + int32 query_1_groups_0 = const()[name = string("query_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2110400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4207616))))[name = string("layers_0_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor layers_0_self_attn_q_proj_bias_to_fp16 = const()[name = string("layers_0_self_attn_q_proj_bias_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4208192)))]; + tensor query_1_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_1_dilations_0, groups = query_1_groups_0, pad = query_1_pad_0, pad_type = query_1_pad_type_0, strides = query_1_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_1_cast_fp16)[name = string("query_1_cast_fp16")]; + string current_key_1_pad_type_0 = const()[name = string("current_key_1_pad_type_0"), val = string("valid")]; + tensor current_key_1_strides_0 = const()[name = string("current_key_1_strides_0"), val = tensor([1, 1])]; + tensor current_key_1_pad_0 = const()[name = string("current_key_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_1_dilations_0 = const()[name = string("current_key_1_dilations_0"), val = tensor([1, 1])]; + int32 current_key_1_groups_0 = const()[name = string("current_key_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4212352))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5260992))))[name = string("layers_0_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_1_cast_fp16 = conv(dilations = current_key_1_dilations_0, groups = current_key_1_groups_0, pad = current_key_1_pad_0, pad_type = current_key_1_pad_type_0, strides = current_key_1_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_1_cast_fp16)[name = string("current_key_1_cast_fp16")]; + string current_value_1_pad_type_0 = const()[name = string("current_value_1_pad_type_0"), val = string("valid")]; + tensor current_value_1_strides_0 = const()[name = string("current_value_1_strides_0"), val = tensor([1, 1])]; + tensor current_value_1_pad_0 = const()[name = string("current_value_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_1_dilations_0 = const()[name = string("current_value_1_dilations_0"), val = tensor([1, 1])]; + int32 current_value_1_groups_0 = const()[name = string("current_value_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5261568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6310208))))[name = string("layers_0_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor layers_0_self_attn_v_proj_bias_to_fp16 = const()[name = string("layers_0_self_attn_v_proj_bias_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6310784)))]; + tensor current_value_1_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_1_dilations_0, groups = current_value_1_groups_0, pad = current_value_1_pad_0, pad_type = current_value_1_pad_type_0, strides = current_value_1_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_1_cast_fp16)[name = string("current_value_1_cast_fp16")]; + tensor var_238 = const()[name = string("op_238"), val = tensor([16, 128, 1, 1])]; + tensor inputs_3_cast_fp16 = reshape(shape = var_238, x = query_1_cast_fp16)[name = string("inputs_3_cast_fp16")]; + tensor inputs_sq_3_cast_fp16 = mul(x = inputs_3_cast_fp16, y = inputs_3_cast_fp16)[name = string("inputs_sq_3_cast_fp16")]; + tensor variance_3_axes_0 = const()[name = string("variance_3_axes_0"), val = tensor([1])]; + bool variance_3_keep_dims_0 = const()[name = string("variance_3_keep_dims_0"), val = bool(true)]; + tensor variance_3_cast_fp16 = reduce_mean(axes = variance_3_axes_0, keep_dims = variance_3_keep_dims_0, x = inputs_sq_3_cast_fp16)[name = string("variance_3_cast_fp16")]; + fp16 var_244_to_fp16 = const()[name = string("op_244_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_245_cast_fp16 = add(x = variance_3_cast_fp16, y = var_244_to_fp16)[name = string("op_245_cast_fp16")]; + fp32 var_246_epsilon_0 = const()[name = string("op_246_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_246_cast_fp16 = rsqrt(epsilon = var_246_epsilon_0, x = var_245_cast_fp16)[name = string("op_246_cast_fp16")]; + tensor hidden_states_3_cast_fp16 = mul(x = inputs_3_cast_fp16, y = var_246_cast_fp16)[name = string("hidden_states_3_cast_fp16")]; + tensor w_3_to_fp16 = const()[name = string("w_3_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6312896)))]; + tensor query_normed_1_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_3_cast_fp16)[name = string("query_normed_1_cast_fp16")]; + tensor var_254 = const()[name = string("op_254"), val = tensor([8, 128, 1, 1])]; + tensor inputs_5_cast_fp16 = reshape(shape = var_254, x = current_key_1_cast_fp16)[name = string("inputs_5_cast_fp16")]; + tensor inputs_sq_5_cast_fp16 = mul(x = inputs_5_cast_fp16, y = inputs_5_cast_fp16)[name = string("inputs_sq_5_cast_fp16")]; + tensor variance_5_axes_0 = const()[name = string("variance_5_axes_0"), val = tensor([1])]; + bool variance_5_keep_dims_0 = const()[name = string("variance_5_keep_dims_0"), val = bool(true)]; + tensor variance_5_cast_fp16 = reduce_mean(axes = variance_5_axes_0, keep_dims = variance_5_keep_dims_0, x = inputs_sq_5_cast_fp16)[name = string("variance_5_cast_fp16")]; + fp16 var_260_to_fp16 = const()[name = string("op_260_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_261_cast_fp16 = add(x = variance_5_cast_fp16, y = var_260_to_fp16)[name = string("op_261_cast_fp16")]; + fp32 var_262_epsilon_0 = const()[name = string("op_262_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_262_cast_fp16 = rsqrt(epsilon = var_262_epsilon_0, x = var_261_cast_fp16)[name = string("op_262_cast_fp16")]; + tensor hidden_states_5_cast_fp16 = mul(x = inputs_5_cast_fp16, y = var_262_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; + tensor w_5_to_fp16 = const()[name = string("w_5_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6313216)))]; + tensor current_key_normed_1_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_5_cast_fp16)[name = string("current_key_normed_1_cast_fp16")]; + tensor var_280 = const()[name = string("op_280"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_1_cast_fp16 = reshape(shape = var_280, x = query_normed_1_cast_fp16)[name = string("mh_q_1_cast_fp16")]; + tensor var_282 = const()[name = string("op_282"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_1_cast_fp16 = reshape(shape = var_282, x = current_key_normed_1_cast_fp16)[name = string("mh_k_1_cast_fp16")]; + tensor var_291_begin_0 = const()[name = string("op_291_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_291_end_0 = const()[name = string("op_291_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_291_end_mask_0 = const()[name = string("op_291_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_291_cast_fp16 = slice_by_index(begin = var_291_begin_0, end = var_291_end_0, end_mask = var_291_end_mask_0, x = mh_q_1_cast_fp16)[name = string("op_291_cast_fp16")]; + tensor var_297_begin_0 = const()[name = string("op_297_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_297_end_0 = const()[name = string("op_297_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_297_end_mask_0 = const()[name = string("op_297_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_297_cast_fp16 = slice_by_index(begin = var_297_begin_0, end = var_297_end_0, end_mask = var_297_end_mask_0, x = mh_q_1_cast_fp16)[name = string("op_297_cast_fp16")]; + fp16 const_14_promoted_to_fp16 = const()[name = string("const_14_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_299_cast_fp16 = mul(x = var_297_cast_fp16, y = const_14_promoted_to_fp16)[name = string("op_299_cast_fp16")]; + bool var_301_interleave_0 = const()[name = string("op_301_interleave_0"), val = bool(false)]; + tensor var_301_cast_fp16 = concat(axis = var_185, interleave = var_301_interleave_0, values = (var_299_cast_fp16, var_291_cast_fp16))[name = string("op_301_cast_fp16")]; + tensor sin_1_to_fp16 = const()[name = string("sin_1_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112273472)))]; + tensor var_302_cast_fp16 = mul(x = var_301_cast_fp16, y = sin_1_to_fp16)[name = string("op_302_cast_fp16")]; + tensor mh_q_3_cast_fp16 = add(x = mh_q_1_cast_fp16, y = var_302_cast_fp16)[name = string("mh_q_3_cast_fp16")]; + tensor var_309_begin_0 = const()[name = string("op_309_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_309_end_0 = const()[name = string("op_309_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_309_end_mask_0 = const()[name = string("op_309_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_309_cast_fp16 = slice_by_index(begin = var_309_begin_0, end = var_309_end_0, end_mask = var_309_end_mask_0, x = mh_k_1_cast_fp16)[name = string("op_309_cast_fp16")]; + tensor var_315_begin_0 = const()[name = string("op_315_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_315_end_0 = const()[name = string("op_315_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_315_end_mask_0 = const()[name = string("op_315_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_315_cast_fp16 = slice_by_index(begin = var_315_begin_0, end = var_315_end_0, end_mask = var_315_end_mask_0, x = mh_k_1_cast_fp16)[name = string("op_315_cast_fp16")]; + fp16 const_17_promoted_to_fp16 = const()[name = string("const_17_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_317_cast_fp16 = mul(x = var_315_cast_fp16, y = const_17_promoted_to_fp16)[name = string("op_317_cast_fp16")]; + bool var_319_interleave_0 = const()[name = string("op_319_interleave_0"), val = bool(false)]; + tensor var_319_cast_fp16 = concat(axis = var_185, interleave = var_319_interleave_0, values = (var_317_cast_fp16, var_309_cast_fp16))[name = string("op_319_cast_fp16")]; + tensor var_320_cast_fp16 = mul(x = var_319_cast_fp16, y = sin_1_to_fp16)[name = string("op_320_cast_fp16")]; + tensor mh_k_3_cast_fp16 = add(x = mh_k_1_cast_fp16, y = var_320_cast_fp16)[name = string("mh_k_3_cast_fp16")]; + tensor var_324 = const()[name = string("op_324"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_3_cast_fp16 = reshape(shape = var_324, x = mh_k_3_cast_fp16)[name = string("current_key_3_cast_fp16")]; + tensor var_328_to_fp16 = const()[name = string("op_328_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112273792)))]; + tensor var_332_cast_fp16 = mul(x = current_key_3_cast_fp16, y = var_328_to_fp16)[name = string("op_332_cast_fp16")]; + tensor var_336_cast_fp16 = mul(x = current_value_1_cast_fp16, y = var_328_to_fp16)[name = string("op_336_cast_fp16")]; + fp16 var_343_to_fp16 = const()[name = string("op_343_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_7_cast_fp16 = mul(x = mh_q_3_cast_fp16, y = var_343_to_fp16)[name = string("mh_q_7_cast_fp16")]; + tensor var_345 = const()[name = string("op_345"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_5_cast_fp16 = reshape(shape = var_345, x = var_332_cast_fp16)[name = string("mh_k_5_cast_fp16")]; + tensor var_347 = const()[name = string("op_347"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_1_cast_fp16 = reshape(shape = var_347, x = var_336_cast_fp16)[name = string("mh_v_1_cast_fp16")]; + tensor transpose_0_perm_0 = const()[name = string("transpose_0_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_0_reps_0 = const()[name = string("tile_0_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_0_cast_fp16 = transpose(perm = transpose_0_perm_0, x = mh_k_5_cast_fp16)[name = string("transpose_473")]; + tensor tile_0_cast_fp16 = tile(reps = tile_0_reps_0, x = transpose_0_cast_fp16)[name = string("tile_0_cast_fp16")]; + tensor concat_4 = const()[name = string("concat_4"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_0_cast_fp16 = reshape(shape = concat_4, x = tile_0_cast_fp16)[name = string("reshape_0_cast_fp16")]; + tensor transpose_1_perm_0 = const()[name = string("transpose_1_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_5 = const()[name = string("concat_5"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_1_cast_fp16 = transpose(perm = transpose_1_perm_0, x = reshape_0_cast_fp16)[name = string("transpose_472")]; + tensor reshape_1_cast_fp16 = reshape(shape = concat_5, x = transpose_1_cast_fp16)[name = string("reshape_1_cast_fp16")]; + tensor transpose_2_perm_0 = const()[name = string("transpose_2_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_1_reps_0 = const()[name = string("tile_1_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_2_cast_fp16 = transpose(perm = transpose_2_perm_0, x = mh_v_1_cast_fp16)[name = string("transpose_471")]; + tensor tile_1_cast_fp16 = tile(reps = tile_1_reps_0, x = transpose_2_cast_fp16)[name = string("tile_1_cast_fp16")]; + tensor concat_6 = const()[name = string("concat_6"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_2_cast_fp16 = reshape(shape = concat_6, x = tile_1_cast_fp16)[name = string("reshape_2_cast_fp16")]; + tensor transpose_3_perm_0 = const()[name = string("transpose_3_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_7 = const()[name = string("concat_7"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_3_cast_fp16 = transpose(perm = transpose_3_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_470")]; + tensor reshape_3_cast_fp16 = reshape(shape = concat_7, x = transpose_3_cast_fp16)[name = string("reshape_3_cast_fp16")]; + tensor transpose_321_perm_0 = const()[name = string("transpose_321_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_1_transpose_x_1 = const()[name = string("mh_w_1_transpose_x_1"), val = bool(true)]; + bool mh_w_1_transpose_y_1 = const()[name = string("mh_w_1_transpose_y_1"), val = bool(false)]; + tensor transpose_321_cast_fp16 = transpose(perm = transpose_321_perm_0, x = reshape_1_cast_fp16)[name = string("transpose_469")]; + tensor mh_w_1_cast_fp16 = matmul(transpose_x = mh_w_1_transpose_x_1, transpose_y = mh_w_1_transpose_y_1, x = mh_q_7_cast_fp16, y = transpose_321_cast_fp16)[name = string("mh_w_1_cast_fp16")]; + tensor var_355_to_fp16 = const()[name = string("op_355_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112273920)))]; + tensor mh_w_3_cast_fp16 = add(x = mh_w_1_cast_fp16, y = var_355_to_fp16)[name = string("mh_w_3_cast_fp16")]; + tensor mh_w_5_cast_fp16 = softmax(axis = var_175, x = mh_w_3_cast_fp16)[name = string("mh_w_5_cast_fp16")]; + tensor transpose_322_perm_0 = const()[name = string("transpose_322_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_1_transpose_x_1 = const()[name = string("attn_1_transpose_x_1"), val = bool(false)]; + bool attn_1_transpose_y_1 = const()[name = string("attn_1_transpose_y_1"), val = bool(true)]; + tensor transpose_322_cast_fp16 = transpose(perm = transpose_322_perm_0, x = reshape_3_cast_fp16)[name = string("transpose_468")]; + tensor attn_1_cast_fp16 = matmul(transpose_x = attn_1_transpose_x_1, transpose_y = attn_1_transpose_y_1, x = transpose_322_cast_fp16, y = mh_w_5_cast_fp16)[name = string("attn_1_cast_fp16")]; + tensor var_361 = const()[name = string("op_361"), val = tensor([1, 2048, 1, 1])]; + tensor input_1_cast_fp16 = reshape(shape = var_361, x = attn_1_cast_fp16)[name = string("input_1_cast_fp16")]; + string obj_11_pad_type_0 = const()[name = string("obj_11_pad_type_0"), val = string("valid")]; + tensor obj_11_strides_0 = const()[name = string("obj_11_strides_0"), val = tensor([1, 1])]; + tensor obj_11_pad_0 = const()[name = string("obj_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_11_dilations_0 = const()[name = string("obj_11_dilations_0"), val = tensor([1, 1])]; + int32 obj_11_groups_0 = const()[name = string("obj_11_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6313536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8410752))))[name = string("layers_0_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_11_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_11_dilations_0, groups = obj_11_groups_0, pad = obj_11_pad_0, pad_type = obj_11_pad_type_0, strides = obj_11_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_1_cast_fp16)[name = string("obj_11_cast_fp16")]; + tensor inputs_7_cast_fp16 = add(x = inputs_1_cast_fp16, y = obj_11_cast_fp16)[name = string("inputs_7_cast_fp16")]; + tensor inputs_sq_7_cast_fp16 = mul(x = inputs_7_cast_fp16, y = inputs_7_cast_fp16)[name = string("inputs_sq_7_cast_fp16")]; + tensor variance_7_axes_0 = const()[name = string("variance_7_axes_0"), val = tensor([1])]; + bool variance_7_keep_dims_0 = const()[name = string("variance_7_keep_dims_0"), val = bool(true)]; + tensor variance_7_cast_fp16 = reduce_mean(axes = variance_7_axes_0, keep_dims = variance_7_keep_dims_0, x = inputs_sq_7_cast_fp16)[name = string("variance_7_cast_fp16")]; + fp16 var_379_to_fp16 = const()[name = string("op_379_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_380_cast_fp16 = add(x = variance_7_cast_fp16, y = var_379_to_fp16)[name = string("op_380_cast_fp16")]; + fp32 var_381_epsilon_0 = const()[name = string("op_381_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_381_cast_fp16 = rsqrt(epsilon = var_381_epsilon_0, x = var_380_cast_fp16)[name = string("op_381_cast_fp16")]; + tensor hidden_states_7_cast_fp16 = mul(x = inputs_7_cast_fp16, y = var_381_cast_fp16)[name = string("hidden_states_7_cast_fp16")]; + tensor w_7_to_fp16 = const()[name = string("w_7_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8411328)))]; + tensor input_3_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_7_cast_fp16)[name = string("input_3_cast_fp16")]; + string input_5_pad_type_0 = const()[name = string("input_5_pad_type_0"), val = string("valid")]; + tensor input_5_strides_0 = const()[name = string("input_5_strides_0"), val = tensor([1, 1])]; + tensor input_5_pad_0 = const()[name = string("input_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_5_dilations_0 = const()[name = string("input_5_dilations_0"), val = tensor([1, 1])]; + int32 input_5_groups_0 = const()[name = string("input_5_groups_0"), val = int32(1)]; + tensor layers_0_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8413440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11559232))))[name = string("layers_0_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_5_cast_fp16 = conv(dilations = input_5_dilations_0, groups = input_5_groups_0, pad = input_5_pad_0, pad_type = input_5_pad_type_0, strides = input_5_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_3_cast_fp16)[name = string("input_5_cast_fp16")]; + tensor var_395_cast_fp16 = silu(x = input_5_cast_fp16)[name = string("op_395_cast_fp16")]; + string var_401_pad_type_0 = const()[name = string("op_401_pad_type_0"), val = string("valid")]; + tensor var_401_strides_0 = const()[name = string("op_401_strides_0"), val = tensor([1, 1])]; + tensor var_401_pad_0 = const()[name = string("op_401_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_401_dilations_0 = const()[name = string("op_401_dilations_0"), val = tensor([1, 1])]; + int32 var_401_groups_0 = const()[name = string("op_401_groups_0"), val = int32(1)]; + tensor layers_0_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11559808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14705600))))[name = string("layers_0_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_401_cast_fp16 = conv(dilations = var_401_dilations_0, groups = var_401_groups_0, pad = var_401_pad_0, pad_type = var_401_pad_type_0, strides = var_401_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_3_cast_fp16)[name = string("op_401_cast_fp16")]; + tensor input_7_cast_fp16 = mul(x = var_395_cast_fp16, y = var_401_cast_fp16)[name = string("input_7_cast_fp16")]; + string hidden_states_9_pad_type_0 = const()[name = string("hidden_states_9_pad_type_0"), val = string("valid")]; + tensor hidden_states_9_strides_0 = const()[name = string("hidden_states_9_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_9_pad_0 = const()[name = string("hidden_states_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_9_dilations_0 = const()[name = string("hidden_states_9_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_9_groups_0 = const()[name = string("hidden_states_9_groups_0"), val = int32(1)]; + tensor layers_0_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14706176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17851968))))[name = string("layers_0_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_9_cast_fp16 = conv(dilations = hidden_states_9_dilations_0, groups = hidden_states_9_groups_0, pad = hidden_states_9_pad_0, pad_type = hidden_states_9_pad_type_0, strides = hidden_states_9_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; + tensor inputs_9_cast_fp16 = add(x = inputs_7_cast_fp16, y = hidden_states_9_cast_fp16)[name = string("inputs_9_cast_fp16")]; + int32 var_449 = const()[name = string("op_449"), val = int32(3)]; + int32 var_459 = const()[name = string("op_459"), val = int32(-2)]; + tensor inputs_sq_9_cast_fp16 = mul(x = inputs_9_cast_fp16, y = inputs_9_cast_fp16)[name = string("inputs_sq_9_cast_fp16")]; + tensor variance_9_axes_0 = const()[name = string("variance_9_axes_0"), val = tensor([1])]; + bool variance_9_keep_dims_0 = const()[name = string("variance_9_keep_dims_0"), val = bool(true)]; + tensor variance_9_cast_fp16 = reduce_mean(axes = variance_9_axes_0, keep_dims = variance_9_keep_dims_0, x = inputs_sq_9_cast_fp16)[name = string("variance_9_cast_fp16")]; + fp16 var_473_to_fp16 = const()[name = string("op_473_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_474_cast_fp16 = add(x = variance_9_cast_fp16, y = var_473_to_fp16)[name = string("op_474_cast_fp16")]; + fp32 var_475_epsilon_0 = const()[name = string("op_475_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_475_cast_fp16 = rsqrt(epsilon = var_475_epsilon_0, x = var_474_cast_fp16)[name = string("op_475_cast_fp16")]; + tensor hidden_states_11_cast_fp16 = mul(x = inputs_9_cast_fp16, y = var_475_cast_fp16)[name = string("hidden_states_11_cast_fp16")]; + tensor w_9_to_fp16 = const()[name = string("w_9_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17852544)))]; + tensor obj_13_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_11_cast_fp16)[name = string("obj_13_cast_fp16")]; + string query_7_pad_type_0 = const()[name = string("query_7_pad_type_0"), val = string("valid")]; + tensor query_7_strides_0 = const()[name = string("query_7_strides_0"), val = tensor([1, 1])]; + tensor query_7_pad_0 = const()[name = string("query_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_7_dilations_0 = const()[name = string("query_7_dilations_0"), val = tensor([1, 1])]; + int32 query_7_groups_0 = const()[name = string("query_7_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17854656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19951872))))[name = string("layers_1_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor query_7_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_7_dilations_0, groups = query_7_groups_0, pad = query_7_pad_0, pad_type = query_7_pad_type_0, strides = query_7_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_13_cast_fp16)[name = string("query_7_cast_fp16")]; + string current_key_5_pad_type_0 = const()[name = string("current_key_5_pad_type_0"), val = string("valid")]; + tensor current_key_5_strides_0 = const()[name = string("current_key_5_strides_0"), val = tensor([1, 1])]; + tensor current_key_5_pad_0 = const()[name = string("current_key_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_5_dilations_0 = const()[name = string("current_key_5_dilations_0"), val = tensor([1, 1])]; + int32 current_key_5_groups_0 = const()[name = string("current_key_5_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19952448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21001088))))[name = string("layers_1_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_5_cast_fp16 = conv(dilations = current_key_5_dilations_0, groups = current_key_5_groups_0, pad = current_key_5_pad_0, pad_type = current_key_5_pad_type_0, strides = current_key_5_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_13_cast_fp16)[name = string("current_key_5_cast_fp16")]; + string current_value_3_pad_type_0 = const()[name = string("current_value_3_pad_type_0"), val = string("valid")]; + tensor current_value_3_strides_0 = const()[name = string("current_value_3_strides_0"), val = tensor([1, 1])]; + tensor current_value_3_pad_0 = const()[name = string("current_value_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_3_dilations_0 = const()[name = string("current_value_3_dilations_0"), val = tensor([1, 1])]; + int32 current_value_3_groups_0 = const()[name = string("current_value_3_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21001664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22050304))))[name = string("layers_1_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor current_value_3_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_3_dilations_0, groups = current_value_3_groups_0, pad = current_value_3_pad_0, pad_type = current_value_3_pad_type_0, strides = current_value_3_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_13_cast_fp16)[name = string("current_value_3_cast_fp16")]; + tensor var_512 = const()[name = string("op_512"), val = tensor([16, 128, 1, 1])]; + tensor inputs_11_cast_fp16 = reshape(shape = var_512, x = query_7_cast_fp16)[name = string("inputs_11_cast_fp16")]; + tensor inputs_sq_11_cast_fp16 = mul(x = inputs_11_cast_fp16, y = inputs_11_cast_fp16)[name = string("inputs_sq_11_cast_fp16")]; + tensor variance_11_axes_0 = const()[name = string("variance_11_axes_0"), val = tensor([1])]; + bool variance_11_keep_dims_0 = const()[name = string("variance_11_keep_dims_0"), val = bool(true)]; + tensor variance_11_cast_fp16 = reduce_mean(axes = variance_11_axes_0, keep_dims = variance_11_keep_dims_0, x = inputs_sq_11_cast_fp16)[name = string("variance_11_cast_fp16")]; + fp16 var_518_to_fp16 = const()[name = string("op_518_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_519_cast_fp16 = add(x = variance_11_cast_fp16, y = var_518_to_fp16)[name = string("op_519_cast_fp16")]; + fp32 var_520_epsilon_0 = const()[name = string("op_520_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_520_cast_fp16 = rsqrt(epsilon = var_520_epsilon_0, x = var_519_cast_fp16)[name = string("op_520_cast_fp16")]; + tensor hidden_states_13_cast_fp16 = mul(x = inputs_11_cast_fp16, y = var_520_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; + tensor w_11_to_fp16 = const()[name = string("w_11_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22050880)))]; + tensor query_normed_3_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_13_cast_fp16)[name = string("query_normed_3_cast_fp16")]; + tensor var_528 = const()[name = string("op_528"), val = tensor([8, 128, 1, 1])]; + tensor inputs_13_cast_fp16 = reshape(shape = var_528, x = current_key_5_cast_fp16)[name = string("inputs_13_cast_fp16")]; + tensor inputs_sq_13_cast_fp16 = mul(x = inputs_13_cast_fp16, y = inputs_13_cast_fp16)[name = string("inputs_sq_13_cast_fp16")]; + tensor variance_13_axes_0 = const()[name = string("variance_13_axes_0"), val = tensor([1])]; + bool variance_13_keep_dims_0 = const()[name = string("variance_13_keep_dims_0"), val = bool(true)]; + tensor variance_13_cast_fp16 = reduce_mean(axes = variance_13_axes_0, keep_dims = variance_13_keep_dims_0, x = inputs_sq_13_cast_fp16)[name = string("variance_13_cast_fp16")]; + fp16 var_534_to_fp16 = const()[name = string("op_534_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_535_cast_fp16 = add(x = variance_13_cast_fp16, y = var_534_to_fp16)[name = string("op_535_cast_fp16")]; + fp32 var_536_epsilon_0 = const()[name = string("op_536_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_536_cast_fp16 = rsqrt(epsilon = var_536_epsilon_0, x = var_535_cast_fp16)[name = string("op_536_cast_fp16")]; + tensor hidden_states_15_cast_fp16 = mul(x = inputs_13_cast_fp16, y = var_536_cast_fp16)[name = string("hidden_states_15_cast_fp16")]; + tensor w_13_to_fp16 = const()[name = string("w_13_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22051200)))]; + tensor current_key_normed_3_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_15_cast_fp16)[name = string("current_key_normed_3_cast_fp16")]; + tensor var_554 = const()[name = string("op_554"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_9_cast_fp16 = reshape(shape = var_554, x = query_normed_3_cast_fp16)[name = string("mh_q_9_cast_fp16")]; + tensor var_556 = const()[name = string("op_556"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_9_cast_fp16 = reshape(shape = var_556, x = current_key_normed_3_cast_fp16)[name = string("mh_k_9_cast_fp16")]; + tensor var_565_begin_0 = const()[name = string("op_565_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_565_end_0 = const()[name = string("op_565_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_565_end_mask_0 = const()[name = string("op_565_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_565_cast_fp16 = slice_by_index(begin = var_565_begin_0, end = var_565_end_0, end_mask = var_565_end_mask_0, x = mh_q_9_cast_fp16)[name = string("op_565_cast_fp16")]; + tensor var_571_begin_0 = const()[name = string("op_571_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_571_end_0 = const()[name = string("op_571_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_571_end_mask_0 = const()[name = string("op_571_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_571_cast_fp16 = slice_by_index(begin = var_571_begin_0, end = var_571_end_0, end_mask = var_571_end_mask_0, x = mh_q_9_cast_fp16)[name = string("op_571_cast_fp16")]; + fp16 const_34_promoted_to_fp16 = const()[name = string("const_34_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_573_cast_fp16 = mul(x = var_571_cast_fp16, y = const_34_promoted_to_fp16)[name = string("op_573_cast_fp16")]; + bool var_575_interleave_0 = const()[name = string("op_575_interleave_0"), val = bool(false)]; + tensor var_575_cast_fp16 = concat(axis = var_459, interleave = var_575_interleave_0, values = (var_573_cast_fp16, var_565_cast_fp16))[name = string("op_575_cast_fp16")]; + tensor var_576_cast_fp16 = mul(x = var_575_cast_fp16, y = sin_1_to_fp16)[name = string("op_576_cast_fp16")]; + tensor mh_q_11_cast_fp16 = add(x = mh_q_9_cast_fp16, y = var_576_cast_fp16)[name = string("mh_q_11_cast_fp16")]; + tensor var_583_begin_0 = const()[name = string("op_583_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_583_end_0 = const()[name = string("op_583_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_583_end_mask_0 = const()[name = string("op_583_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_583_cast_fp16 = slice_by_index(begin = var_583_begin_0, end = var_583_end_0, end_mask = var_583_end_mask_0, x = mh_k_9_cast_fp16)[name = string("op_583_cast_fp16")]; + tensor var_589_begin_0 = const()[name = string("op_589_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_589_end_0 = const()[name = string("op_589_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_589_end_mask_0 = const()[name = string("op_589_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_589_cast_fp16 = slice_by_index(begin = var_589_begin_0, end = var_589_end_0, end_mask = var_589_end_mask_0, x = mh_k_9_cast_fp16)[name = string("op_589_cast_fp16")]; + fp16 const_37_promoted_to_fp16 = const()[name = string("const_37_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_591_cast_fp16 = mul(x = var_589_cast_fp16, y = const_37_promoted_to_fp16)[name = string("op_591_cast_fp16")]; + bool var_593_interleave_0 = const()[name = string("op_593_interleave_0"), val = bool(false)]; + tensor var_593_cast_fp16 = concat(axis = var_459, interleave = var_593_interleave_0, values = (var_591_cast_fp16, var_583_cast_fp16))[name = string("op_593_cast_fp16")]; + tensor var_594_cast_fp16 = mul(x = var_593_cast_fp16, y = sin_1_to_fp16)[name = string("op_594_cast_fp16")]; + tensor mh_k_11_cast_fp16 = add(x = mh_k_9_cast_fp16, y = var_594_cast_fp16)[name = string("mh_k_11_cast_fp16")]; + tensor var_598 = const()[name = string("op_598"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_7_cast_fp16 = reshape(shape = var_598, x = mh_k_11_cast_fp16)[name = string("current_key_7_cast_fp16")]; + tensor var_602_to_fp16 = const()[name = string("op_602_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112273792)))]; + tensor var_606_cast_fp16 = mul(x = current_key_7_cast_fp16, y = var_602_to_fp16)[name = string("op_606_cast_fp16")]; + tensor var_610_cast_fp16 = mul(x = current_value_3_cast_fp16, y = var_602_to_fp16)[name = string("op_610_cast_fp16")]; + fp16 var_617_to_fp16 = const()[name = string("op_617_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_15_cast_fp16 = mul(x = mh_q_11_cast_fp16, y = var_617_to_fp16)[name = string("mh_q_15_cast_fp16")]; + tensor var_619 = const()[name = string("op_619"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_13_cast_fp16 = reshape(shape = var_619, x = var_606_cast_fp16)[name = string("mh_k_13_cast_fp16")]; + tensor var_621 = const()[name = string("op_621"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_5_cast_fp16 = reshape(shape = var_621, x = var_610_cast_fp16)[name = string("mh_v_5_cast_fp16")]; + tensor transpose_4_perm_0 = const()[name = string("transpose_4_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_2_reps_0 = const()[name = string("tile_2_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_4_cast_fp16 = transpose(perm = transpose_4_perm_0, x = mh_k_13_cast_fp16)[name = string("transpose_467")]; + tensor tile_2_cast_fp16 = tile(reps = tile_2_reps_0, x = transpose_4_cast_fp16)[name = string("tile_2_cast_fp16")]; + tensor concat_8 = const()[name = string("concat_8"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_4_cast_fp16 = reshape(shape = concat_8, x = tile_2_cast_fp16)[name = string("reshape_4_cast_fp16")]; + tensor transpose_5_perm_0 = const()[name = string("transpose_5_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_9 = const()[name = string("concat_9"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_5_cast_fp16 = transpose(perm = transpose_5_perm_0, x = reshape_4_cast_fp16)[name = string("transpose_466")]; + tensor reshape_5_cast_fp16 = reshape(shape = concat_9, x = transpose_5_cast_fp16)[name = string("reshape_5_cast_fp16")]; + tensor transpose_6_perm_0 = const()[name = string("transpose_6_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_3_reps_0 = const()[name = string("tile_3_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_6_cast_fp16 = transpose(perm = transpose_6_perm_0, x = mh_v_5_cast_fp16)[name = string("transpose_465")]; + tensor tile_3_cast_fp16 = tile(reps = tile_3_reps_0, x = transpose_6_cast_fp16)[name = string("tile_3_cast_fp16")]; + tensor concat_10 = const()[name = string("concat_10"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_6_cast_fp16 = reshape(shape = concat_10, x = tile_3_cast_fp16)[name = string("reshape_6_cast_fp16")]; + tensor transpose_7_perm_0 = const()[name = string("transpose_7_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_11 = const()[name = string("concat_11"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_7_cast_fp16 = transpose(perm = transpose_7_perm_0, x = reshape_6_cast_fp16)[name = string("transpose_464")]; + tensor reshape_7_cast_fp16 = reshape(shape = concat_11, x = transpose_7_cast_fp16)[name = string("reshape_7_cast_fp16")]; + tensor transpose_325_perm_0 = const()[name = string("transpose_325_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_7_transpose_x_1 = const()[name = string("mh_w_7_transpose_x_1"), val = bool(true)]; + bool mh_w_7_transpose_y_1 = const()[name = string("mh_w_7_transpose_y_1"), val = bool(false)]; + tensor transpose_325_cast_fp16 = transpose(perm = transpose_325_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_463")]; + tensor mh_w_7_cast_fp16 = matmul(transpose_x = mh_w_7_transpose_x_1, transpose_y = mh_w_7_transpose_y_1, x = mh_q_15_cast_fp16, y = transpose_325_cast_fp16)[name = string("mh_w_7_cast_fp16")]; + tensor var_629_to_fp16 = const()[name = string("op_629_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112273920)))]; + tensor mh_w_9_cast_fp16 = add(x = mh_w_7_cast_fp16, y = var_629_to_fp16)[name = string("mh_w_9_cast_fp16")]; + tensor mh_w_11_cast_fp16 = softmax(axis = var_449, x = mh_w_9_cast_fp16)[name = string("mh_w_11_cast_fp16")]; + tensor transpose_326_perm_0 = const()[name = string("transpose_326_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_3_transpose_x_1 = const()[name = string("attn_3_transpose_x_1"), val = bool(false)]; + bool attn_3_transpose_y_1 = const()[name = string("attn_3_transpose_y_1"), val = bool(true)]; + tensor transpose_326_cast_fp16 = transpose(perm = transpose_326_perm_0, x = reshape_7_cast_fp16)[name = string("transpose_462")]; + tensor attn_3_cast_fp16 = matmul(transpose_x = attn_3_transpose_x_1, transpose_y = attn_3_transpose_y_1, x = transpose_326_cast_fp16, y = mh_w_11_cast_fp16)[name = string("attn_3_cast_fp16")]; + tensor var_635 = const()[name = string("op_635"), val = tensor([1, 2048, 1, 1])]; + tensor input_9_cast_fp16 = reshape(shape = var_635, x = attn_3_cast_fp16)[name = string("input_9_cast_fp16")]; + string obj_19_pad_type_0 = const()[name = string("obj_19_pad_type_0"), val = string("valid")]; + tensor obj_19_strides_0 = const()[name = string("obj_19_strides_0"), val = tensor([1, 1])]; + tensor obj_19_pad_0 = const()[name = string("obj_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_19_dilations_0 = const()[name = string("obj_19_dilations_0"), val = tensor([1, 1])]; + int32 obj_19_groups_0 = const()[name = string("obj_19_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22051520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24148736))))[name = string("layers_1_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_19_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_19_dilations_0, groups = obj_19_groups_0, pad = obj_19_pad_0, pad_type = obj_19_pad_type_0, strides = obj_19_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_9_cast_fp16)[name = string("obj_19_cast_fp16")]; + tensor inputs_15_cast_fp16 = add(x = inputs_9_cast_fp16, y = obj_19_cast_fp16)[name = string("inputs_15_cast_fp16")]; + tensor inputs_sq_15_cast_fp16 = mul(x = inputs_15_cast_fp16, y = inputs_15_cast_fp16)[name = string("inputs_sq_15_cast_fp16")]; + tensor variance_15_axes_0 = const()[name = string("variance_15_axes_0"), val = tensor([1])]; + bool variance_15_keep_dims_0 = const()[name = string("variance_15_keep_dims_0"), val = bool(true)]; + tensor variance_15_cast_fp16 = reduce_mean(axes = variance_15_axes_0, keep_dims = variance_15_keep_dims_0, x = inputs_sq_15_cast_fp16)[name = string("variance_15_cast_fp16")]; + fp16 var_653_to_fp16 = const()[name = string("op_653_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_654_cast_fp16 = add(x = variance_15_cast_fp16, y = var_653_to_fp16)[name = string("op_654_cast_fp16")]; + fp32 var_655_epsilon_0 = const()[name = string("op_655_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_655_cast_fp16 = rsqrt(epsilon = var_655_epsilon_0, x = var_654_cast_fp16)[name = string("op_655_cast_fp16")]; + tensor hidden_states_17_cast_fp16 = mul(x = inputs_15_cast_fp16, y = var_655_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; + tensor w_15_to_fp16 = const()[name = string("w_15_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24149312)))]; + tensor input_11_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_17_cast_fp16)[name = string("input_11_cast_fp16")]; + string input_13_pad_type_0 = const()[name = string("input_13_pad_type_0"), val = string("valid")]; + tensor input_13_strides_0 = const()[name = string("input_13_strides_0"), val = tensor([1, 1])]; + tensor input_13_pad_0 = const()[name = string("input_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_13_dilations_0 = const()[name = string("input_13_dilations_0"), val = tensor([1, 1])]; + int32 input_13_groups_0 = const()[name = string("input_13_groups_0"), val = int32(1)]; + tensor layers_1_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24151424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(27297216))))[name = string("layers_1_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_13_cast_fp16 = conv(dilations = input_13_dilations_0, groups = input_13_groups_0, pad = input_13_pad_0, pad_type = input_13_pad_type_0, strides = input_13_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_11_cast_fp16)[name = string("input_13_cast_fp16")]; + tensor var_669_cast_fp16 = silu(x = input_13_cast_fp16)[name = string("op_669_cast_fp16")]; + string var_675_pad_type_0 = const()[name = string("op_675_pad_type_0"), val = string("valid")]; + tensor var_675_strides_0 = const()[name = string("op_675_strides_0"), val = tensor([1, 1])]; + tensor var_675_pad_0 = const()[name = string("op_675_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_675_dilations_0 = const()[name = string("op_675_dilations_0"), val = tensor([1, 1])]; + int32 var_675_groups_0 = const()[name = string("op_675_groups_0"), val = int32(1)]; + tensor layers_1_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(27297792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30443584))))[name = string("layers_1_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_675_cast_fp16 = conv(dilations = var_675_dilations_0, groups = var_675_groups_0, pad = var_675_pad_0, pad_type = var_675_pad_type_0, strides = var_675_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_11_cast_fp16)[name = string("op_675_cast_fp16")]; + tensor input_15_cast_fp16 = mul(x = var_669_cast_fp16, y = var_675_cast_fp16)[name = string("input_15_cast_fp16")]; + string hidden_states_19_pad_type_0 = const()[name = string("hidden_states_19_pad_type_0"), val = string("valid")]; + tensor hidden_states_19_strides_0 = const()[name = string("hidden_states_19_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_19_pad_0 = const()[name = string("hidden_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_19_dilations_0 = const()[name = string("hidden_states_19_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_19_groups_0 = const()[name = string("hidden_states_19_groups_0"), val = int32(1)]; + tensor layers_1_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30444160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33589952))))[name = string("layers_1_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_19_cast_fp16 = conv(dilations = hidden_states_19_dilations_0, groups = hidden_states_19_groups_0, pad = hidden_states_19_pad_0, pad_type = hidden_states_19_pad_type_0, strides = hidden_states_19_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_15_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; + tensor inputs_17_cast_fp16 = add(x = inputs_15_cast_fp16, y = hidden_states_19_cast_fp16)[name = string("inputs_17_cast_fp16")]; + int32 var_723 = const()[name = string("op_723"), val = int32(3)]; + int32 var_733 = const()[name = string("op_733"), val = int32(-2)]; + tensor inputs_sq_17_cast_fp16 = mul(x = inputs_17_cast_fp16, y = inputs_17_cast_fp16)[name = string("inputs_sq_17_cast_fp16")]; + tensor variance_17_axes_0 = const()[name = string("variance_17_axes_0"), val = tensor([1])]; + bool variance_17_keep_dims_0 = const()[name = string("variance_17_keep_dims_0"), val = bool(true)]; + tensor variance_17_cast_fp16 = reduce_mean(axes = variance_17_axes_0, keep_dims = variance_17_keep_dims_0, x = inputs_sq_17_cast_fp16)[name = string("variance_17_cast_fp16")]; + fp16 var_747_to_fp16 = const()[name = string("op_747_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_748_cast_fp16 = add(x = variance_17_cast_fp16, y = var_747_to_fp16)[name = string("op_748_cast_fp16")]; + fp32 var_749_epsilon_0 = const()[name = string("op_749_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_749_cast_fp16 = rsqrt(epsilon = var_749_epsilon_0, x = var_748_cast_fp16)[name = string("op_749_cast_fp16")]; + tensor hidden_states_21_cast_fp16 = mul(x = inputs_17_cast_fp16, y = var_749_cast_fp16)[name = string("hidden_states_21_cast_fp16")]; + tensor w_17_to_fp16 = const()[name = string("w_17_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33590528)))]; + tensor obj_21_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_21_cast_fp16)[name = string("obj_21_cast_fp16")]; + string query_13_pad_type_0 = const()[name = string("query_13_pad_type_0"), val = string("valid")]; + tensor query_13_strides_0 = const()[name = string("query_13_strides_0"), val = tensor([1, 1])]; + tensor query_13_pad_0 = const()[name = string("query_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_13_dilations_0 = const()[name = string("query_13_dilations_0"), val = tensor([1, 1])]; + int32 query_13_groups_0 = const()[name = string("query_13_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33592640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35689856))))[name = string("layers_2_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor query_13_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_13_dilations_0, groups = query_13_groups_0, pad = query_13_pad_0, pad_type = query_13_pad_type_0, strides = query_13_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_21_cast_fp16)[name = string("query_13_cast_fp16")]; + string current_key_9_pad_type_0 = const()[name = string("current_key_9_pad_type_0"), val = string("valid")]; + tensor current_key_9_strides_0 = const()[name = string("current_key_9_strides_0"), val = tensor([1, 1])]; + tensor current_key_9_pad_0 = const()[name = string("current_key_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_9_dilations_0 = const()[name = string("current_key_9_dilations_0"), val = tensor([1, 1])]; + int32 current_key_9_groups_0 = const()[name = string("current_key_9_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35690432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36739072))))[name = string("layers_2_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_9_cast_fp16 = conv(dilations = current_key_9_dilations_0, groups = current_key_9_groups_0, pad = current_key_9_pad_0, pad_type = current_key_9_pad_type_0, strides = current_key_9_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_21_cast_fp16)[name = string("current_key_9_cast_fp16")]; + string current_value_5_pad_type_0 = const()[name = string("current_value_5_pad_type_0"), val = string("valid")]; + tensor current_value_5_strides_0 = const()[name = string("current_value_5_strides_0"), val = tensor([1, 1])]; + tensor current_value_5_pad_0 = const()[name = string("current_value_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_5_dilations_0 = const()[name = string("current_value_5_dilations_0"), val = tensor([1, 1])]; + int32 current_value_5_groups_0 = const()[name = string("current_value_5_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36739648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37788288))))[name = string("layers_2_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor current_value_5_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_5_dilations_0, groups = current_value_5_groups_0, pad = current_value_5_pad_0, pad_type = current_value_5_pad_type_0, strides = current_value_5_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_21_cast_fp16)[name = string("current_value_5_cast_fp16")]; + tensor var_786 = const()[name = string("op_786"), val = tensor([16, 128, 1, 1])]; + tensor inputs_19_cast_fp16 = reshape(shape = var_786, x = query_13_cast_fp16)[name = string("inputs_19_cast_fp16")]; + tensor inputs_sq_19_cast_fp16 = mul(x = inputs_19_cast_fp16, y = inputs_19_cast_fp16)[name = string("inputs_sq_19_cast_fp16")]; + tensor variance_19_axes_0 = const()[name = string("variance_19_axes_0"), val = tensor([1])]; + bool variance_19_keep_dims_0 = const()[name = string("variance_19_keep_dims_0"), val = bool(true)]; + tensor variance_19_cast_fp16 = reduce_mean(axes = variance_19_axes_0, keep_dims = variance_19_keep_dims_0, x = inputs_sq_19_cast_fp16)[name = string("variance_19_cast_fp16")]; + fp16 var_792_to_fp16 = const()[name = string("op_792_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_793_cast_fp16 = add(x = variance_19_cast_fp16, y = var_792_to_fp16)[name = string("op_793_cast_fp16")]; + fp32 var_794_epsilon_0 = const()[name = string("op_794_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_794_cast_fp16 = rsqrt(epsilon = var_794_epsilon_0, x = var_793_cast_fp16)[name = string("op_794_cast_fp16")]; + tensor hidden_states_23_cast_fp16 = mul(x = inputs_19_cast_fp16, y = var_794_cast_fp16)[name = string("hidden_states_23_cast_fp16")]; + tensor w_19_to_fp16 = const()[name = string("w_19_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37788864)))]; + tensor query_normed_5_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_23_cast_fp16)[name = string("query_normed_5_cast_fp16")]; + tensor var_802 = const()[name = string("op_802"), val = tensor([8, 128, 1, 1])]; + tensor inputs_21_cast_fp16 = reshape(shape = var_802, x = current_key_9_cast_fp16)[name = string("inputs_21_cast_fp16")]; + tensor inputs_sq_21_cast_fp16 = mul(x = inputs_21_cast_fp16, y = inputs_21_cast_fp16)[name = string("inputs_sq_21_cast_fp16")]; + tensor variance_21_axes_0 = const()[name = string("variance_21_axes_0"), val = tensor([1])]; + bool variance_21_keep_dims_0 = const()[name = string("variance_21_keep_dims_0"), val = bool(true)]; + tensor variance_21_cast_fp16 = reduce_mean(axes = variance_21_axes_0, keep_dims = variance_21_keep_dims_0, x = inputs_sq_21_cast_fp16)[name = string("variance_21_cast_fp16")]; + fp16 var_808_to_fp16 = const()[name = string("op_808_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_809_cast_fp16 = add(x = variance_21_cast_fp16, y = var_808_to_fp16)[name = string("op_809_cast_fp16")]; + fp32 var_810_epsilon_0 = const()[name = string("op_810_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_810_cast_fp16 = rsqrt(epsilon = var_810_epsilon_0, x = var_809_cast_fp16)[name = string("op_810_cast_fp16")]; + tensor hidden_states_25_cast_fp16 = mul(x = inputs_21_cast_fp16, y = var_810_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; + tensor w_21_to_fp16 = const()[name = string("w_21_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37789184)))]; + tensor current_key_normed_5_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_25_cast_fp16)[name = string("current_key_normed_5_cast_fp16")]; + tensor var_828 = const()[name = string("op_828"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_17_cast_fp16 = reshape(shape = var_828, x = query_normed_5_cast_fp16)[name = string("mh_q_17_cast_fp16")]; + tensor var_830 = const()[name = string("op_830"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_17_cast_fp16 = reshape(shape = var_830, x = current_key_normed_5_cast_fp16)[name = string("mh_k_17_cast_fp16")]; + tensor var_839_begin_0 = const()[name = string("op_839_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_839_end_0 = const()[name = string("op_839_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_839_end_mask_0 = const()[name = string("op_839_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_839_cast_fp16 = slice_by_index(begin = var_839_begin_0, end = var_839_end_0, end_mask = var_839_end_mask_0, x = mh_q_17_cast_fp16)[name = string("op_839_cast_fp16")]; + tensor var_845_begin_0 = const()[name = string("op_845_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_845_end_0 = const()[name = string("op_845_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_845_end_mask_0 = const()[name = string("op_845_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_845_cast_fp16 = slice_by_index(begin = var_845_begin_0, end = var_845_end_0, end_mask = var_845_end_mask_0, x = mh_q_17_cast_fp16)[name = string("op_845_cast_fp16")]; + fp16 const_54_promoted_to_fp16 = const()[name = string("const_54_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_847_cast_fp16 = mul(x = var_845_cast_fp16, y = const_54_promoted_to_fp16)[name = string("op_847_cast_fp16")]; + bool var_849_interleave_0 = const()[name = string("op_849_interleave_0"), val = bool(false)]; + tensor var_849_cast_fp16 = concat(axis = var_733, interleave = var_849_interleave_0, values = (var_847_cast_fp16, var_839_cast_fp16))[name = string("op_849_cast_fp16")]; + tensor var_850_cast_fp16 = mul(x = var_849_cast_fp16, y = sin_1_to_fp16)[name = string("op_850_cast_fp16")]; + tensor mh_q_19_cast_fp16 = add(x = mh_q_17_cast_fp16, y = var_850_cast_fp16)[name = string("mh_q_19_cast_fp16")]; + tensor var_857_begin_0 = const()[name = string("op_857_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_857_end_0 = const()[name = string("op_857_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_857_end_mask_0 = const()[name = string("op_857_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_857_cast_fp16 = slice_by_index(begin = var_857_begin_0, end = var_857_end_0, end_mask = var_857_end_mask_0, x = mh_k_17_cast_fp16)[name = string("op_857_cast_fp16")]; + tensor var_863_begin_0 = const()[name = string("op_863_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_863_end_0 = const()[name = string("op_863_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_863_end_mask_0 = const()[name = string("op_863_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_863_cast_fp16 = slice_by_index(begin = var_863_begin_0, end = var_863_end_0, end_mask = var_863_end_mask_0, x = mh_k_17_cast_fp16)[name = string("op_863_cast_fp16")]; + fp16 const_57_promoted_to_fp16 = const()[name = string("const_57_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_865_cast_fp16 = mul(x = var_863_cast_fp16, y = const_57_promoted_to_fp16)[name = string("op_865_cast_fp16")]; + bool var_867_interleave_0 = const()[name = string("op_867_interleave_0"), val = bool(false)]; + tensor var_867_cast_fp16 = concat(axis = var_733, interleave = var_867_interleave_0, values = (var_865_cast_fp16, var_857_cast_fp16))[name = string("op_867_cast_fp16")]; + tensor var_868_cast_fp16 = mul(x = var_867_cast_fp16, y = sin_1_to_fp16)[name = string("op_868_cast_fp16")]; + tensor mh_k_19_cast_fp16 = add(x = mh_k_17_cast_fp16, y = var_868_cast_fp16)[name = string("mh_k_19_cast_fp16")]; + tensor var_872 = const()[name = string("op_872"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_11_cast_fp16 = reshape(shape = var_872, x = mh_k_19_cast_fp16)[name = string("current_key_11_cast_fp16")]; + tensor var_876_to_fp16 = const()[name = string("op_876_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112273792)))]; + tensor var_880_cast_fp16 = mul(x = current_key_11_cast_fp16, y = var_876_to_fp16)[name = string("op_880_cast_fp16")]; + tensor var_884_cast_fp16 = mul(x = current_value_5_cast_fp16, y = var_876_to_fp16)[name = string("op_884_cast_fp16")]; + fp16 var_891_to_fp16 = const()[name = string("op_891_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_23_cast_fp16 = mul(x = mh_q_19_cast_fp16, y = var_891_to_fp16)[name = string("mh_q_23_cast_fp16")]; + tensor var_893 = const()[name = string("op_893"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_21_cast_fp16 = reshape(shape = var_893, x = var_880_cast_fp16)[name = string("mh_k_21_cast_fp16")]; + tensor var_895 = const()[name = string("op_895"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_9_cast_fp16 = reshape(shape = var_895, x = var_884_cast_fp16)[name = string("mh_v_9_cast_fp16")]; + tensor transpose_8_perm_0 = const()[name = string("transpose_8_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_4_reps_0 = const()[name = string("tile_4_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_8_cast_fp16 = transpose(perm = transpose_8_perm_0, x = mh_k_21_cast_fp16)[name = string("transpose_461")]; + tensor tile_4_cast_fp16 = tile(reps = tile_4_reps_0, x = transpose_8_cast_fp16)[name = string("tile_4_cast_fp16")]; + tensor concat_12 = const()[name = string("concat_12"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_8_cast_fp16 = reshape(shape = concat_12, x = tile_4_cast_fp16)[name = string("reshape_8_cast_fp16")]; + tensor transpose_9_perm_0 = const()[name = string("transpose_9_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_13 = const()[name = string("concat_13"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_9_cast_fp16 = transpose(perm = transpose_9_perm_0, x = reshape_8_cast_fp16)[name = string("transpose_460")]; + tensor reshape_9_cast_fp16 = reshape(shape = concat_13, x = transpose_9_cast_fp16)[name = string("reshape_9_cast_fp16")]; + tensor transpose_10_perm_0 = const()[name = string("transpose_10_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_5_reps_0 = const()[name = string("tile_5_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_10_cast_fp16 = transpose(perm = transpose_10_perm_0, x = mh_v_9_cast_fp16)[name = string("transpose_459")]; + tensor tile_5_cast_fp16 = tile(reps = tile_5_reps_0, x = transpose_10_cast_fp16)[name = string("tile_5_cast_fp16")]; + tensor concat_14 = const()[name = string("concat_14"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_10_cast_fp16 = reshape(shape = concat_14, x = tile_5_cast_fp16)[name = string("reshape_10_cast_fp16")]; + tensor transpose_11_perm_0 = const()[name = string("transpose_11_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_15 = const()[name = string("concat_15"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_11_cast_fp16 = transpose(perm = transpose_11_perm_0, x = reshape_10_cast_fp16)[name = string("transpose_458")]; + tensor reshape_11_cast_fp16 = reshape(shape = concat_15, x = transpose_11_cast_fp16)[name = string("reshape_11_cast_fp16")]; + tensor transpose_329_perm_0 = const()[name = string("transpose_329_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_13_transpose_x_1 = const()[name = string("mh_w_13_transpose_x_1"), val = bool(true)]; + bool mh_w_13_transpose_y_1 = const()[name = string("mh_w_13_transpose_y_1"), val = bool(false)]; + tensor transpose_329_cast_fp16 = transpose(perm = transpose_329_perm_0, x = reshape_9_cast_fp16)[name = string("transpose_457")]; + tensor mh_w_13_cast_fp16 = matmul(transpose_x = mh_w_13_transpose_x_1, transpose_y = mh_w_13_transpose_y_1, x = mh_q_23_cast_fp16, y = transpose_329_cast_fp16)[name = string("mh_w_13_cast_fp16")]; + tensor var_903_to_fp16 = const()[name = string("op_903_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112273920)))]; + tensor mh_w_15_cast_fp16 = add(x = mh_w_13_cast_fp16, y = var_903_to_fp16)[name = string("mh_w_15_cast_fp16")]; + tensor mh_w_17_cast_fp16 = softmax(axis = var_723, x = mh_w_15_cast_fp16)[name = string("mh_w_17_cast_fp16")]; + tensor transpose_330_perm_0 = const()[name = string("transpose_330_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_5_transpose_x_1 = const()[name = string("attn_5_transpose_x_1"), val = bool(false)]; + bool attn_5_transpose_y_1 = const()[name = string("attn_5_transpose_y_1"), val = bool(true)]; + tensor transpose_330_cast_fp16 = transpose(perm = transpose_330_perm_0, x = reshape_11_cast_fp16)[name = string("transpose_456")]; + tensor attn_5_cast_fp16 = matmul(transpose_x = attn_5_transpose_x_1, transpose_y = attn_5_transpose_y_1, x = transpose_330_cast_fp16, y = mh_w_17_cast_fp16)[name = string("attn_5_cast_fp16")]; + tensor var_909 = const()[name = string("op_909"), val = tensor([1, 2048, 1, 1])]; + tensor input_17_cast_fp16 = reshape(shape = var_909, x = attn_5_cast_fp16)[name = string("input_17_cast_fp16")]; + string obj_27_pad_type_0 = const()[name = string("obj_27_pad_type_0"), val = string("valid")]; + tensor obj_27_strides_0 = const()[name = string("obj_27_strides_0"), val = tensor([1, 1])]; + tensor obj_27_pad_0 = const()[name = string("obj_27_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_27_dilations_0 = const()[name = string("obj_27_dilations_0"), val = tensor([1, 1])]; + int32 obj_27_groups_0 = const()[name = string("obj_27_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37789504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39886720))))[name = string("layers_2_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_27_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_27_dilations_0, groups = obj_27_groups_0, pad = obj_27_pad_0, pad_type = obj_27_pad_type_0, strides = obj_27_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_17_cast_fp16)[name = string("obj_27_cast_fp16")]; + tensor inputs_23_cast_fp16 = add(x = inputs_17_cast_fp16, y = obj_27_cast_fp16)[name = string("inputs_23_cast_fp16")]; + tensor inputs_sq_23_cast_fp16 = mul(x = inputs_23_cast_fp16, y = inputs_23_cast_fp16)[name = string("inputs_sq_23_cast_fp16")]; + tensor variance_23_axes_0 = const()[name = string("variance_23_axes_0"), val = tensor([1])]; + bool variance_23_keep_dims_0 = const()[name = string("variance_23_keep_dims_0"), val = bool(true)]; + tensor variance_23_cast_fp16 = reduce_mean(axes = variance_23_axes_0, keep_dims = variance_23_keep_dims_0, x = inputs_sq_23_cast_fp16)[name = string("variance_23_cast_fp16")]; + fp16 var_927_to_fp16 = const()[name = string("op_927_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_928_cast_fp16 = add(x = variance_23_cast_fp16, y = var_927_to_fp16)[name = string("op_928_cast_fp16")]; + fp32 var_929_epsilon_0 = const()[name = string("op_929_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_929_cast_fp16 = rsqrt(epsilon = var_929_epsilon_0, x = var_928_cast_fp16)[name = string("op_929_cast_fp16")]; + tensor hidden_states_27_cast_fp16 = mul(x = inputs_23_cast_fp16, y = var_929_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; + tensor w_23_to_fp16 = const()[name = string("w_23_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39887296)))]; + tensor input_19_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_27_cast_fp16)[name = string("input_19_cast_fp16")]; + string input_21_pad_type_0 = const()[name = string("input_21_pad_type_0"), val = string("valid")]; + tensor input_21_strides_0 = const()[name = string("input_21_strides_0"), val = tensor([1, 1])]; + tensor input_21_pad_0 = const()[name = string("input_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_21_dilations_0 = const()[name = string("input_21_dilations_0"), val = tensor([1, 1])]; + int32 input_21_groups_0 = const()[name = string("input_21_groups_0"), val = int32(1)]; + tensor layers_2_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39889408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43035200))))[name = string("layers_2_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_21_cast_fp16 = conv(dilations = input_21_dilations_0, groups = input_21_groups_0, pad = input_21_pad_0, pad_type = input_21_pad_type_0, strides = input_21_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_19_cast_fp16)[name = string("input_21_cast_fp16")]; + tensor var_943_cast_fp16 = silu(x = input_21_cast_fp16)[name = string("op_943_cast_fp16")]; + string var_949_pad_type_0 = const()[name = string("op_949_pad_type_0"), val = string("valid")]; + tensor var_949_strides_0 = const()[name = string("op_949_strides_0"), val = tensor([1, 1])]; + tensor var_949_pad_0 = const()[name = string("op_949_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_949_dilations_0 = const()[name = string("op_949_dilations_0"), val = tensor([1, 1])]; + int32 var_949_groups_0 = const()[name = string("op_949_groups_0"), val = int32(1)]; + tensor layers_2_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43035776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46181568))))[name = string("layers_2_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_949_cast_fp16 = conv(dilations = var_949_dilations_0, groups = var_949_groups_0, pad = var_949_pad_0, pad_type = var_949_pad_type_0, strides = var_949_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_19_cast_fp16)[name = string("op_949_cast_fp16")]; + tensor input_23_cast_fp16 = mul(x = var_943_cast_fp16, y = var_949_cast_fp16)[name = string("input_23_cast_fp16")]; + string hidden_states_29_pad_type_0 = const()[name = string("hidden_states_29_pad_type_0"), val = string("valid")]; + tensor hidden_states_29_strides_0 = const()[name = string("hidden_states_29_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_29_pad_0 = const()[name = string("hidden_states_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_29_dilations_0 = const()[name = string("hidden_states_29_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_29_groups_0 = const()[name = string("hidden_states_29_groups_0"), val = int32(1)]; + tensor layers_2_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46182144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49327936))))[name = string("layers_2_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_29_cast_fp16 = conv(dilations = hidden_states_29_dilations_0, groups = hidden_states_29_groups_0, pad = hidden_states_29_pad_0, pad_type = hidden_states_29_pad_type_0, strides = hidden_states_29_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_23_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; + tensor inputs_25_cast_fp16 = add(x = inputs_23_cast_fp16, y = hidden_states_29_cast_fp16)[name = string("inputs_25_cast_fp16")]; + int32 var_997 = const()[name = string("op_997"), val = int32(3)]; + int32 var_1007 = const()[name = string("op_1007"), val = int32(-2)]; + tensor inputs_sq_25_cast_fp16 = mul(x = inputs_25_cast_fp16, y = inputs_25_cast_fp16)[name = string("inputs_sq_25_cast_fp16")]; + tensor variance_25_axes_0 = const()[name = string("variance_25_axes_0"), val = tensor([1])]; + bool variance_25_keep_dims_0 = const()[name = string("variance_25_keep_dims_0"), val = bool(true)]; + tensor variance_25_cast_fp16 = reduce_mean(axes = variance_25_axes_0, keep_dims = variance_25_keep_dims_0, x = inputs_sq_25_cast_fp16)[name = string("variance_25_cast_fp16")]; + fp16 var_1021_to_fp16 = const()[name = string("op_1021_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1022_cast_fp16 = add(x = variance_25_cast_fp16, y = var_1021_to_fp16)[name = string("op_1022_cast_fp16")]; + fp32 var_1023_epsilon_0 = const()[name = string("op_1023_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1023_cast_fp16 = rsqrt(epsilon = var_1023_epsilon_0, x = var_1022_cast_fp16)[name = string("op_1023_cast_fp16")]; + tensor hidden_states_31_cast_fp16 = mul(x = inputs_25_cast_fp16, y = var_1023_cast_fp16)[name = string("hidden_states_31_cast_fp16")]; + tensor w_25_to_fp16 = const()[name = string("w_25_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49328512)))]; + tensor obj_29_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_31_cast_fp16)[name = string("obj_29_cast_fp16")]; + string query_19_pad_type_0 = const()[name = string("query_19_pad_type_0"), val = string("valid")]; + tensor query_19_strides_0 = const()[name = string("query_19_strides_0"), val = tensor([1, 1])]; + tensor query_19_pad_0 = const()[name = string("query_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_19_dilations_0 = const()[name = string("query_19_dilations_0"), val = tensor([1, 1])]; + int32 query_19_groups_0 = const()[name = string("query_19_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49330624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51427840))))[name = string("layers_3_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor query_19_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_19_dilations_0, groups = query_19_groups_0, pad = query_19_pad_0, pad_type = query_19_pad_type_0, strides = query_19_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_29_cast_fp16)[name = string("query_19_cast_fp16")]; + string current_key_13_pad_type_0 = const()[name = string("current_key_13_pad_type_0"), val = string("valid")]; + tensor current_key_13_strides_0 = const()[name = string("current_key_13_strides_0"), val = tensor([1, 1])]; + tensor current_key_13_pad_0 = const()[name = string("current_key_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_13_dilations_0 = const()[name = string("current_key_13_dilations_0"), val = tensor([1, 1])]; + int32 current_key_13_groups_0 = const()[name = string("current_key_13_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51428416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52477056))))[name = string("layers_3_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_13_cast_fp16 = conv(dilations = current_key_13_dilations_0, groups = current_key_13_groups_0, pad = current_key_13_pad_0, pad_type = current_key_13_pad_type_0, strides = current_key_13_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_29_cast_fp16)[name = string("current_key_13_cast_fp16")]; + string current_value_7_pad_type_0 = const()[name = string("current_value_7_pad_type_0"), val = string("valid")]; + tensor current_value_7_strides_0 = const()[name = string("current_value_7_strides_0"), val = tensor([1, 1])]; + tensor current_value_7_pad_0 = const()[name = string("current_value_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_7_dilations_0 = const()[name = string("current_value_7_dilations_0"), val = tensor([1, 1])]; + int32 current_value_7_groups_0 = const()[name = string("current_value_7_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52477632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53526272))))[name = string("layers_3_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor current_value_7_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_7_dilations_0, groups = current_value_7_groups_0, pad = current_value_7_pad_0, pad_type = current_value_7_pad_type_0, strides = current_value_7_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_29_cast_fp16)[name = string("current_value_7_cast_fp16")]; + tensor var_1060 = const()[name = string("op_1060"), val = tensor([16, 128, 1, 1])]; + tensor inputs_27_cast_fp16 = reshape(shape = var_1060, x = query_19_cast_fp16)[name = string("inputs_27_cast_fp16")]; + tensor inputs_sq_27_cast_fp16 = mul(x = inputs_27_cast_fp16, y = inputs_27_cast_fp16)[name = string("inputs_sq_27_cast_fp16")]; + tensor variance_27_axes_0 = const()[name = string("variance_27_axes_0"), val = tensor([1])]; + bool variance_27_keep_dims_0 = const()[name = string("variance_27_keep_dims_0"), val = bool(true)]; + tensor variance_27_cast_fp16 = reduce_mean(axes = variance_27_axes_0, keep_dims = variance_27_keep_dims_0, x = inputs_sq_27_cast_fp16)[name = string("variance_27_cast_fp16")]; + fp16 var_1066_to_fp16 = const()[name = string("op_1066_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1067_cast_fp16 = add(x = variance_27_cast_fp16, y = var_1066_to_fp16)[name = string("op_1067_cast_fp16")]; + fp32 var_1068_epsilon_0 = const()[name = string("op_1068_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1068_cast_fp16 = rsqrt(epsilon = var_1068_epsilon_0, x = var_1067_cast_fp16)[name = string("op_1068_cast_fp16")]; + tensor hidden_states_33_cast_fp16 = mul(x = inputs_27_cast_fp16, y = var_1068_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; + tensor w_27_to_fp16 = const()[name = string("w_27_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53526848)))]; + tensor query_normed_7_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_33_cast_fp16)[name = string("query_normed_7_cast_fp16")]; + tensor var_1076 = const()[name = string("op_1076"), val = tensor([8, 128, 1, 1])]; + tensor inputs_29_cast_fp16 = reshape(shape = var_1076, x = current_key_13_cast_fp16)[name = string("inputs_29_cast_fp16")]; + tensor inputs_sq_29_cast_fp16 = mul(x = inputs_29_cast_fp16, y = inputs_29_cast_fp16)[name = string("inputs_sq_29_cast_fp16")]; + tensor variance_29_axes_0 = const()[name = string("variance_29_axes_0"), val = tensor([1])]; + bool variance_29_keep_dims_0 = const()[name = string("variance_29_keep_dims_0"), val = bool(true)]; + tensor variance_29_cast_fp16 = reduce_mean(axes = variance_29_axes_0, keep_dims = variance_29_keep_dims_0, x = inputs_sq_29_cast_fp16)[name = string("variance_29_cast_fp16")]; + fp16 var_1082_to_fp16 = const()[name = string("op_1082_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1083_cast_fp16 = add(x = variance_29_cast_fp16, y = var_1082_to_fp16)[name = string("op_1083_cast_fp16")]; + fp32 var_1084_epsilon_0 = const()[name = string("op_1084_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1084_cast_fp16 = rsqrt(epsilon = var_1084_epsilon_0, x = var_1083_cast_fp16)[name = string("op_1084_cast_fp16")]; + tensor hidden_states_35_cast_fp16 = mul(x = inputs_29_cast_fp16, y = var_1084_cast_fp16)[name = string("hidden_states_35_cast_fp16")]; + tensor w_29_to_fp16 = const()[name = string("w_29_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53527168)))]; + tensor current_key_normed_7_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_35_cast_fp16)[name = string("current_key_normed_7_cast_fp16")]; + tensor var_1102 = const()[name = string("op_1102"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_25_cast_fp16 = reshape(shape = var_1102, x = query_normed_7_cast_fp16)[name = string("mh_q_25_cast_fp16")]; + tensor var_1104 = const()[name = string("op_1104"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_25_cast_fp16 = reshape(shape = var_1104, x = current_key_normed_7_cast_fp16)[name = string("mh_k_25_cast_fp16")]; + tensor var_1113_begin_0 = const()[name = string("op_1113_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1113_end_0 = const()[name = string("op_1113_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_1113_end_mask_0 = const()[name = string("op_1113_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1113_cast_fp16 = slice_by_index(begin = var_1113_begin_0, end = var_1113_end_0, end_mask = var_1113_end_mask_0, x = mh_q_25_cast_fp16)[name = string("op_1113_cast_fp16")]; + tensor var_1119_begin_0 = const()[name = string("op_1119_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1119_end_0 = const()[name = string("op_1119_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_1119_end_mask_0 = const()[name = string("op_1119_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1119_cast_fp16 = slice_by_index(begin = var_1119_begin_0, end = var_1119_end_0, end_mask = var_1119_end_mask_0, x = mh_q_25_cast_fp16)[name = string("op_1119_cast_fp16")]; + fp16 const_74_promoted_to_fp16 = const()[name = string("const_74_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1121_cast_fp16 = mul(x = var_1119_cast_fp16, y = const_74_promoted_to_fp16)[name = string("op_1121_cast_fp16")]; + bool var_1123_interleave_0 = const()[name = string("op_1123_interleave_0"), val = bool(false)]; + tensor var_1123_cast_fp16 = concat(axis = var_1007, interleave = var_1123_interleave_0, values = (var_1121_cast_fp16, var_1113_cast_fp16))[name = string("op_1123_cast_fp16")]; + tensor var_1124_cast_fp16 = mul(x = var_1123_cast_fp16, y = sin_1_to_fp16)[name = string("op_1124_cast_fp16")]; + tensor mh_q_27_cast_fp16 = add(x = mh_q_25_cast_fp16, y = var_1124_cast_fp16)[name = string("mh_q_27_cast_fp16")]; + tensor var_1131_begin_0 = const()[name = string("op_1131_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1131_end_0 = const()[name = string("op_1131_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_1131_end_mask_0 = const()[name = string("op_1131_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1131_cast_fp16 = slice_by_index(begin = var_1131_begin_0, end = var_1131_end_0, end_mask = var_1131_end_mask_0, x = mh_k_25_cast_fp16)[name = string("op_1131_cast_fp16")]; + tensor var_1137_begin_0 = const()[name = string("op_1137_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1137_end_0 = const()[name = string("op_1137_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_1137_end_mask_0 = const()[name = string("op_1137_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1137_cast_fp16 = slice_by_index(begin = var_1137_begin_0, end = var_1137_end_0, end_mask = var_1137_end_mask_0, x = mh_k_25_cast_fp16)[name = string("op_1137_cast_fp16")]; + fp16 const_77_promoted_to_fp16 = const()[name = string("const_77_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1139_cast_fp16 = mul(x = var_1137_cast_fp16, y = const_77_promoted_to_fp16)[name = string("op_1139_cast_fp16")]; + bool var_1141_interleave_0 = const()[name = string("op_1141_interleave_0"), val = bool(false)]; + tensor var_1141_cast_fp16 = concat(axis = var_1007, interleave = var_1141_interleave_0, values = (var_1139_cast_fp16, var_1131_cast_fp16))[name = string("op_1141_cast_fp16")]; + tensor var_1142_cast_fp16 = mul(x = var_1141_cast_fp16, y = sin_1_to_fp16)[name = string("op_1142_cast_fp16")]; + tensor mh_k_27_cast_fp16 = add(x = mh_k_25_cast_fp16, y = var_1142_cast_fp16)[name = string("mh_k_27_cast_fp16")]; + tensor var_1146 = const()[name = string("op_1146"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_15_cast_fp16 = reshape(shape = var_1146, x = mh_k_27_cast_fp16)[name = string("current_key_15_cast_fp16")]; + tensor var_1150_to_fp16 = const()[name = string("op_1150_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112273792)))]; + tensor var_1154_cast_fp16 = mul(x = current_key_15_cast_fp16, y = var_1150_to_fp16)[name = string("op_1154_cast_fp16")]; + tensor var_1158_cast_fp16 = mul(x = current_value_7_cast_fp16, y = var_1150_to_fp16)[name = string("op_1158_cast_fp16")]; + fp16 var_1165_to_fp16 = const()[name = string("op_1165_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_31_cast_fp16 = mul(x = mh_q_27_cast_fp16, y = var_1165_to_fp16)[name = string("mh_q_31_cast_fp16")]; + tensor var_1167 = const()[name = string("op_1167"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_29_cast_fp16 = reshape(shape = var_1167, x = var_1154_cast_fp16)[name = string("mh_k_29_cast_fp16")]; + tensor var_1169 = const()[name = string("op_1169"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_13_cast_fp16 = reshape(shape = var_1169, x = var_1158_cast_fp16)[name = string("mh_v_13_cast_fp16")]; + tensor transpose_12_perm_0 = const()[name = string("transpose_12_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_6_reps_0 = const()[name = string("tile_6_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_12_cast_fp16 = transpose(perm = transpose_12_perm_0, x = mh_k_29_cast_fp16)[name = string("transpose_455")]; + tensor tile_6_cast_fp16 = tile(reps = tile_6_reps_0, x = transpose_12_cast_fp16)[name = string("tile_6_cast_fp16")]; + tensor concat_16 = const()[name = string("concat_16"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_12_cast_fp16 = reshape(shape = concat_16, x = tile_6_cast_fp16)[name = string("reshape_12_cast_fp16")]; + tensor transpose_13_perm_0 = const()[name = string("transpose_13_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_17 = const()[name = string("concat_17"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_13_cast_fp16 = transpose(perm = transpose_13_perm_0, x = reshape_12_cast_fp16)[name = string("transpose_454")]; + tensor reshape_13_cast_fp16 = reshape(shape = concat_17, x = transpose_13_cast_fp16)[name = string("reshape_13_cast_fp16")]; + tensor transpose_14_perm_0 = const()[name = string("transpose_14_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_7_reps_0 = const()[name = string("tile_7_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_14_cast_fp16 = transpose(perm = transpose_14_perm_0, x = mh_v_13_cast_fp16)[name = string("transpose_453")]; + tensor tile_7_cast_fp16 = tile(reps = tile_7_reps_0, x = transpose_14_cast_fp16)[name = string("tile_7_cast_fp16")]; + tensor concat_18 = const()[name = string("concat_18"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_14_cast_fp16 = reshape(shape = concat_18, x = tile_7_cast_fp16)[name = string("reshape_14_cast_fp16")]; + tensor transpose_15_perm_0 = const()[name = string("transpose_15_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_19 = const()[name = string("concat_19"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_15_cast_fp16 = transpose(perm = transpose_15_perm_0, x = reshape_14_cast_fp16)[name = string("transpose_452")]; + tensor reshape_15_cast_fp16 = reshape(shape = concat_19, x = transpose_15_cast_fp16)[name = string("reshape_15_cast_fp16")]; + tensor transpose_333_perm_0 = const()[name = string("transpose_333_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_19_transpose_x_1 = const()[name = string("mh_w_19_transpose_x_1"), val = bool(true)]; + bool mh_w_19_transpose_y_1 = const()[name = string("mh_w_19_transpose_y_1"), val = bool(false)]; + tensor transpose_333_cast_fp16 = transpose(perm = transpose_333_perm_0, x = reshape_13_cast_fp16)[name = string("transpose_451")]; + tensor mh_w_19_cast_fp16 = matmul(transpose_x = mh_w_19_transpose_x_1, transpose_y = mh_w_19_transpose_y_1, x = mh_q_31_cast_fp16, y = transpose_333_cast_fp16)[name = string("mh_w_19_cast_fp16")]; + tensor var_1177_to_fp16 = const()[name = string("op_1177_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112273920)))]; + tensor mh_w_21_cast_fp16 = add(x = mh_w_19_cast_fp16, y = var_1177_to_fp16)[name = string("mh_w_21_cast_fp16")]; + tensor mh_w_23_cast_fp16 = softmax(axis = var_997, x = mh_w_21_cast_fp16)[name = string("mh_w_23_cast_fp16")]; + tensor transpose_334_perm_0 = const()[name = string("transpose_334_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_7_transpose_x_1 = const()[name = string("attn_7_transpose_x_1"), val = bool(false)]; + bool attn_7_transpose_y_1 = const()[name = string("attn_7_transpose_y_1"), val = bool(true)]; + tensor transpose_334_cast_fp16 = transpose(perm = transpose_334_perm_0, x = reshape_15_cast_fp16)[name = string("transpose_450")]; + tensor attn_7_cast_fp16 = matmul(transpose_x = attn_7_transpose_x_1, transpose_y = attn_7_transpose_y_1, x = transpose_334_cast_fp16, y = mh_w_23_cast_fp16)[name = string("attn_7_cast_fp16")]; + tensor var_1183 = const()[name = string("op_1183"), val = tensor([1, 2048, 1, 1])]; + tensor input_25_cast_fp16 = reshape(shape = var_1183, x = attn_7_cast_fp16)[name = string("input_25_cast_fp16")]; + string obj_35_pad_type_0 = const()[name = string("obj_35_pad_type_0"), val = string("valid")]; + tensor obj_35_strides_0 = const()[name = string("obj_35_strides_0"), val = tensor([1, 1])]; + tensor obj_35_pad_0 = const()[name = string("obj_35_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_35_dilations_0 = const()[name = string("obj_35_dilations_0"), val = tensor([1, 1])]; + int32 obj_35_groups_0 = const()[name = string("obj_35_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53527488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55624704))))[name = string("layers_3_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_35_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_35_dilations_0, groups = obj_35_groups_0, pad = obj_35_pad_0, pad_type = obj_35_pad_type_0, strides = obj_35_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_25_cast_fp16)[name = string("obj_35_cast_fp16")]; + tensor inputs_31_cast_fp16 = add(x = inputs_25_cast_fp16, y = obj_35_cast_fp16)[name = string("inputs_31_cast_fp16")]; + tensor inputs_sq_31_cast_fp16 = mul(x = inputs_31_cast_fp16, y = inputs_31_cast_fp16)[name = string("inputs_sq_31_cast_fp16")]; + tensor variance_31_axes_0 = const()[name = string("variance_31_axes_0"), val = tensor([1])]; + bool variance_31_keep_dims_0 = const()[name = string("variance_31_keep_dims_0"), val = bool(true)]; + tensor variance_31_cast_fp16 = reduce_mean(axes = variance_31_axes_0, keep_dims = variance_31_keep_dims_0, x = inputs_sq_31_cast_fp16)[name = string("variance_31_cast_fp16")]; + fp16 var_1201_to_fp16 = const()[name = string("op_1201_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1202_cast_fp16 = add(x = variance_31_cast_fp16, y = var_1201_to_fp16)[name = string("op_1202_cast_fp16")]; + fp32 var_1203_epsilon_0 = const()[name = string("op_1203_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1203_cast_fp16 = rsqrt(epsilon = var_1203_epsilon_0, x = var_1202_cast_fp16)[name = string("op_1203_cast_fp16")]; + tensor hidden_states_37_cast_fp16 = mul(x = inputs_31_cast_fp16, y = var_1203_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; + tensor w_31_to_fp16 = const()[name = string("w_31_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55625280)))]; + tensor input_27_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_37_cast_fp16)[name = string("input_27_cast_fp16")]; + string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; + tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; + tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; + int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; + tensor layers_3_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55627392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(58773184))))[name = string("layers_3_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_29_cast_fp16 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_27_cast_fp16)[name = string("input_29_cast_fp16")]; + tensor var_1217_cast_fp16 = silu(x = input_29_cast_fp16)[name = string("op_1217_cast_fp16")]; + string var_1223_pad_type_0 = const()[name = string("op_1223_pad_type_0"), val = string("valid")]; + tensor var_1223_strides_0 = const()[name = string("op_1223_strides_0"), val = tensor([1, 1])]; + tensor var_1223_pad_0 = const()[name = string("op_1223_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1223_dilations_0 = const()[name = string("op_1223_dilations_0"), val = tensor([1, 1])]; + int32 var_1223_groups_0 = const()[name = string("op_1223_groups_0"), val = int32(1)]; + tensor layers_3_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(58773760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61919552))))[name = string("layers_3_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_1223_cast_fp16 = conv(dilations = var_1223_dilations_0, groups = var_1223_groups_0, pad = var_1223_pad_0, pad_type = var_1223_pad_type_0, strides = var_1223_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_27_cast_fp16)[name = string("op_1223_cast_fp16")]; + tensor input_31_cast_fp16 = mul(x = var_1217_cast_fp16, y = var_1223_cast_fp16)[name = string("input_31_cast_fp16")]; + string hidden_states_39_pad_type_0 = const()[name = string("hidden_states_39_pad_type_0"), val = string("valid")]; + tensor hidden_states_39_strides_0 = const()[name = string("hidden_states_39_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_39_pad_0 = const()[name = string("hidden_states_39_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_39_dilations_0 = const()[name = string("hidden_states_39_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_39_groups_0 = const()[name = string("hidden_states_39_groups_0"), val = int32(1)]; + tensor layers_3_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61920128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65065920))))[name = string("layers_3_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_39_cast_fp16 = conv(dilations = hidden_states_39_dilations_0, groups = hidden_states_39_groups_0, pad = hidden_states_39_pad_0, pad_type = hidden_states_39_pad_type_0, strides = hidden_states_39_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_31_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; + tensor inputs_33_cast_fp16 = add(x = inputs_31_cast_fp16, y = hidden_states_39_cast_fp16)[name = string("inputs_33_cast_fp16")]; + int32 var_1281 = const()[name = string("op_1281"), val = int32(-2)]; + tensor inputs_sq_33_cast_fp16 = mul(x = inputs_33_cast_fp16, y = inputs_33_cast_fp16)[name = string("inputs_sq_33_cast_fp16")]; + tensor variance_33_axes_0 = const()[name = string("variance_33_axes_0"), val = tensor([1])]; + bool variance_33_keep_dims_0 = const()[name = string("variance_33_keep_dims_0"), val = bool(true)]; + tensor variance_33_cast_fp16 = reduce_mean(axes = variance_33_axes_0, keep_dims = variance_33_keep_dims_0, x = inputs_sq_33_cast_fp16)[name = string("variance_33_cast_fp16")]; + fp16 var_1293_to_fp16 = const()[name = string("op_1293_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1294_cast_fp16 = add(x = variance_33_cast_fp16, y = var_1293_to_fp16)[name = string("op_1294_cast_fp16")]; + fp32 var_1295_epsilon_0 = const()[name = string("op_1295_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1295_cast_fp16 = rsqrt(epsilon = var_1295_epsilon_0, x = var_1294_cast_fp16)[name = string("op_1295_cast_fp16")]; + tensor hidden_states_41_cast_fp16 = mul(x = inputs_33_cast_fp16, y = var_1295_cast_fp16)[name = string("hidden_states_41_cast_fp16")]; + tensor w_33_to_fp16 = const()[name = string("w_33_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65066496)))]; + tensor obj_37_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_41_cast_fp16)[name = string("obj_37_cast_fp16")]; + string current_key_17_pad_type_0 = const()[name = string("current_key_17_pad_type_0"), val = string("valid")]; + tensor current_key_17_strides_0 = const()[name = string("current_key_17_strides_0"), val = tensor([1, 1])]; + tensor current_key_17_pad_0 = const()[name = string("current_key_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_17_dilations_0 = const()[name = string("current_key_17_dilations_0"), val = tensor([1, 1])]; + int32 current_key_17_groups_0 = const()[name = string("current_key_17_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67166400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68215040))))[name = string("layers_4_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_17_cast_fp16 = conv(dilations = current_key_17_dilations_0, groups = current_key_17_groups_0, pad = current_key_17_pad_0, pad_type = current_key_17_pad_type_0, strides = current_key_17_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_37_cast_fp16)[name = string("current_key_17_cast_fp16")]; + string current_value_9_pad_type_0 = const()[name = string("current_value_9_pad_type_0"), val = string("valid")]; + tensor current_value_9_strides_0 = const()[name = string("current_value_9_strides_0"), val = tensor([1, 1])]; + tensor current_value_9_pad_0 = const()[name = string("current_value_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_9_dilations_0 = const()[name = string("current_value_9_dilations_0"), val = tensor([1, 1])]; + int32 current_value_9_groups_0 = const()[name = string("current_value_9_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68215616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69264256))))[name = string("layers_4_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor current_value_9_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_9_dilations_0, groups = current_value_9_groups_0, pad = current_value_9_pad_0, pad_type = current_value_9_pad_type_0, strides = current_value_9_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_37_cast_fp16)[name = string("current_value_9_cast_fp16")]; + tensor var_1348 = const()[name = string("op_1348"), val = tensor([8, 128, 1, 1])]; + tensor inputs_37_cast_fp16 = reshape(shape = var_1348, x = current_key_17_cast_fp16)[name = string("inputs_37_cast_fp16")]; + tensor inputs_sq_37_cast_fp16 = mul(x = inputs_37_cast_fp16, y = inputs_37_cast_fp16)[name = string("inputs_sq_37_cast_fp16")]; + tensor variance_37_axes_0 = const()[name = string("variance_37_axes_0"), val = tensor([1])]; + bool variance_37_keep_dims_0 = const()[name = string("variance_37_keep_dims_0"), val = bool(true)]; + tensor variance_37_cast_fp16 = reduce_mean(axes = variance_37_axes_0, keep_dims = variance_37_keep_dims_0, x = inputs_sq_37_cast_fp16)[name = string("variance_37_cast_fp16")]; + fp16 var_1354_to_fp16 = const()[name = string("op_1354_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1355_cast_fp16 = add(x = variance_37_cast_fp16, y = var_1354_to_fp16)[name = string("op_1355_cast_fp16")]; + fp32 var_1356_epsilon_0 = const()[name = string("op_1356_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1356_cast_fp16 = rsqrt(epsilon = var_1356_epsilon_0, x = var_1355_cast_fp16)[name = string("op_1356_cast_fp16")]; + tensor hidden_states_45_cast_fp16 = mul(x = inputs_37_cast_fp16, y = var_1356_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; + tensor w_37_to_fp16 = const()[name = string("w_37_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69265152)))]; + tensor current_key_normed_9_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_45_cast_fp16)[name = string("current_key_normed_9_cast_fp16")]; + tensor var_1376 = const()[name = string("op_1376"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_33_cast_fp16 = reshape(shape = var_1376, x = current_key_normed_9_cast_fp16)[name = string("mh_k_33_cast_fp16")]; + tensor var_1403_begin_0 = const()[name = string("op_1403_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1403_end_0 = const()[name = string("op_1403_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_1403_end_mask_0 = const()[name = string("op_1403_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1403_cast_fp16 = slice_by_index(begin = var_1403_begin_0, end = var_1403_end_0, end_mask = var_1403_end_mask_0, x = mh_k_33_cast_fp16)[name = string("op_1403_cast_fp16")]; + tensor var_1409_begin_0 = const()[name = string("op_1409_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1409_end_0 = const()[name = string("op_1409_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_1409_end_mask_0 = const()[name = string("op_1409_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1409_cast_fp16 = slice_by_index(begin = var_1409_begin_0, end = var_1409_end_0, end_mask = var_1409_end_mask_0, x = mh_k_33_cast_fp16)[name = string("op_1409_cast_fp16")]; + fp16 const_97_promoted_to_fp16 = const()[name = string("const_97_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1411_cast_fp16 = mul(x = var_1409_cast_fp16, y = const_97_promoted_to_fp16)[name = string("op_1411_cast_fp16")]; + bool var_1413_interleave_0 = const()[name = string("op_1413_interleave_0"), val = bool(false)]; + tensor var_1413_cast_fp16 = concat(axis = var_1281, interleave = var_1413_interleave_0, values = (var_1411_cast_fp16, var_1403_cast_fp16))[name = string("op_1413_cast_fp16")]; + tensor var_1414_cast_fp16 = mul(x = var_1413_cast_fp16, y = sin_1_to_fp16)[name = string("op_1414_cast_fp16")]; + tensor mh_k_35_cast_fp16 = add(x = mh_k_33_cast_fp16, y = var_1414_cast_fp16)[name = string("mh_k_35_cast_fp16")]; + tensor var_1418 = const()[name = string("op_1418"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_19_cast_fp16 = reshape(shape = var_1418, x = mh_k_35_cast_fp16)[name = string("current_key_19_cast_fp16")]; + tensor var_148_to_fp16 = const()[name = string("op_148_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112273792)))]; + tensor var_1475_cast_fp16 = mul(x = current_key_19_cast_fp16, y = var_148_to_fp16)[name = string("op_1475_cast_fp16")]; + tensor var_1482_cast_fp16 = mul(x = current_value_9_cast_fp16, y = var_148_to_fp16)[name = string("op_1482_cast_fp16")]; + int32 var_1486 = const()[name = string("op_1486"), val = int32(1)]; + bool key_caches_3_interleave_0 = const()[name = string("key_caches_3_interleave_0"), val = bool(false)]; + tensor key_caches_3_cast_fp16 = concat(axis = var_1486, interleave = key_caches_3_interleave_0, values = (var_332_cast_fp16, var_606_cast_fp16, var_880_cast_fp16, var_1154_cast_fp16, var_1475_cast_fp16))[name = string("key_caches_3_cast_fp16")]; + int32 var_1489 = const()[name = string("op_1489"), val = int32(1)]; + bool value_caches_3_interleave_0 = const()[name = string("value_caches_3_interleave_0"), val = bool(false)]; + tensor value_caches_3_cast_fp16 = concat(axis = var_1489, interleave = value_caches_3_interleave_0, values = (var_336_cast_fp16, var_610_cast_fp16, var_884_cast_fp16, var_1158_cast_fp16, var_1482_cast_fp16))[name = string("value_caches_3_cast_fp16")]; + string inputs_41_pad_type_0 = const()[name = string("inputs_41_pad_type_0"), val = string("valid")]; + tensor inputs_41_strides_0 = const()[name = string("inputs_41_strides_0"), val = tensor([1, 1])]; + tensor inputs_41_pad_0 = const()[name = string("inputs_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_41_dilations_0 = const()[name = string("inputs_41_dilations_0"), val = tensor([1, 1])]; + int32 inputs_41_groups_0 = const()[name = string("inputs_41_groups_0"), val = int32(1)]; + tensor inputs_41_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_41_dilations_0, groups = inputs_41_groups_0, pad = inputs_41_pad_0, pad_type = inputs_41_pad_type_0, strides = inputs_41_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code0_embed)[name = string("inputs_41_cast_fp16")]; + tensor obj_47_begin_0 = const()[name = string("obj_47_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_47_end_0 = const()[name = string("obj_47_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_47_end_mask_0 = const()[name = string("obj_47_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_47_cast_fp16 = slice_by_index(begin = obj_47_begin_0, end = obj_47_end_0, end_mask = obj_47_end_mask_0, x = key_caches_3_cast_fp16)[name = string("obj_47_cast_fp16")]; + tensor obj_49_begin_0 = const()[name = string("obj_49_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_49_end_0 = const()[name = string("obj_49_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_49_end_mask_0 = const()[name = string("obj_49_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_49_cast_fp16 = slice_by_index(begin = obj_49_begin_0, end = obj_49_end_0, end_mask = obj_49_end_mask_0, x = value_caches_3_cast_fp16)[name = string("obj_49_cast_fp16")]; + int32 var_1589 = const()[name = string("op_1589"), val = int32(3)]; + int32 var_1599 = const()[name = string("op_1599"), val = int32(-2)]; + tensor inputs_sq_41_cast_fp16 = mul(x = inputs_41_cast_fp16, y = inputs_41_cast_fp16)[name = string("inputs_sq_41_cast_fp16")]; + tensor variance_41_axes_0 = const()[name = string("variance_41_axes_0"), val = tensor([1])]; + bool variance_41_keep_dims_0 = const()[name = string("variance_41_keep_dims_0"), val = bool(true)]; + tensor variance_41_cast_fp16 = reduce_mean(axes = variance_41_axes_0, keep_dims = variance_41_keep_dims_0, x = inputs_sq_41_cast_fp16)[name = string("variance_41_cast_fp16")]; + fp16 var_1613_to_fp16 = const()[name = string("op_1613_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1614_cast_fp16 = add(x = variance_41_cast_fp16, y = var_1613_to_fp16)[name = string("op_1614_cast_fp16")]; + fp32 var_1615_epsilon_0 = const()[name = string("op_1615_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1615_cast_fp16 = rsqrt(epsilon = var_1615_epsilon_0, x = var_1614_cast_fp16)[name = string("op_1615_cast_fp16")]; + tensor hidden_states_51_cast_fp16 = mul(x = inputs_41_cast_fp16, y = var_1615_cast_fp16)[name = string("hidden_states_51_cast_fp16")]; + tensor obj_45_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_51_cast_fp16)[name = string("obj_45_cast_fp16")]; + string query_31_pad_type_0 = const()[name = string("query_31_pad_type_0"), val = string("valid")]; + tensor query_31_strides_0 = const()[name = string("query_31_strides_0"), val = tensor([1, 1])]; + tensor query_31_pad_0 = const()[name = string("query_31_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_31_dilations_0 = const()[name = string("query_31_dilations_0"), val = tensor([1, 1])]; + int32 query_31_groups_0 = const()[name = string("query_31_groups_0"), val = int32(1)]; + tensor query_31_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_31_dilations_0, groups = query_31_groups_0, pad = query_31_pad_0, pad_type = query_31_pad_type_0, strides = query_31_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_45_cast_fp16)[name = string("query_31_cast_fp16")]; + string current_key_21_pad_type_0 = const()[name = string("current_key_21_pad_type_0"), val = string("valid")]; + tensor current_key_21_strides_0 = const()[name = string("current_key_21_strides_0"), val = tensor([1, 1])]; + tensor current_key_21_pad_0 = const()[name = string("current_key_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_21_dilations_0 = const()[name = string("current_key_21_dilations_0"), val = tensor([1, 1])]; + int32 current_key_21_groups_0 = const()[name = string("current_key_21_groups_0"), val = int32(1)]; + tensor current_key_21_cast_fp16 = conv(dilations = current_key_21_dilations_0, groups = current_key_21_groups_0, pad = current_key_21_pad_0, pad_type = current_key_21_pad_type_0, strides = current_key_21_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_45_cast_fp16)[name = string("current_key_21_cast_fp16")]; + string current_value_11_pad_type_0 = const()[name = string("current_value_11_pad_type_0"), val = string("valid")]; + tensor current_value_11_strides_0 = const()[name = string("current_value_11_strides_0"), val = tensor([1, 1])]; + tensor current_value_11_pad_0 = const()[name = string("current_value_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_11_dilations_0 = const()[name = string("current_value_11_dilations_0"), val = tensor([1, 1])]; + int32 current_value_11_groups_0 = const()[name = string("current_value_11_groups_0"), val = int32(1)]; + tensor current_value_11_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_11_dilations_0, groups = current_value_11_groups_0, pad = current_value_11_pad_0, pad_type = current_value_11_pad_type_0, strides = current_value_11_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_45_cast_fp16)[name = string("current_value_11_cast_fp16")]; + tensor var_1652 = const()[name = string("op_1652"), val = tensor([16, 128, 1, 1])]; + tensor inputs_43_cast_fp16 = reshape(shape = var_1652, x = query_31_cast_fp16)[name = string("inputs_43_cast_fp16")]; + tensor inputs_sq_43_cast_fp16 = mul(x = inputs_43_cast_fp16, y = inputs_43_cast_fp16)[name = string("inputs_sq_43_cast_fp16")]; + tensor variance_43_axes_0 = const()[name = string("variance_43_axes_0"), val = tensor([1])]; + bool variance_43_keep_dims_0 = const()[name = string("variance_43_keep_dims_0"), val = bool(true)]; + tensor variance_43_cast_fp16 = reduce_mean(axes = variance_43_axes_0, keep_dims = variance_43_keep_dims_0, x = inputs_sq_43_cast_fp16)[name = string("variance_43_cast_fp16")]; + fp16 var_1658_to_fp16 = const()[name = string("op_1658_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1659_cast_fp16 = add(x = variance_43_cast_fp16, y = var_1658_to_fp16)[name = string("op_1659_cast_fp16")]; + fp32 var_1660_epsilon_0 = const()[name = string("op_1660_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1660_cast_fp16 = rsqrt(epsilon = var_1660_epsilon_0, x = var_1659_cast_fp16)[name = string("op_1660_cast_fp16")]; + tensor hidden_states_53_cast_fp16 = mul(x = inputs_43_cast_fp16, y = var_1660_cast_fp16)[name = string("hidden_states_53_cast_fp16")]; + tensor query_normed_11_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_53_cast_fp16)[name = string("query_normed_11_cast_fp16")]; + tensor var_1668 = const()[name = string("op_1668"), val = tensor([8, 128, 1, 1])]; + tensor inputs_45_cast_fp16 = reshape(shape = var_1668, x = current_key_21_cast_fp16)[name = string("inputs_45_cast_fp16")]; + tensor inputs_sq_45_cast_fp16 = mul(x = inputs_45_cast_fp16, y = inputs_45_cast_fp16)[name = string("inputs_sq_45_cast_fp16")]; + tensor variance_45_axes_0 = const()[name = string("variance_45_axes_0"), val = tensor([1])]; + bool variance_45_keep_dims_0 = const()[name = string("variance_45_keep_dims_0"), val = bool(true)]; + tensor variance_45_cast_fp16 = reduce_mean(axes = variance_45_axes_0, keep_dims = variance_45_keep_dims_0, x = inputs_sq_45_cast_fp16)[name = string("variance_45_cast_fp16")]; + fp16 var_1674_to_fp16 = const()[name = string("op_1674_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1675_cast_fp16 = add(x = variance_45_cast_fp16, y = var_1674_to_fp16)[name = string("op_1675_cast_fp16")]; + fp32 var_1676_epsilon_0 = const()[name = string("op_1676_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1676_cast_fp16 = rsqrt(epsilon = var_1676_epsilon_0, x = var_1675_cast_fp16)[name = string("op_1676_cast_fp16")]; + tensor hidden_states_55_cast_fp16 = mul(x = inputs_45_cast_fp16, y = var_1676_cast_fp16)[name = string("hidden_states_55_cast_fp16")]; + tensor current_key_normed_11_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_55_cast_fp16)[name = string("current_key_normed_11_cast_fp16")]; + tensor var_1694 = const()[name = string("op_1694"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_41_cast_fp16 = reshape(shape = var_1694, x = query_normed_11_cast_fp16)[name = string("mh_q_41_cast_fp16")]; + tensor var_1696 = const()[name = string("op_1696"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_41_cast_fp16 = reshape(shape = var_1696, x = current_key_normed_11_cast_fp16)[name = string("mh_k_41_cast_fp16")]; + tensor cos_11_to_fp16 = const()[name = string("cos_11_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274048)))]; + tensor var_1700_cast_fp16 = mul(x = mh_q_41_cast_fp16, y = cos_11_to_fp16)[name = string("op_1700_cast_fp16")]; + tensor var_1705_begin_0 = const()[name = string("op_1705_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1705_end_0 = const()[name = string("op_1705_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_1705_end_mask_0 = const()[name = string("op_1705_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1705_cast_fp16 = slice_by_index(begin = var_1705_begin_0, end = var_1705_end_0, end_mask = var_1705_end_mask_0, x = mh_q_41_cast_fp16)[name = string("op_1705_cast_fp16")]; + tensor var_1711_begin_0 = const()[name = string("op_1711_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1711_end_0 = const()[name = string("op_1711_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_1711_end_mask_0 = const()[name = string("op_1711_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1711_cast_fp16 = slice_by_index(begin = var_1711_begin_0, end = var_1711_end_0, end_mask = var_1711_end_mask_0, x = mh_q_41_cast_fp16)[name = string("op_1711_cast_fp16")]; + fp16 const_115_promoted_to_fp16 = const()[name = string("const_115_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1713_cast_fp16 = mul(x = var_1711_cast_fp16, y = const_115_promoted_to_fp16)[name = string("op_1713_cast_fp16")]; + bool var_1715_interleave_0 = const()[name = string("op_1715_interleave_0"), val = bool(false)]; + tensor var_1715_cast_fp16 = concat(axis = var_1599, interleave = var_1715_interleave_0, values = (var_1713_cast_fp16, var_1705_cast_fp16))[name = string("op_1715_cast_fp16")]; + tensor sin_11_to_fp16 = const()[name = string("sin_11_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274368)))]; + tensor var_1716_cast_fp16 = mul(x = var_1715_cast_fp16, y = sin_11_to_fp16)[name = string("op_1716_cast_fp16")]; + tensor mh_q_43_cast_fp16 = add(x = var_1700_cast_fp16, y = var_1716_cast_fp16)[name = string("mh_q_43_cast_fp16")]; + tensor var_1718_cast_fp16 = mul(x = mh_k_41_cast_fp16, y = cos_11_to_fp16)[name = string("op_1718_cast_fp16")]; + tensor var_1723_begin_0 = const()[name = string("op_1723_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1723_end_0 = const()[name = string("op_1723_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_1723_end_mask_0 = const()[name = string("op_1723_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1723_cast_fp16 = slice_by_index(begin = var_1723_begin_0, end = var_1723_end_0, end_mask = var_1723_end_mask_0, x = mh_k_41_cast_fp16)[name = string("op_1723_cast_fp16")]; + tensor var_1729_begin_0 = const()[name = string("op_1729_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1729_end_0 = const()[name = string("op_1729_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_1729_end_mask_0 = const()[name = string("op_1729_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1729_cast_fp16 = slice_by_index(begin = var_1729_begin_0, end = var_1729_end_0, end_mask = var_1729_end_mask_0, x = mh_k_41_cast_fp16)[name = string("op_1729_cast_fp16")]; + fp16 const_118_promoted_to_fp16 = const()[name = string("const_118_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1731_cast_fp16 = mul(x = var_1729_cast_fp16, y = const_118_promoted_to_fp16)[name = string("op_1731_cast_fp16")]; + bool var_1733_interleave_0 = const()[name = string("op_1733_interleave_0"), val = bool(false)]; + tensor var_1733_cast_fp16 = concat(axis = var_1599, interleave = var_1733_interleave_0, values = (var_1731_cast_fp16, var_1723_cast_fp16))[name = string("op_1733_cast_fp16")]; + tensor var_1734_cast_fp16 = mul(x = var_1733_cast_fp16, y = sin_11_to_fp16)[name = string("op_1734_cast_fp16")]; + tensor mh_k_43_cast_fp16 = add(x = var_1718_cast_fp16, y = var_1734_cast_fp16)[name = string("mh_k_43_cast_fp16")]; + tensor var_1738 = const()[name = string("op_1738"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_23_cast_fp16 = reshape(shape = var_1738, x = mh_k_43_cast_fp16)[name = string("current_key_23_cast_fp16")]; + tensor var_1744_to_fp16 = const()[name = string("op_1744_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274688)))]; + tensor var_1745_cast_fp16 = mul(x = obj_47_cast_fp16, y = var_1744_to_fp16)[name = string("op_1745_cast_fp16")]; + tensor var_1742_to_fp16 = const()[name = string("op_1742_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274816)))]; + tensor var_1746_cast_fp16 = mul(x = current_key_23_cast_fp16, y = var_1742_to_fp16)[name = string("op_1746_cast_fp16")]; + tensor key_23_cast_fp16 = add(x = var_1745_cast_fp16, y = var_1746_cast_fp16)[name = string("key_23_cast_fp16")]; + tensor var_1748_to_fp16 = const()[name = string("op_1748_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274688)))]; + tensor var_1749_cast_fp16 = mul(x = obj_49_cast_fp16, y = var_1748_to_fp16)[name = string("op_1749_cast_fp16")]; + tensor var_1750_cast_fp16 = mul(x = current_value_11_cast_fp16, y = var_1742_to_fp16)[name = string("op_1750_cast_fp16")]; + tensor value_11_cast_fp16 = add(x = var_1749_cast_fp16, y = var_1750_cast_fp16)[name = string("value_11_cast_fp16")]; + fp16 var_1757_to_fp16 = const()[name = string("op_1757_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_47_cast_fp16 = mul(x = mh_q_43_cast_fp16, y = var_1757_to_fp16)[name = string("mh_q_47_cast_fp16")]; + tensor var_1759 = const()[name = string("op_1759"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_45_cast_fp16 = reshape(shape = var_1759, x = key_23_cast_fp16)[name = string("mh_k_45_cast_fp16")]; + tensor var_1761 = const()[name = string("op_1761"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_21_cast_fp16 = reshape(shape = var_1761, x = value_11_cast_fp16)[name = string("mh_v_21_cast_fp16")]; + tensor transpose_20_perm_0 = const()[name = string("transpose_20_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_10_reps_0 = const()[name = string("tile_10_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_20_cast_fp16 = transpose(perm = transpose_20_perm_0, x = mh_k_45_cast_fp16)[name = string("transpose_449")]; + tensor tile_10_cast_fp16 = tile(reps = tile_10_reps_0, x = transpose_20_cast_fp16)[name = string("tile_10_cast_fp16")]; + tensor concat_28 = const()[name = string("concat_28"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_20_cast_fp16 = reshape(shape = concat_28, x = tile_10_cast_fp16)[name = string("reshape_20_cast_fp16")]; + tensor transpose_21_perm_0 = const()[name = string("transpose_21_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_29 = const()[name = string("concat_29"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_21_cast_fp16 = transpose(perm = transpose_21_perm_0, x = reshape_20_cast_fp16)[name = string("transpose_448")]; + tensor reshape_21_cast_fp16 = reshape(shape = concat_29, x = transpose_21_cast_fp16)[name = string("reshape_21_cast_fp16")]; + tensor transpose_22_perm_0 = const()[name = string("transpose_22_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_11_reps_0 = const()[name = string("tile_11_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_22_cast_fp16 = transpose(perm = transpose_22_perm_0, x = mh_v_21_cast_fp16)[name = string("transpose_447")]; + tensor tile_11_cast_fp16 = tile(reps = tile_11_reps_0, x = transpose_22_cast_fp16)[name = string("tile_11_cast_fp16")]; + tensor concat_30 = const()[name = string("concat_30"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_22_cast_fp16 = reshape(shape = concat_30, x = tile_11_cast_fp16)[name = string("reshape_22_cast_fp16")]; + tensor transpose_23_perm_0 = const()[name = string("transpose_23_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_31 = const()[name = string("concat_31"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_23_cast_fp16 = transpose(perm = transpose_23_perm_0, x = reshape_22_cast_fp16)[name = string("transpose_446")]; + tensor reshape_23_cast_fp16 = reshape(shape = concat_31, x = transpose_23_cast_fp16)[name = string("reshape_23_cast_fp16")]; + tensor transpose_337_perm_0 = const()[name = string("transpose_337_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_31_transpose_x_1 = const()[name = string("mh_w_31_transpose_x_1"), val = bool(true)]; + bool mh_w_31_transpose_y_1 = const()[name = string("mh_w_31_transpose_y_1"), val = bool(false)]; + tensor transpose_337_cast_fp16 = transpose(perm = transpose_337_perm_0, x = reshape_21_cast_fp16)[name = string("transpose_445")]; + tensor mh_w_31_cast_fp16 = matmul(transpose_x = mh_w_31_transpose_x_1, transpose_y = mh_w_31_transpose_y_1, x = mh_q_47_cast_fp16, y = transpose_337_cast_fp16)[name = string("mh_w_31_cast_fp16")]; + tensor var_1769_to_fp16 = const()[name = string("op_1769_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274944)))]; + tensor mh_w_33_cast_fp16 = add(x = mh_w_31_cast_fp16, y = var_1769_to_fp16)[name = string("mh_w_33_cast_fp16")]; + tensor mh_w_35_cast_fp16 = softmax(axis = var_1589, x = mh_w_33_cast_fp16)[name = string("mh_w_35_cast_fp16")]; + tensor transpose_338_perm_0 = const()[name = string("transpose_338_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_11_transpose_x_1 = const()[name = string("attn_11_transpose_x_1"), val = bool(false)]; + bool attn_11_transpose_y_1 = const()[name = string("attn_11_transpose_y_1"), val = bool(true)]; + tensor transpose_338_cast_fp16 = transpose(perm = transpose_338_perm_0, x = reshape_23_cast_fp16)[name = string("transpose_444")]; + tensor attn_11_cast_fp16 = matmul(transpose_x = attn_11_transpose_x_1, transpose_y = attn_11_transpose_y_1, x = transpose_338_cast_fp16, y = mh_w_35_cast_fp16)[name = string("attn_11_cast_fp16")]; + tensor var_1775 = const()[name = string("op_1775"), val = tensor([1, 2048, 1, 1])]; + tensor input_41_cast_fp16 = reshape(shape = var_1775, x = attn_11_cast_fp16)[name = string("input_41_cast_fp16")]; + string obj_55_pad_type_0 = const()[name = string("obj_55_pad_type_0"), val = string("valid")]; + tensor obj_55_strides_0 = const()[name = string("obj_55_strides_0"), val = tensor([1, 1])]; + tensor obj_55_pad_0 = const()[name = string("obj_55_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_55_dilations_0 = const()[name = string("obj_55_dilations_0"), val = tensor([1, 1])]; + int32 obj_55_groups_0 = const()[name = string("obj_55_groups_0"), val = int32(1)]; + tensor obj_55_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_55_dilations_0, groups = obj_55_groups_0, pad = obj_55_pad_0, pad_type = obj_55_pad_type_0, strides = obj_55_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_41_cast_fp16)[name = string("obj_55_cast_fp16")]; + tensor inputs_47_cast_fp16 = add(x = inputs_41_cast_fp16, y = obj_55_cast_fp16)[name = string("inputs_47_cast_fp16")]; + tensor inputs_sq_47_cast_fp16 = mul(x = inputs_47_cast_fp16, y = inputs_47_cast_fp16)[name = string("inputs_sq_47_cast_fp16")]; + tensor variance_47_axes_0 = const()[name = string("variance_47_axes_0"), val = tensor([1])]; + bool variance_47_keep_dims_0 = const()[name = string("variance_47_keep_dims_0"), val = bool(true)]; + tensor variance_47_cast_fp16 = reduce_mean(axes = variance_47_axes_0, keep_dims = variance_47_keep_dims_0, x = inputs_sq_47_cast_fp16)[name = string("variance_47_cast_fp16")]; + fp16 var_1793_to_fp16 = const()[name = string("op_1793_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1794_cast_fp16 = add(x = variance_47_cast_fp16, y = var_1793_to_fp16)[name = string("op_1794_cast_fp16")]; + fp32 var_1795_epsilon_0 = const()[name = string("op_1795_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1795_cast_fp16 = rsqrt(epsilon = var_1795_epsilon_0, x = var_1794_cast_fp16)[name = string("op_1795_cast_fp16")]; + tensor hidden_states_57_cast_fp16 = mul(x = inputs_47_cast_fp16, y = var_1795_cast_fp16)[name = string("hidden_states_57_cast_fp16")]; + tensor input_43_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_57_cast_fp16)[name = string("input_43_cast_fp16")]; + string input_45_pad_type_0 = const()[name = string("input_45_pad_type_0"), val = string("valid")]; + tensor input_45_strides_0 = const()[name = string("input_45_strides_0"), val = tensor([1, 1])]; + tensor input_45_pad_0 = const()[name = string("input_45_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_45_dilations_0 = const()[name = string("input_45_dilations_0"), val = tensor([1, 1])]; + int32 input_45_groups_0 = const()[name = string("input_45_groups_0"), val = int32(1)]; + tensor input_45_cast_fp16 = conv(dilations = input_45_dilations_0, groups = input_45_groups_0, pad = input_45_pad_0, pad_type = input_45_pad_type_0, strides = input_45_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_43_cast_fp16)[name = string("input_45_cast_fp16")]; + tensor var_1809_cast_fp16 = silu(x = input_45_cast_fp16)[name = string("op_1809_cast_fp16")]; + string var_1815_pad_type_0 = const()[name = string("op_1815_pad_type_0"), val = string("valid")]; + tensor var_1815_strides_0 = const()[name = string("op_1815_strides_0"), val = tensor([1, 1])]; + tensor var_1815_pad_0 = const()[name = string("op_1815_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1815_dilations_0 = const()[name = string("op_1815_dilations_0"), val = tensor([1, 1])]; + int32 var_1815_groups_0 = const()[name = string("op_1815_groups_0"), val = int32(1)]; + tensor var_1815_cast_fp16 = conv(dilations = var_1815_dilations_0, groups = var_1815_groups_0, pad = var_1815_pad_0, pad_type = var_1815_pad_type_0, strides = var_1815_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_43_cast_fp16)[name = string("op_1815_cast_fp16")]; + tensor input_47_cast_fp16 = mul(x = var_1809_cast_fp16, y = var_1815_cast_fp16)[name = string("input_47_cast_fp16")]; + string hidden_states_59_pad_type_0 = const()[name = string("hidden_states_59_pad_type_0"), val = string("valid")]; + tensor hidden_states_59_strides_0 = const()[name = string("hidden_states_59_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_59_pad_0 = const()[name = string("hidden_states_59_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_59_dilations_0 = const()[name = string("hidden_states_59_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_59_groups_0 = const()[name = string("hidden_states_59_groups_0"), val = int32(1)]; + tensor hidden_states_59_cast_fp16 = conv(dilations = hidden_states_59_dilations_0, groups = hidden_states_59_groups_0, pad = hidden_states_59_pad_0, pad_type = hidden_states_59_pad_type_0, strides = hidden_states_59_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_47_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; + tensor inputs_49_cast_fp16 = add(x = inputs_47_cast_fp16, y = hidden_states_59_cast_fp16)[name = string("inputs_49_cast_fp16")]; + tensor obj_59_begin_0 = const()[name = string("obj_59_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_59_end_0 = const()[name = string("obj_59_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_59_end_mask_0 = const()[name = string("obj_59_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_59_cast_fp16 = slice_by_index(begin = obj_59_begin_0, end = obj_59_end_0, end_mask = obj_59_end_mask_0, x = key_caches_3_cast_fp16)[name = string("obj_59_cast_fp16")]; + tensor obj_61_begin_0 = const()[name = string("obj_61_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_61_end_0 = const()[name = string("obj_61_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_61_end_mask_0 = const()[name = string("obj_61_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_61_cast_fp16 = slice_by_index(begin = obj_61_begin_0, end = obj_61_end_0, end_mask = obj_61_end_mask_0, x = value_caches_3_cast_fp16)[name = string("obj_61_cast_fp16")]; + int32 var_1863 = const()[name = string("op_1863"), val = int32(3)]; + int32 var_1873 = const()[name = string("op_1873"), val = int32(-2)]; + tensor inputs_sq_49_cast_fp16 = mul(x = inputs_49_cast_fp16, y = inputs_49_cast_fp16)[name = string("inputs_sq_49_cast_fp16")]; + tensor variance_49_axes_0 = const()[name = string("variance_49_axes_0"), val = tensor([1])]; + bool variance_49_keep_dims_0 = const()[name = string("variance_49_keep_dims_0"), val = bool(true)]; + tensor variance_49_cast_fp16 = reduce_mean(axes = variance_49_axes_0, keep_dims = variance_49_keep_dims_0, x = inputs_sq_49_cast_fp16)[name = string("variance_49_cast_fp16")]; + fp16 var_1887_to_fp16 = const()[name = string("op_1887_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1888_cast_fp16 = add(x = variance_49_cast_fp16, y = var_1887_to_fp16)[name = string("op_1888_cast_fp16")]; + fp32 var_1889_epsilon_0 = const()[name = string("op_1889_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1889_cast_fp16 = rsqrt(epsilon = var_1889_epsilon_0, x = var_1888_cast_fp16)[name = string("op_1889_cast_fp16")]; + tensor hidden_states_61_cast_fp16 = mul(x = inputs_49_cast_fp16, y = var_1889_cast_fp16)[name = string("hidden_states_61_cast_fp16")]; + tensor obj_57_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_61_cast_fp16)[name = string("obj_57_cast_fp16")]; + string query_37_pad_type_0 = const()[name = string("query_37_pad_type_0"), val = string("valid")]; + tensor query_37_strides_0 = const()[name = string("query_37_strides_0"), val = tensor([1, 1])]; + tensor query_37_pad_0 = const()[name = string("query_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_37_dilations_0 = const()[name = string("query_37_dilations_0"), val = tensor([1, 1])]; + int32 query_37_groups_0 = const()[name = string("query_37_groups_0"), val = int32(1)]; + tensor query_37_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_37_dilations_0, groups = query_37_groups_0, pad = query_37_pad_0, pad_type = query_37_pad_type_0, strides = query_37_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_57_cast_fp16)[name = string("query_37_cast_fp16")]; + string current_key_25_pad_type_0 = const()[name = string("current_key_25_pad_type_0"), val = string("valid")]; + tensor current_key_25_strides_0 = const()[name = string("current_key_25_strides_0"), val = tensor([1, 1])]; + tensor current_key_25_pad_0 = const()[name = string("current_key_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_25_dilations_0 = const()[name = string("current_key_25_dilations_0"), val = tensor([1, 1])]; + int32 current_key_25_groups_0 = const()[name = string("current_key_25_groups_0"), val = int32(1)]; + tensor current_key_25_cast_fp16 = conv(dilations = current_key_25_dilations_0, groups = current_key_25_groups_0, pad = current_key_25_pad_0, pad_type = current_key_25_pad_type_0, strides = current_key_25_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_57_cast_fp16)[name = string("current_key_25_cast_fp16")]; + string current_value_13_pad_type_0 = const()[name = string("current_value_13_pad_type_0"), val = string("valid")]; + tensor current_value_13_strides_0 = const()[name = string("current_value_13_strides_0"), val = tensor([1, 1])]; + tensor current_value_13_pad_0 = const()[name = string("current_value_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_13_dilations_0 = const()[name = string("current_value_13_dilations_0"), val = tensor([1, 1])]; + int32 current_value_13_groups_0 = const()[name = string("current_value_13_groups_0"), val = int32(1)]; + tensor current_value_13_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_13_dilations_0, groups = current_value_13_groups_0, pad = current_value_13_pad_0, pad_type = current_value_13_pad_type_0, strides = current_value_13_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_57_cast_fp16)[name = string("current_value_13_cast_fp16")]; + tensor var_1926 = const()[name = string("op_1926"), val = tensor([16, 128, 1, 1])]; + tensor inputs_51_cast_fp16 = reshape(shape = var_1926, x = query_37_cast_fp16)[name = string("inputs_51_cast_fp16")]; + tensor inputs_sq_51_cast_fp16 = mul(x = inputs_51_cast_fp16, y = inputs_51_cast_fp16)[name = string("inputs_sq_51_cast_fp16")]; + tensor variance_51_axes_0 = const()[name = string("variance_51_axes_0"), val = tensor([1])]; + bool variance_51_keep_dims_0 = const()[name = string("variance_51_keep_dims_0"), val = bool(true)]; + tensor variance_51_cast_fp16 = reduce_mean(axes = variance_51_axes_0, keep_dims = variance_51_keep_dims_0, x = inputs_sq_51_cast_fp16)[name = string("variance_51_cast_fp16")]; + fp16 var_1932_to_fp16 = const()[name = string("op_1932_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1933_cast_fp16 = add(x = variance_51_cast_fp16, y = var_1932_to_fp16)[name = string("op_1933_cast_fp16")]; + fp32 var_1934_epsilon_0 = const()[name = string("op_1934_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1934_cast_fp16 = rsqrt(epsilon = var_1934_epsilon_0, x = var_1933_cast_fp16)[name = string("op_1934_cast_fp16")]; + tensor hidden_states_63_cast_fp16 = mul(x = inputs_51_cast_fp16, y = var_1934_cast_fp16)[name = string("hidden_states_63_cast_fp16")]; + tensor query_normed_13_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_63_cast_fp16)[name = string("query_normed_13_cast_fp16")]; + tensor var_1942 = const()[name = string("op_1942"), val = tensor([8, 128, 1, 1])]; + tensor inputs_53_cast_fp16 = reshape(shape = var_1942, x = current_key_25_cast_fp16)[name = string("inputs_53_cast_fp16")]; + tensor inputs_sq_53_cast_fp16 = mul(x = inputs_53_cast_fp16, y = inputs_53_cast_fp16)[name = string("inputs_sq_53_cast_fp16")]; + tensor variance_53_axes_0 = const()[name = string("variance_53_axes_0"), val = tensor([1])]; + bool variance_53_keep_dims_0 = const()[name = string("variance_53_keep_dims_0"), val = bool(true)]; + tensor variance_53_cast_fp16 = reduce_mean(axes = variance_53_axes_0, keep_dims = variance_53_keep_dims_0, x = inputs_sq_53_cast_fp16)[name = string("variance_53_cast_fp16")]; + fp16 var_1948_to_fp16 = const()[name = string("op_1948_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1949_cast_fp16 = add(x = variance_53_cast_fp16, y = var_1948_to_fp16)[name = string("op_1949_cast_fp16")]; + fp32 var_1950_epsilon_0 = const()[name = string("op_1950_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1950_cast_fp16 = rsqrt(epsilon = var_1950_epsilon_0, x = var_1949_cast_fp16)[name = string("op_1950_cast_fp16")]; + tensor hidden_states_65_cast_fp16 = mul(x = inputs_53_cast_fp16, y = var_1950_cast_fp16)[name = string("hidden_states_65_cast_fp16")]; + tensor current_key_normed_13_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_65_cast_fp16)[name = string("current_key_normed_13_cast_fp16")]; + tensor var_1968 = const()[name = string("op_1968"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_49_cast_fp16 = reshape(shape = var_1968, x = query_normed_13_cast_fp16)[name = string("mh_q_49_cast_fp16")]; + tensor var_1970 = const()[name = string("op_1970"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_49_cast_fp16 = reshape(shape = var_1970, x = current_key_normed_13_cast_fp16)[name = string("mh_k_49_cast_fp16")]; + tensor var_1974_cast_fp16 = mul(x = mh_q_49_cast_fp16, y = cos_11_to_fp16)[name = string("op_1974_cast_fp16")]; + tensor var_1979_begin_0 = const()[name = string("op_1979_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1979_end_0 = const()[name = string("op_1979_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_1979_end_mask_0 = const()[name = string("op_1979_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1979_cast_fp16 = slice_by_index(begin = var_1979_begin_0, end = var_1979_end_0, end_mask = var_1979_end_mask_0, x = mh_q_49_cast_fp16)[name = string("op_1979_cast_fp16")]; + tensor var_1985_begin_0 = const()[name = string("op_1985_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1985_end_0 = const()[name = string("op_1985_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_1985_end_mask_0 = const()[name = string("op_1985_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1985_cast_fp16 = slice_by_index(begin = var_1985_begin_0, end = var_1985_end_0, end_mask = var_1985_end_mask_0, x = mh_q_49_cast_fp16)[name = string("op_1985_cast_fp16")]; + fp16 const_135_promoted_to_fp16 = const()[name = string("const_135_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1987_cast_fp16 = mul(x = var_1985_cast_fp16, y = const_135_promoted_to_fp16)[name = string("op_1987_cast_fp16")]; + bool var_1989_interleave_0 = const()[name = string("op_1989_interleave_0"), val = bool(false)]; + tensor var_1989_cast_fp16 = concat(axis = var_1873, interleave = var_1989_interleave_0, values = (var_1987_cast_fp16, var_1979_cast_fp16))[name = string("op_1989_cast_fp16")]; + tensor var_1990_cast_fp16 = mul(x = var_1989_cast_fp16, y = sin_11_to_fp16)[name = string("op_1990_cast_fp16")]; + tensor mh_q_51_cast_fp16 = add(x = var_1974_cast_fp16, y = var_1990_cast_fp16)[name = string("mh_q_51_cast_fp16")]; + tensor var_1992_cast_fp16 = mul(x = mh_k_49_cast_fp16, y = cos_11_to_fp16)[name = string("op_1992_cast_fp16")]; + tensor var_1997_begin_0 = const()[name = string("op_1997_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1997_end_0 = const()[name = string("op_1997_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_1997_end_mask_0 = const()[name = string("op_1997_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1997_cast_fp16 = slice_by_index(begin = var_1997_begin_0, end = var_1997_end_0, end_mask = var_1997_end_mask_0, x = mh_k_49_cast_fp16)[name = string("op_1997_cast_fp16")]; + tensor var_2003_begin_0 = const()[name = string("op_2003_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_2003_end_0 = const()[name = string("op_2003_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_2003_end_mask_0 = const()[name = string("op_2003_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_2003_cast_fp16 = slice_by_index(begin = var_2003_begin_0, end = var_2003_end_0, end_mask = var_2003_end_mask_0, x = mh_k_49_cast_fp16)[name = string("op_2003_cast_fp16")]; + fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2005_cast_fp16 = mul(x = var_2003_cast_fp16, y = const_138_promoted_to_fp16)[name = string("op_2005_cast_fp16")]; + bool var_2007_interleave_0 = const()[name = string("op_2007_interleave_0"), val = bool(false)]; + tensor var_2007_cast_fp16 = concat(axis = var_1873, interleave = var_2007_interleave_0, values = (var_2005_cast_fp16, var_1997_cast_fp16))[name = string("op_2007_cast_fp16")]; + tensor var_2008_cast_fp16 = mul(x = var_2007_cast_fp16, y = sin_11_to_fp16)[name = string("op_2008_cast_fp16")]; + tensor mh_k_51_cast_fp16 = add(x = var_1992_cast_fp16, y = var_2008_cast_fp16)[name = string("mh_k_51_cast_fp16")]; + tensor var_2012 = const()[name = string("op_2012"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_27_cast_fp16 = reshape(shape = var_2012, x = mh_k_51_cast_fp16)[name = string("current_key_27_cast_fp16")]; + tensor var_2018_to_fp16 = const()[name = string("op_2018_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274688)))]; + tensor var_2019_cast_fp16 = mul(x = obj_59_cast_fp16, y = var_2018_to_fp16)[name = string("op_2019_cast_fp16")]; + tensor var_2016_to_fp16 = const()[name = string("op_2016_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274816)))]; + tensor var_2020_cast_fp16 = mul(x = current_key_27_cast_fp16, y = var_2016_to_fp16)[name = string("op_2020_cast_fp16")]; + tensor key_27_cast_fp16 = add(x = var_2019_cast_fp16, y = var_2020_cast_fp16)[name = string("key_27_cast_fp16")]; + tensor var_2022_to_fp16 = const()[name = string("op_2022_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274688)))]; + tensor var_2023_cast_fp16 = mul(x = obj_61_cast_fp16, y = var_2022_to_fp16)[name = string("op_2023_cast_fp16")]; + tensor var_2024_cast_fp16 = mul(x = current_value_13_cast_fp16, y = var_2016_to_fp16)[name = string("op_2024_cast_fp16")]; + tensor value_13_cast_fp16 = add(x = var_2023_cast_fp16, y = var_2024_cast_fp16)[name = string("value_13_cast_fp16")]; + fp16 var_2031_to_fp16 = const()[name = string("op_2031_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_55_cast_fp16 = mul(x = mh_q_51_cast_fp16, y = var_2031_to_fp16)[name = string("mh_q_55_cast_fp16")]; + tensor var_2033 = const()[name = string("op_2033"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_53_cast_fp16 = reshape(shape = var_2033, x = key_27_cast_fp16)[name = string("mh_k_53_cast_fp16")]; + tensor var_2035 = const()[name = string("op_2035"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_25_cast_fp16 = reshape(shape = var_2035, x = value_13_cast_fp16)[name = string("mh_v_25_cast_fp16")]; + tensor transpose_24_perm_0 = const()[name = string("transpose_24_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_12_reps_0 = const()[name = string("tile_12_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_24_cast_fp16 = transpose(perm = transpose_24_perm_0, x = mh_k_53_cast_fp16)[name = string("transpose_443")]; + tensor tile_12_cast_fp16 = tile(reps = tile_12_reps_0, x = transpose_24_cast_fp16)[name = string("tile_12_cast_fp16")]; + tensor concat_32 = const()[name = string("concat_32"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_24_cast_fp16 = reshape(shape = concat_32, x = tile_12_cast_fp16)[name = string("reshape_24_cast_fp16")]; + tensor transpose_25_perm_0 = const()[name = string("transpose_25_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_33 = const()[name = string("concat_33"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_25_cast_fp16 = transpose(perm = transpose_25_perm_0, x = reshape_24_cast_fp16)[name = string("transpose_442")]; + tensor reshape_25_cast_fp16 = reshape(shape = concat_33, x = transpose_25_cast_fp16)[name = string("reshape_25_cast_fp16")]; + tensor transpose_26_perm_0 = const()[name = string("transpose_26_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_13_reps_0 = const()[name = string("tile_13_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_26_cast_fp16 = transpose(perm = transpose_26_perm_0, x = mh_v_25_cast_fp16)[name = string("transpose_441")]; + tensor tile_13_cast_fp16 = tile(reps = tile_13_reps_0, x = transpose_26_cast_fp16)[name = string("tile_13_cast_fp16")]; + tensor concat_34 = const()[name = string("concat_34"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_26_cast_fp16 = reshape(shape = concat_34, x = tile_13_cast_fp16)[name = string("reshape_26_cast_fp16")]; + tensor transpose_27_perm_0 = const()[name = string("transpose_27_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_35 = const()[name = string("concat_35"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_27_cast_fp16 = transpose(perm = transpose_27_perm_0, x = reshape_26_cast_fp16)[name = string("transpose_440")]; + tensor reshape_27_cast_fp16 = reshape(shape = concat_35, x = transpose_27_cast_fp16)[name = string("reshape_27_cast_fp16")]; + tensor transpose_341_perm_0 = const()[name = string("transpose_341_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_37_transpose_x_1 = const()[name = string("mh_w_37_transpose_x_1"), val = bool(true)]; + bool mh_w_37_transpose_y_1 = const()[name = string("mh_w_37_transpose_y_1"), val = bool(false)]; + tensor transpose_341_cast_fp16 = transpose(perm = transpose_341_perm_0, x = reshape_25_cast_fp16)[name = string("transpose_439")]; + tensor mh_w_37_cast_fp16 = matmul(transpose_x = mh_w_37_transpose_x_1, transpose_y = mh_w_37_transpose_y_1, x = mh_q_55_cast_fp16, y = transpose_341_cast_fp16)[name = string("mh_w_37_cast_fp16")]; + tensor var_2043_to_fp16 = const()[name = string("op_2043_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274944)))]; + tensor mh_w_39_cast_fp16 = add(x = mh_w_37_cast_fp16, y = var_2043_to_fp16)[name = string("mh_w_39_cast_fp16")]; + tensor mh_w_41_cast_fp16 = softmax(axis = var_1863, x = mh_w_39_cast_fp16)[name = string("mh_w_41_cast_fp16")]; + tensor transpose_342_perm_0 = const()[name = string("transpose_342_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_13_transpose_x_1 = const()[name = string("attn_13_transpose_x_1"), val = bool(false)]; + bool attn_13_transpose_y_1 = const()[name = string("attn_13_transpose_y_1"), val = bool(true)]; + tensor transpose_342_cast_fp16 = transpose(perm = transpose_342_perm_0, x = reshape_27_cast_fp16)[name = string("transpose_438")]; + tensor attn_13_cast_fp16 = matmul(transpose_x = attn_13_transpose_x_1, transpose_y = attn_13_transpose_y_1, x = transpose_342_cast_fp16, y = mh_w_41_cast_fp16)[name = string("attn_13_cast_fp16")]; + tensor var_2049 = const()[name = string("op_2049"), val = tensor([1, 2048, 1, 1])]; + tensor input_49_cast_fp16 = reshape(shape = var_2049, x = attn_13_cast_fp16)[name = string("input_49_cast_fp16")]; + string obj_63_pad_type_0 = const()[name = string("obj_63_pad_type_0"), val = string("valid")]; + tensor obj_63_strides_0 = const()[name = string("obj_63_strides_0"), val = tensor([1, 1])]; + tensor obj_63_pad_0 = const()[name = string("obj_63_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_63_dilations_0 = const()[name = string("obj_63_dilations_0"), val = tensor([1, 1])]; + int32 obj_63_groups_0 = const()[name = string("obj_63_groups_0"), val = int32(1)]; + tensor obj_63_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_63_dilations_0, groups = obj_63_groups_0, pad = obj_63_pad_0, pad_type = obj_63_pad_type_0, strides = obj_63_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_49_cast_fp16)[name = string("obj_63_cast_fp16")]; + tensor inputs_55_cast_fp16 = add(x = inputs_49_cast_fp16, y = obj_63_cast_fp16)[name = string("inputs_55_cast_fp16")]; + tensor inputs_sq_55_cast_fp16 = mul(x = inputs_55_cast_fp16, y = inputs_55_cast_fp16)[name = string("inputs_sq_55_cast_fp16")]; + tensor variance_55_axes_0 = const()[name = string("variance_55_axes_0"), val = tensor([1])]; + bool variance_55_keep_dims_0 = const()[name = string("variance_55_keep_dims_0"), val = bool(true)]; + tensor variance_55_cast_fp16 = reduce_mean(axes = variance_55_axes_0, keep_dims = variance_55_keep_dims_0, x = inputs_sq_55_cast_fp16)[name = string("variance_55_cast_fp16")]; + fp16 var_2067_to_fp16 = const()[name = string("op_2067_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2068_cast_fp16 = add(x = variance_55_cast_fp16, y = var_2067_to_fp16)[name = string("op_2068_cast_fp16")]; + fp32 var_2069_epsilon_0 = const()[name = string("op_2069_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2069_cast_fp16 = rsqrt(epsilon = var_2069_epsilon_0, x = var_2068_cast_fp16)[name = string("op_2069_cast_fp16")]; + tensor hidden_states_67_cast_fp16 = mul(x = inputs_55_cast_fp16, y = var_2069_cast_fp16)[name = string("hidden_states_67_cast_fp16")]; + tensor input_51_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_67_cast_fp16)[name = string("input_51_cast_fp16")]; + string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; + tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; + tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; + int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; + tensor input_53_cast_fp16 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_51_cast_fp16)[name = string("input_53_cast_fp16")]; + tensor var_2083_cast_fp16 = silu(x = input_53_cast_fp16)[name = string("op_2083_cast_fp16")]; + string var_2089_pad_type_0 = const()[name = string("op_2089_pad_type_0"), val = string("valid")]; + tensor var_2089_strides_0 = const()[name = string("op_2089_strides_0"), val = tensor([1, 1])]; + tensor var_2089_pad_0 = const()[name = string("op_2089_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2089_dilations_0 = const()[name = string("op_2089_dilations_0"), val = tensor([1, 1])]; + int32 var_2089_groups_0 = const()[name = string("op_2089_groups_0"), val = int32(1)]; + tensor var_2089_cast_fp16 = conv(dilations = var_2089_dilations_0, groups = var_2089_groups_0, pad = var_2089_pad_0, pad_type = var_2089_pad_type_0, strides = var_2089_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_51_cast_fp16)[name = string("op_2089_cast_fp16")]; + tensor input_55_cast_fp16 = mul(x = var_2083_cast_fp16, y = var_2089_cast_fp16)[name = string("input_55_cast_fp16")]; + string hidden_states_69_pad_type_0 = const()[name = string("hidden_states_69_pad_type_0"), val = string("valid")]; + tensor hidden_states_69_strides_0 = const()[name = string("hidden_states_69_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_69_pad_0 = const()[name = string("hidden_states_69_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_69_dilations_0 = const()[name = string("hidden_states_69_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_69_groups_0 = const()[name = string("hidden_states_69_groups_0"), val = int32(1)]; + tensor hidden_states_69_cast_fp16 = conv(dilations = hidden_states_69_dilations_0, groups = hidden_states_69_groups_0, pad = hidden_states_69_pad_0, pad_type = hidden_states_69_pad_type_0, strides = hidden_states_69_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_55_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; + tensor inputs_57_cast_fp16 = add(x = inputs_55_cast_fp16, y = hidden_states_69_cast_fp16)[name = string("inputs_57_cast_fp16")]; + tensor obj_67_begin_0 = const()[name = string("obj_67_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_67_end_0 = const()[name = string("obj_67_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_67_end_mask_0 = const()[name = string("obj_67_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_67_cast_fp16 = slice_by_index(begin = obj_67_begin_0, end = obj_67_end_0, end_mask = obj_67_end_mask_0, x = key_caches_3_cast_fp16)[name = string("obj_67_cast_fp16")]; + tensor obj_69_begin_0 = const()[name = string("obj_69_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_69_end_0 = const()[name = string("obj_69_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_69_end_mask_0 = const()[name = string("obj_69_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_69_cast_fp16 = slice_by_index(begin = obj_69_begin_0, end = obj_69_end_0, end_mask = obj_69_end_mask_0, x = value_caches_3_cast_fp16)[name = string("obj_69_cast_fp16")]; + int32 var_2137 = const()[name = string("op_2137"), val = int32(3)]; + int32 var_2147 = const()[name = string("op_2147"), val = int32(-2)]; + tensor inputs_sq_57_cast_fp16 = mul(x = inputs_57_cast_fp16, y = inputs_57_cast_fp16)[name = string("inputs_sq_57_cast_fp16")]; + tensor variance_57_axes_0 = const()[name = string("variance_57_axes_0"), val = tensor([1])]; + bool variance_57_keep_dims_0 = const()[name = string("variance_57_keep_dims_0"), val = bool(true)]; + tensor variance_57_cast_fp16 = reduce_mean(axes = variance_57_axes_0, keep_dims = variance_57_keep_dims_0, x = inputs_sq_57_cast_fp16)[name = string("variance_57_cast_fp16")]; + fp16 var_2161_to_fp16 = const()[name = string("op_2161_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2162_cast_fp16 = add(x = variance_57_cast_fp16, y = var_2161_to_fp16)[name = string("op_2162_cast_fp16")]; + fp32 var_2163_epsilon_0 = const()[name = string("op_2163_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2163_cast_fp16 = rsqrt(epsilon = var_2163_epsilon_0, x = var_2162_cast_fp16)[name = string("op_2163_cast_fp16")]; + tensor hidden_states_71_cast_fp16 = mul(x = inputs_57_cast_fp16, y = var_2163_cast_fp16)[name = string("hidden_states_71_cast_fp16")]; + tensor obj_65_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_71_cast_fp16)[name = string("obj_65_cast_fp16")]; + string query_43_pad_type_0 = const()[name = string("query_43_pad_type_0"), val = string("valid")]; + tensor query_43_strides_0 = const()[name = string("query_43_strides_0"), val = tensor([1, 1])]; + tensor query_43_pad_0 = const()[name = string("query_43_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_43_dilations_0 = const()[name = string("query_43_dilations_0"), val = tensor([1, 1])]; + int32 query_43_groups_0 = const()[name = string("query_43_groups_0"), val = int32(1)]; + tensor query_43_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_43_dilations_0, groups = query_43_groups_0, pad = query_43_pad_0, pad_type = query_43_pad_type_0, strides = query_43_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_65_cast_fp16)[name = string("query_43_cast_fp16")]; + string current_key_29_pad_type_0 = const()[name = string("current_key_29_pad_type_0"), val = string("valid")]; + tensor current_key_29_strides_0 = const()[name = string("current_key_29_strides_0"), val = tensor([1, 1])]; + tensor current_key_29_pad_0 = const()[name = string("current_key_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_29_dilations_0 = const()[name = string("current_key_29_dilations_0"), val = tensor([1, 1])]; + int32 current_key_29_groups_0 = const()[name = string("current_key_29_groups_0"), val = int32(1)]; + tensor current_key_29_cast_fp16 = conv(dilations = current_key_29_dilations_0, groups = current_key_29_groups_0, pad = current_key_29_pad_0, pad_type = current_key_29_pad_type_0, strides = current_key_29_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_65_cast_fp16)[name = string("current_key_29_cast_fp16")]; + string current_value_15_pad_type_0 = const()[name = string("current_value_15_pad_type_0"), val = string("valid")]; + tensor current_value_15_strides_0 = const()[name = string("current_value_15_strides_0"), val = tensor([1, 1])]; + tensor current_value_15_pad_0 = const()[name = string("current_value_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_15_dilations_0 = const()[name = string("current_value_15_dilations_0"), val = tensor([1, 1])]; + int32 current_value_15_groups_0 = const()[name = string("current_value_15_groups_0"), val = int32(1)]; + tensor current_value_15_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_15_dilations_0, groups = current_value_15_groups_0, pad = current_value_15_pad_0, pad_type = current_value_15_pad_type_0, strides = current_value_15_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_65_cast_fp16)[name = string("current_value_15_cast_fp16")]; + tensor var_2200 = const()[name = string("op_2200"), val = tensor([16, 128, 1, 1])]; + tensor inputs_59_cast_fp16 = reshape(shape = var_2200, x = query_43_cast_fp16)[name = string("inputs_59_cast_fp16")]; + tensor inputs_sq_59_cast_fp16 = mul(x = inputs_59_cast_fp16, y = inputs_59_cast_fp16)[name = string("inputs_sq_59_cast_fp16")]; + tensor variance_59_axes_0 = const()[name = string("variance_59_axes_0"), val = tensor([1])]; + bool variance_59_keep_dims_0 = const()[name = string("variance_59_keep_dims_0"), val = bool(true)]; + tensor variance_59_cast_fp16 = reduce_mean(axes = variance_59_axes_0, keep_dims = variance_59_keep_dims_0, x = inputs_sq_59_cast_fp16)[name = string("variance_59_cast_fp16")]; + fp16 var_2206_to_fp16 = const()[name = string("op_2206_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2207_cast_fp16 = add(x = variance_59_cast_fp16, y = var_2206_to_fp16)[name = string("op_2207_cast_fp16")]; + fp32 var_2208_epsilon_0 = const()[name = string("op_2208_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2208_cast_fp16 = rsqrt(epsilon = var_2208_epsilon_0, x = var_2207_cast_fp16)[name = string("op_2208_cast_fp16")]; + tensor hidden_states_73_cast_fp16 = mul(x = inputs_59_cast_fp16, y = var_2208_cast_fp16)[name = string("hidden_states_73_cast_fp16")]; + tensor query_normed_15_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_73_cast_fp16)[name = string("query_normed_15_cast_fp16")]; + tensor var_2216 = const()[name = string("op_2216"), val = tensor([8, 128, 1, 1])]; + tensor inputs_61_cast_fp16 = reshape(shape = var_2216, x = current_key_29_cast_fp16)[name = string("inputs_61_cast_fp16")]; + tensor inputs_sq_61_cast_fp16 = mul(x = inputs_61_cast_fp16, y = inputs_61_cast_fp16)[name = string("inputs_sq_61_cast_fp16")]; + tensor variance_61_axes_0 = const()[name = string("variance_61_axes_0"), val = tensor([1])]; + bool variance_61_keep_dims_0 = const()[name = string("variance_61_keep_dims_0"), val = bool(true)]; + tensor variance_61_cast_fp16 = reduce_mean(axes = variance_61_axes_0, keep_dims = variance_61_keep_dims_0, x = inputs_sq_61_cast_fp16)[name = string("variance_61_cast_fp16")]; + fp16 var_2222_to_fp16 = const()[name = string("op_2222_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2223_cast_fp16 = add(x = variance_61_cast_fp16, y = var_2222_to_fp16)[name = string("op_2223_cast_fp16")]; + fp32 var_2224_epsilon_0 = const()[name = string("op_2224_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2224_cast_fp16 = rsqrt(epsilon = var_2224_epsilon_0, x = var_2223_cast_fp16)[name = string("op_2224_cast_fp16")]; + tensor hidden_states_75_cast_fp16 = mul(x = inputs_61_cast_fp16, y = var_2224_cast_fp16)[name = string("hidden_states_75_cast_fp16")]; + tensor current_key_normed_15_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_75_cast_fp16)[name = string("current_key_normed_15_cast_fp16")]; + tensor var_2242 = const()[name = string("op_2242"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_57_cast_fp16 = reshape(shape = var_2242, x = query_normed_15_cast_fp16)[name = string("mh_q_57_cast_fp16")]; + tensor var_2244 = const()[name = string("op_2244"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_57_cast_fp16 = reshape(shape = var_2244, x = current_key_normed_15_cast_fp16)[name = string("mh_k_57_cast_fp16")]; + tensor var_2248_cast_fp16 = mul(x = mh_q_57_cast_fp16, y = cos_11_to_fp16)[name = string("op_2248_cast_fp16")]; + tensor var_2253_begin_0 = const()[name = string("op_2253_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2253_end_0 = const()[name = string("op_2253_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_2253_end_mask_0 = const()[name = string("op_2253_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_2253_cast_fp16 = slice_by_index(begin = var_2253_begin_0, end = var_2253_end_0, end_mask = var_2253_end_mask_0, x = mh_q_57_cast_fp16)[name = string("op_2253_cast_fp16")]; + tensor var_2259_begin_0 = const()[name = string("op_2259_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_2259_end_0 = const()[name = string("op_2259_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_2259_end_mask_0 = const()[name = string("op_2259_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_2259_cast_fp16 = slice_by_index(begin = var_2259_begin_0, end = var_2259_end_0, end_mask = var_2259_end_mask_0, x = mh_q_57_cast_fp16)[name = string("op_2259_cast_fp16")]; + fp16 const_155_promoted_to_fp16 = const()[name = string("const_155_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2261_cast_fp16 = mul(x = var_2259_cast_fp16, y = const_155_promoted_to_fp16)[name = string("op_2261_cast_fp16")]; + bool var_2263_interleave_0 = const()[name = string("op_2263_interleave_0"), val = bool(false)]; + tensor var_2263_cast_fp16 = concat(axis = var_2147, interleave = var_2263_interleave_0, values = (var_2261_cast_fp16, var_2253_cast_fp16))[name = string("op_2263_cast_fp16")]; + tensor var_2264_cast_fp16 = mul(x = var_2263_cast_fp16, y = sin_11_to_fp16)[name = string("op_2264_cast_fp16")]; + tensor mh_q_59_cast_fp16 = add(x = var_2248_cast_fp16, y = var_2264_cast_fp16)[name = string("mh_q_59_cast_fp16")]; + tensor var_2266_cast_fp16 = mul(x = mh_k_57_cast_fp16, y = cos_11_to_fp16)[name = string("op_2266_cast_fp16")]; + tensor var_2271_begin_0 = const()[name = string("op_2271_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2271_end_0 = const()[name = string("op_2271_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_2271_end_mask_0 = const()[name = string("op_2271_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_2271_cast_fp16 = slice_by_index(begin = var_2271_begin_0, end = var_2271_end_0, end_mask = var_2271_end_mask_0, x = mh_k_57_cast_fp16)[name = string("op_2271_cast_fp16")]; + tensor var_2277_begin_0 = const()[name = string("op_2277_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_2277_end_0 = const()[name = string("op_2277_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_2277_end_mask_0 = const()[name = string("op_2277_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_2277_cast_fp16 = slice_by_index(begin = var_2277_begin_0, end = var_2277_end_0, end_mask = var_2277_end_mask_0, x = mh_k_57_cast_fp16)[name = string("op_2277_cast_fp16")]; + fp16 const_158_promoted_to_fp16 = const()[name = string("const_158_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2279_cast_fp16 = mul(x = var_2277_cast_fp16, y = const_158_promoted_to_fp16)[name = string("op_2279_cast_fp16")]; + bool var_2281_interleave_0 = const()[name = string("op_2281_interleave_0"), val = bool(false)]; + tensor var_2281_cast_fp16 = concat(axis = var_2147, interleave = var_2281_interleave_0, values = (var_2279_cast_fp16, var_2271_cast_fp16))[name = string("op_2281_cast_fp16")]; + tensor var_2282_cast_fp16 = mul(x = var_2281_cast_fp16, y = sin_11_to_fp16)[name = string("op_2282_cast_fp16")]; + tensor mh_k_59_cast_fp16 = add(x = var_2266_cast_fp16, y = var_2282_cast_fp16)[name = string("mh_k_59_cast_fp16")]; + tensor var_2286 = const()[name = string("op_2286"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_31_cast_fp16 = reshape(shape = var_2286, x = mh_k_59_cast_fp16)[name = string("current_key_31_cast_fp16")]; + tensor var_2292_to_fp16 = const()[name = string("op_2292_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274688)))]; + tensor var_2293_cast_fp16 = mul(x = obj_67_cast_fp16, y = var_2292_to_fp16)[name = string("op_2293_cast_fp16")]; + tensor var_2290_to_fp16 = const()[name = string("op_2290_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274816)))]; + tensor var_2294_cast_fp16 = mul(x = current_key_31_cast_fp16, y = var_2290_to_fp16)[name = string("op_2294_cast_fp16")]; + tensor key_31_cast_fp16 = add(x = var_2293_cast_fp16, y = var_2294_cast_fp16)[name = string("key_31_cast_fp16")]; + tensor var_2296_to_fp16 = const()[name = string("op_2296_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274688)))]; + tensor var_2297_cast_fp16 = mul(x = obj_69_cast_fp16, y = var_2296_to_fp16)[name = string("op_2297_cast_fp16")]; + tensor var_2298_cast_fp16 = mul(x = current_value_15_cast_fp16, y = var_2290_to_fp16)[name = string("op_2298_cast_fp16")]; + tensor value_15_cast_fp16 = add(x = var_2297_cast_fp16, y = var_2298_cast_fp16)[name = string("value_15_cast_fp16")]; + fp16 var_2305_to_fp16 = const()[name = string("op_2305_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_63_cast_fp16 = mul(x = mh_q_59_cast_fp16, y = var_2305_to_fp16)[name = string("mh_q_63_cast_fp16")]; + tensor var_2307 = const()[name = string("op_2307"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_61_cast_fp16 = reshape(shape = var_2307, x = key_31_cast_fp16)[name = string("mh_k_61_cast_fp16")]; + tensor var_2309 = const()[name = string("op_2309"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_29_cast_fp16 = reshape(shape = var_2309, x = value_15_cast_fp16)[name = string("mh_v_29_cast_fp16")]; + tensor transpose_28_perm_0 = const()[name = string("transpose_28_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_14_reps_0 = const()[name = string("tile_14_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_28_cast_fp16 = transpose(perm = transpose_28_perm_0, x = mh_k_61_cast_fp16)[name = string("transpose_437")]; + tensor tile_14_cast_fp16 = tile(reps = tile_14_reps_0, x = transpose_28_cast_fp16)[name = string("tile_14_cast_fp16")]; + tensor concat_36 = const()[name = string("concat_36"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_28_cast_fp16 = reshape(shape = concat_36, x = tile_14_cast_fp16)[name = string("reshape_28_cast_fp16")]; + tensor transpose_29_perm_0 = const()[name = string("transpose_29_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_37 = const()[name = string("concat_37"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_29_cast_fp16 = transpose(perm = transpose_29_perm_0, x = reshape_28_cast_fp16)[name = string("transpose_436")]; + tensor reshape_29_cast_fp16 = reshape(shape = concat_37, x = transpose_29_cast_fp16)[name = string("reshape_29_cast_fp16")]; + tensor transpose_30_perm_0 = const()[name = string("transpose_30_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_15_reps_0 = const()[name = string("tile_15_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_30_cast_fp16 = transpose(perm = transpose_30_perm_0, x = mh_v_29_cast_fp16)[name = string("transpose_435")]; + tensor tile_15_cast_fp16 = tile(reps = tile_15_reps_0, x = transpose_30_cast_fp16)[name = string("tile_15_cast_fp16")]; + tensor concat_38 = const()[name = string("concat_38"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_30_cast_fp16 = reshape(shape = concat_38, x = tile_15_cast_fp16)[name = string("reshape_30_cast_fp16")]; + tensor transpose_31_perm_0 = const()[name = string("transpose_31_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_39 = const()[name = string("concat_39"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_31_cast_fp16 = transpose(perm = transpose_31_perm_0, x = reshape_30_cast_fp16)[name = string("transpose_434")]; + tensor reshape_31_cast_fp16 = reshape(shape = concat_39, x = transpose_31_cast_fp16)[name = string("reshape_31_cast_fp16")]; + tensor transpose_345_perm_0 = const()[name = string("transpose_345_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_43_transpose_x_1 = const()[name = string("mh_w_43_transpose_x_1"), val = bool(true)]; + bool mh_w_43_transpose_y_1 = const()[name = string("mh_w_43_transpose_y_1"), val = bool(false)]; + tensor transpose_345_cast_fp16 = transpose(perm = transpose_345_perm_0, x = reshape_29_cast_fp16)[name = string("transpose_433")]; + tensor mh_w_43_cast_fp16 = matmul(transpose_x = mh_w_43_transpose_x_1, transpose_y = mh_w_43_transpose_y_1, x = mh_q_63_cast_fp16, y = transpose_345_cast_fp16)[name = string("mh_w_43_cast_fp16")]; + tensor var_2317_to_fp16 = const()[name = string("op_2317_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274944)))]; + tensor mh_w_45_cast_fp16 = add(x = mh_w_43_cast_fp16, y = var_2317_to_fp16)[name = string("mh_w_45_cast_fp16")]; + tensor mh_w_47_cast_fp16 = softmax(axis = var_2137, x = mh_w_45_cast_fp16)[name = string("mh_w_47_cast_fp16")]; + tensor transpose_346_perm_0 = const()[name = string("transpose_346_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_15_transpose_x_1 = const()[name = string("attn_15_transpose_x_1"), val = bool(false)]; + bool attn_15_transpose_y_1 = const()[name = string("attn_15_transpose_y_1"), val = bool(true)]; + tensor transpose_346_cast_fp16 = transpose(perm = transpose_346_perm_0, x = reshape_31_cast_fp16)[name = string("transpose_432")]; + tensor attn_15_cast_fp16 = matmul(transpose_x = attn_15_transpose_x_1, transpose_y = attn_15_transpose_y_1, x = transpose_346_cast_fp16, y = mh_w_47_cast_fp16)[name = string("attn_15_cast_fp16")]; + tensor var_2323 = const()[name = string("op_2323"), val = tensor([1, 2048, 1, 1])]; + tensor input_57_cast_fp16 = reshape(shape = var_2323, x = attn_15_cast_fp16)[name = string("input_57_cast_fp16")]; + string obj_71_pad_type_0 = const()[name = string("obj_71_pad_type_0"), val = string("valid")]; + tensor obj_71_strides_0 = const()[name = string("obj_71_strides_0"), val = tensor([1, 1])]; + tensor obj_71_pad_0 = const()[name = string("obj_71_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_71_dilations_0 = const()[name = string("obj_71_dilations_0"), val = tensor([1, 1])]; + int32 obj_71_groups_0 = const()[name = string("obj_71_groups_0"), val = int32(1)]; + tensor obj_71_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_71_dilations_0, groups = obj_71_groups_0, pad = obj_71_pad_0, pad_type = obj_71_pad_type_0, strides = obj_71_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_57_cast_fp16)[name = string("obj_71_cast_fp16")]; + tensor inputs_63_cast_fp16 = add(x = inputs_57_cast_fp16, y = obj_71_cast_fp16)[name = string("inputs_63_cast_fp16")]; + tensor inputs_sq_63_cast_fp16 = mul(x = inputs_63_cast_fp16, y = inputs_63_cast_fp16)[name = string("inputs_sq_63_cast_fp16")]; + tensor variance_63_axes_0 = const()[name = string("variance_63_axes_0"), val = tensor([1])]; + bool variance_63_keep_dims_0 = const()[name = string("variance_63_keep_dims_0"), val = bool(true)]; + tensor variance_63_cast_fp16 = reduce_mean(axes = variance_63_axes_0, keep_dims = variance_63_keep_dims_0, x = inputs_sq_63_cast_fp16)[name = string("variance_63_cast_fp16")]; + fp16 var_2341_to_fp16 = const()[name = string("op_2341_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2342_cast_fp16 = add(x = variance_63_cast_fp16, y = var_2341_to_fp16)[name = string("op_2342_cast_fp16")]; + fp32 var_2343_epsilon_0 = const()[name = string("op_2343_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2343_cast_fp16 = rsqrt(epsilon = var_2343_epsilon_0, x = var_2342_cast_fp16)[name = string("op_2343_cast_fp16")]; + tensor hidden_states_77_cast_fp16 = mul(x = inputs_63_cast_fp16, y = var_2343_cast_fp16)[name = string("hidden_states_77_cast_fp16")]; + tensor input_59_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_77_cast_fp16)[name = string("input_59_cast_fp16")]; + string input_61_pad_type_0 = const()[name = string("input_61_pad_type_0"), val = string("valid")]; + tensor input_61_strides_0 = const()[name = string("input_61_strides_0"), val = tensor([1, 1])]; + tensor input_61_pad_0 = const()[name = string("input_61_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_61_dilations_0 = const()[name = string("input_61_dilations_0"), val = tensor([1, 1])]; + int32 input_61_groups_0 = const()[name = string("input_61_groups_0"), val = int32(1)]; + tensor input_61_cast_fp16 = conv(dilations = input_61_dilations_0, groups = input_61_groups_0, pad = input_61_pad_0, pad_type = input_61_pad_type_0, strides = input_61_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_59_cast_fp16)[name = string("input_61_cast_fp16")]; + tensor var_2357_cast_fp16 = silu(x = input_61_cast_fp16)[name = string("op_2357_cast_fp16")]; + string var_2363_pad_type_0 = const()[name = string("op_2363_pad_type_0"), val = string("valid")]; + tensor var_2363_strides_0 = const()[name = string("op_2363_strides_0"), val = tensor([1, 1])]; + tensor var_2363_pad_0 = const()[name = string("op_2363_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2363_dilations_0 = const()[name = string("op_2363_dilations_0"), val = tensor([1, 1])]; + int32 var_2363_groups_0 = const()[name = string("op_2363_groups_0"), val = int32(1)]; + tensor var_2363_cast_fp16 = conv(dilations = var_2363_dilations_0, groups = var_2363_groups_0, pad = var_2363_pad_0, pad_type = var_2363_pad_type_0, strides = var_2363_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_59_cast_fp16)[name = string("op_2363_cast_fp16")]; + tensor input_63_cast_fp16 = mul(x = var_2357_cast_fp16, y = var_2363_cast_fp16)[name = string("input_63_cast_fp16")]; + string hidden_states_79_pad_type_0 = const()[name = string("hidden_states_79_pad_type_0"), val = string("valid")]; + tensor hidden_states_79_strides_0 = const()[name = string("hidden_states_79_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_79_pad_0 = const()[name = string("hidden_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_79_dilations_0 = const()[name = string("hidden_states_79_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_79_groups_0 = const()[name = string("hidden_states_79_groups_0"), val = int32(1)]; + tensor hidden_states_79_cast_fp16 = conv(dilations = hidden_states_79_dilations_0, groups = hidden_states_79_groups_0, pad = hidden_states_79_pad_0, pad_type = hidden_states_79_pad_type_0, strides = hidden_states_79_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_63_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; + tensor inputs_65_cast_fp16 = add(x = inputs_63_cast_fp16, y = hidden_states_79_cast_fp16)[name = string("inputs_65_cast_fp16")]; + tensor obj_75_begin_0 = const()[name = string("obj_75_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_75_end_0 = const()[name = string("obj_75_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_75_end_mask_0 = const()[name = string("obj_75_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_75_cast_fp16 = slice_by_index(begin = obj_75_begin_0, end = obj_75_end_0, end_mask = obj_75_end_mask_0, x = key_caches_3_cast_fp16)[name = string("obj_75_cast_fp16")]; + tensor obj_77_begin_0 = const()[name = string("obj_77_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_77_end_0 = const()[name = string("obj_77_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_77_end_mask_0 = const()[name = string("obj_77_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_77_cast_fp16 = slice_by_index(begin = obj_77_begin_0, end = obj_77_end_0, end_mask = obj_77_end_mask_0, x = value_caches_3_cast_fp16)[name = string("obj_77_cast_fp16")]; + int32 var_2411 = const()[name = string("op_2411"), val = int32(3)]; + int32 var_2421 = const()[name = string("op_2421"), val = int32(-2)]; + tensor inputs_sq_65_cast_fp16 = mul(x = inputs_65_cast_fp16, y = inputs_65_cast_fp16)[name = string("inputs_sq_65_cast_fp16")]; + tensor variance_65_axes_0 = const()[name = string("variance_65_axes_0"), val = tensor([1])]; + bool variance_65_keep_dims_0 = const()[name = string("variance_65_keep_dims_0"), val = bool(true)]; + tensor variance_65_cast_fp16 = reduce_mean(axes = variance_65_axes_0, keep_dims = variance_65_keep_dims_0, x = inputs_sq_65_cast_fp16)[name = string("variance_65_cast_fp16")]; + fp16 var_2435_to_fp16 = const()[name = string("op_2435_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2436_cast_fp16 = add(x = variance_65_cast_fp16, y = var_2435_to_fp16)[name = string("op_2436_cast_fp16")]; + fp32 var_2437_epsilon_0 = const()[name = string("op_2437_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2437_cast_fp16 = rsqrt(epsilon = var_2437_epsilon_0, x = var_2436_cast_fp16)[name = string("op_2437_cast_fp16")]; + tensor hidden_states_81_cast_fp16 = mul(x = inputs_65_cast_fp16, y = var_2437_cast_fp16)[name = string("hidden_states_81_cast_fp16")]; + tensor obj_73_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_81_cast_fp16)[name = string("obj_73_cast_fp16")]; + string query_49_pad_type_0 = const()[name = string("query_49_pad_type_0"), val = string("valid")]; + tensor query_49_strides_0 = const()[name = string("query_49_strides_0"), val = tensor([1, 1])]; + tensor query_49_pad_0 = const()[name = string("query_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_49_dilations_0 = const()[name = string("query_49_dilations_0"), val = tensor([1, 1])]; + int32 query_49_groups_0 = const()[name = string("query_49_groups_0"), val = int32(1)]; + tensor query_49_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_49_dilations_0, groups = query_49_groups_0, pad = query_49_pad_0, pad_type = query_49_pad_type_0, strides = query_49_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_73_cast_fp16)[name = string("query_49_cast_fp16")]; + string current_key_33_pad_type_0 = const()[name = string("current_key_33_pad_type_0"), val = string("valid")]; + tensor current_key_33_strides_0 = const()[name = string("current_key_33_strides_0"), val = tensor([1, 1])]; + tensor current_key_33_pad_0 = const()[name = string("current_key_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_33_dilations_0 = const()[name = string("current_key_33_dilations_0"), val = tensor([1, 1])]; + int32 current_key_33_groups_0 = const()[name = string("current_key_33_groups_0"), val = int32(1)]; + tensor current_key_33_cast_fp16 = conv(dilations = current_key_33_dilations_0, groups = current_key_33_groups_0, pad = current_key_33_pad_0, pad_type = current_key_33_pad_type_0, strides = current_key_33_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_73_cast_fp16)[name = string("current_key_33_cast_fp16")]; + string current_value_17_pad_type_0 = const()[name = string("current_value_17_pad_type_0"), val = string("valid")]; + tensor current_value_17_strides_0 = const()[name = string("current_value_17_strides_0"), val = tensor([1, 1])]; + tensor current_value_17_pad_0 = const()[name = string("current_value_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_17_dilations_0 = const()[name = string("current_value_17_dilations_0"), val = tensor([1, 1])]; + int32 current_value_17_groups_0 = const()[name = string("current_value_17_groups_0"), val = int32(1)]; + tensor current_value_17_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_17_dilations_0, groups = current_value_17_groups_0, pad = current_value_17_pad_0, pad_type = current_value_17_pad_type_0, strides = current_value_17_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_73_cast_fp16)[name = string("current_value_17_cast_fp16")]; + tensor var_2474 = const()[name = string("op_2474"), val = tensor([16, 128, 1, 1])]; + tensor inputs_67_cast_fp16 = reshape(shape = var_2474, x = query_49_cast_fp16)[name = string("inputs_67_cast_fp16")]; + tensor inputs_sq_67_cast_fp16 = mul(x = inputs_67_cast_fp16, y = inputs_67_cast_fp16)[name = string("inputs_sq_67_cast_fp16")]; + tensor variance_67_axes_0 = const()[name = string("variance_67_axes_0"), val = tensor([1])]; + bool variance_67_keep_dims_0 = const()[name = string("variance_67_keep_dims_0"), val = bool(true)]; + tensor variance_67_cast_fp16 = reduce_mean(axes = variance_67_axes_0, keep_dims = variance_67_keep_dims_0, x = inputs_sq_67_cast_fp16)[name = string("variance_67_cast_fp16")]; + fp16 var_2480_to_fp16 = const()[name = string("op_2480_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2481_cast_fp16 = add(x = variance_67_cast_fp16, y = var_2480_to_fp16)[name = string("op_2481_cast_fp16")]; + fp32 var_2482_epsilon_0 = const()[name = string("op_2482_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2482_cast_fp16 = rsqrt(epsilon = var_2482_epsilon_0, x = var_2481_cast_fp16)[name = string("op_2482_cast_fp16")]; + tensor hidden_states_83_cast_fp16 = mul(x = inputs_67_cast_fp16, y = var_2482_cast_fp16)[name = string("hidden_states_83_cast_fp16")]; + tensor query_normed_17_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_83_cast_fp16)[name = string("query_normed_17_cast_fp16")]; + tensor var_2490 = const()[name = string("op_2490"), val = tensor([8, 128, 1, 1])]; + tensor inputs_69_cast_fp16 = reshape(shape = var_2490, x = current_key_33_cast_fp16)[name = string("inputs_69_cast_fp16")]; + tensor inputs_sq_69_cast_fp16 = mul(x = inputs_69_cast_fp16, y = inputs_69_cast_fp16)[name = string("inputs_sq_69_cast_fp16")]; + tensor variance_69_axes_0 = const()[name = string("variance_69_axes_0"), val = tensor([1])]; + bool variance_69_keep_dims_0 = const()[name = string("variance_69_keep_dims_0"), val = bool(true)]; + tensor variance_69_cast_fp16 = reduce_mean(axes = variance_69_axes_0, keep_dims = variance_69_keep_dims_0, x = inputs_sq_69_cast_fp16)[name = string("variance_69_cast_fp16")]; + fp16 var_2496_to_fp16 = const()[name = string("op_2496_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2497_cast_fp16 = add(x = variance_69_cast_fp16, y = var_2496_to_fp16)[name = string("op_2497_cast_fp16")]; + fp32 var_2498_epsilon_0 = const()[name = string("op_2498_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2498_cast_fp16 = rsqrt(epsilon = var_2498_epsilon_0, x = var_2497_cast_fp16)[name = string("op_2498_cast_fp16")]; + tensor hidden_states_85_cast_fp16 = mul(x = inputs_69_cast_fp16, y = var_2498_cast_fp16)[name = string("hidden_states_85_cast_fp16")]; + tensor current_key_normed_17_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_85_cast_fp16)[name = string("current_key_normed_17_cast_fp16")]; + tensor var_2516 = const()[name = string("op_2516"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_65_cast_fp16 = reshape(shape = var_2516, x = query_normed_17_cast_fp16)[name = string("mh_q_65_cast_fp16")]; + tensor var_2518 = const()[name = string("op_2518"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_65_cast_fp16 = reshape(shape = var_2518, x = current_key_normed_17_cast_fp16)[name = string("mh_k_65_cast_fp16")]; + tensor var_2522_cast_fp16 = mul(x = mh_q_65_cast_fp16, y = cos_11_to_fp16)[name = string("op_2522_cast_fp16")]; + tensor var_2527_begin_0 = const()[name = string("op_2527_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2527_end_0 = const()[name = string("op_2527_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_2527_end_mask_0 = const()[name = string("op_2527_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_2527_cast_fp16 = slice_by_index(begin = var_2527_begin_0, end = var_2527_end_0, end_mask = var_2527_end_mask_0, x = mh_q_65_cast_fp16)[name = string("op_2527_cast_fp16")]; + tensor var_2533_begin_0 = const()[name = string("op_2533_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_2533_end_0 = const()[name = string("op_2533_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_2533_end_mask_0 = const()[name = string("op_2533_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_2533_cast_fp16 = slice_by_index(begin = var_2533_begin_0, end = var_2533_end_0, end_mask = var_2533_end_mask_0, x = mh_q_65_cast_fp16)[name = string("op_2533_cast_fp16")]; + fp16 const_175_promoted_to_fp16 = const()[name = string("const_175_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2535_cast_fp16 = mul(x = var_2533_cast_fp16, y = const_175_promoted_to_fp16)[name = string("op_2535_cast_fp16")]; + bool var_2537_interleave_0 = const()[name = string("op_2537_interleave_0"), val = bool(false)]; + tensor var_2537_cast_fp16 = concat(axis = var_2421, interleave = var_2537_interleave_0, values = (var_2535_cast_fp16, var_2527_cast_fp16))[name = string("op_2537_cast_fp16")]; + tensor var_2538_cast_fp16 = mul(x = var_2537_cast_fp16, y = sin_11_to_fp16)[name = string("op_2538_cast_fp16")]; + tensor mh_q_67_cast_fp16 = add(x = var_2522_cast_fp16, y = var_2538_cast_fp16)[name = string("mh_q_67_cast_fp16")]; + tensor var_2540_cast_fp16 = mul(x = mh_k_65_cast_fp16, y = cos_11_to_fp16)[name = string("op_2540_cast_fp16")]; + tensor var_2545_begin_0 = const()[name = string("op_2545_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2545_end_0 = const()[name = string("op_2545_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_2545_end_mask_0 = const()[name = string("op_2545_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_2545_cast_fp16 = slice_by_index(begin = var_2545_begin_0, end = var_2545_end_0, end_mask = var_2545_end_mask_0, x = mh_k_65_cast_fp16)[name = string("op_2545_cast_fp16")]; + tensor var_2551_begin_0 = const()[name = string("op_2551_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_2551_end_0 = const()[name = string("op_2551_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_2551_end_mask_0 = const()[name = string("op_2551_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_2551_cast_fp16 = slice_by_index(begin = var_2551_begin_0, end = var_2551_end_0, end_mask = var_2551_end_mask_0, x = mh_k_65_cast_fp16)[name = string("op_2551_cast_fp16")]; + fp16 const_178_promoted_to_fp16 = const()[name = string("const_178_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2553_cast_fp16 = mul(x = var_2551_cast_fp16, y = const_178_promoted_to_fp16)[name = string("op_2553_cast_fp16")]; + bool var_2555_interleave_0 = const()[name = string("op_2555_interleave_0"), val = bool(false)]; + tensor var_2555_cast_fp16 = concat(axis = var_2421, interleave = var_2555_interleave_0, values = (var_2553_cast_fp16, var_2545_cast_fp16))[name = string("op_2555_cast_fp16")]; + tensor var_2556_cast_fp16 = mul(x = var_2555_cast_fp16, y = sin_11_to_fp16)[name = string("op_2556_cast_fp16")]; + tensor mh_k_67_cast_fp16 = add(x = var_2540_cast_fp16, y = var_2556_cast_fp16)[name = string("mh_k_67_cast_fp16")]; + tensor var_2560 = const()[name = string("op_2560"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_35_cast_fp16 = reshape(shape = var_2560, x = mh_k_67_cast_fp16)[name = string("current_key_35_cast_fp16")]; + tensor var_2566_to_fp16 = const()[name = string("op_2566_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274688)))]; + tensor var_2567_cast_fp16 = mul(x = obj_75_cast_fp16, y = var_2566_to_fp16)[name = string("op_2567_cast_fp16")]; + tensor var_2564_to_fp16 = const()[name = string("op_2564_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274816)))]; + tensor var_2568_cast_fp16 = mul(x = current_key_35_cast_fp16, y = var_2564_to_fp16)[name = string("op_2568_cast_fp16")]; + tensor key_35_cast_fp16 = add(x = var_2567_cast_fp16, y = var_2568_cast_fp16)[name = string("key_35_cast_fp16")]; + tensor var_2570_to_fp16 = const()[name = string("op_2570_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274688)))]; + tensor var_2571_cast_fp16 = mul(x = obj_77_cast_fp16, y = var_2570_to_fp16)[name = string("op_2571_cast_fp16")]; + tensor var_2572_cast_fp16 = mul(x = current_value_17_cast_fp16, y = var_2564_to_fp16)[name = string("op_2572_cast_fp16")]; + tensor value_17_cast_fp16 = add(x = var_2571_cast_fp16, y = var_2572_cast_fp16)[name = string("value_17_cast_fp16")]; + fp16 var_2579_to_fp16 = const()[name = string("op_2579_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_71_cast_fp16 = mul(x = mh_q_67_cast_fp16, y = var_2579_to_fp16)[name = string("mh_q_71_cast_fp16")]; + tensor var_2581 = const()[name = string("op_2581"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_69_cast_fp16 = reshape(shape = var_2581, x = key_35_cast_fp16)[name = string("mh_k_69_cast_fp16")]; + tensor var_2583 = const()[name = string("op_2583"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_33_cast_fp16 = reshape(shape = var_2583, x = value_17_cast_fp16)[name = string("mh_v_33_cast_fp16")]; + tensor transpose_32_perm_0 = const()[name = string("transpose_32_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_16_reps_0 = const()[name = string("tile_16_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_32_cast_fp16 = transpose(perm = transpose_32_perm_0, x = mh_k_69_cast_fp16)[name = string("transpose_431")]; + tensor tile_16_cast_fp16 = tile(reps = tile_16_reps_0, x = transpose_32_cast_fp16)[name = string("tile_16_cast_fp16")]; + tensor concat_40 = const()[name = string("concat_40"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_32_cast_fp16 = reshape(shape = concat_40, x = tile_16_cast_fp16)[name = string("reshape_32_cast_fp16")]; + tensor transpose_33_perm_0 = const()[name = string("transpose_33_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_41 = const()[name = string("concat_41"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_33_cast_fp16 = transpose(perm = transpose_33_perm_0, x = reshape_32_cast_fp16)[name = string("transpose_430")]; + tensor reshape_33_cast_fp16 = reshape(shape = concat_41, x = transpose_33_cast_fp16)[name = string("reshape_33_cast_fp16")]; + tensor transpose_34_perm_0 = const()[name = string("transpose_34_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_17_reps_0 = const()[name = string("tile_17_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_34_cast_fp16 = transpose(perm = transpose_34_perm_0, x = mh_v_33_cast_fp16)[name = string("transpose_429")]; + tensor tile_17_cast_fp16 = tile(reps = tile_17_reps_0, x = transpose_34_cast_fp16)[name = string("tile_17_cast_fp16")]; + tensor concat_42 = const()[name = string("concat_42"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_34_cast_fp16 = reshape(shape = concat_42, x = tile_17_cast_fp16)[name = string("reshape_34_cast_fp16")]; + tensor transpose_35_perm_0 = const()[name = string("transpose_35_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_43 = const()[name = string("concat_43"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_35_cast_fp16 = transpose(perm = transpose_35_perm_0, x = reshape_34_cast_fp16)[name = string("transpose_428")]; + tensor reshape_35_cast_fp16 = reshape(shape = concat_43, x = transpose_35_cast_fp16)[name = string("reshape_35_cast_fp16")]; + tensor transpose_349_perm_0 = const()[name = string("transpose_349_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_49_transpose_x_1 = const()[name = string("mh_w_49_transpose_x_1"), val = bool(true)]; + bool mh_w_49_transpose_y_1 = const()[name = string("mh_w_49_transpose_y_1"), val = bool(false)]; + tensor transpose_349_cast_fp16 = transpose(perm = transpose_349_perm_0, x = reshape_33_cast_fp16)[name = string("transpose_427")]; + tensor mh_w_49_cast_fp16 = matmul(transpose_x = mh_w_49_transpose_x_1, transpose_y = mh_w_49_transpose_y_1, x = mh_q_71_cast_fp16, y = transpose_349_cast_fp16)[name = string("mh_w_49_cast_fp16")]; + tensor var_2591_to_fp16 = const()[name = string("op_2591_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274944)))]; + tensor mh_w_51_cast_fp16 = add(x = mh_w_49_cast_fp16, y = var_2591_to_fp16)[name = string("mh_w_51_cast_fp16")]; + tensor mh_w_53_cast_fp16 = softmax(axis = var_2411, x = mh_w_51_cast_fp16)[name = string("mh_w_53_cast_fp16")]; + tensor transpose_350_perm_0 = const()[name = string("transpose_350_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_17_transpose_x_1 = const()[name = string("attn_17_transpose_x_1"), val = bool(false)]; + bool attn_17_transpose_y_1 = const()[name = string("attn_17_transpose_y_1"), val = bool(true)]; + tensor transpose_350_cast_fp16 = transpose(perm = transpose_350_perm_0, x = reshape_35_cast_fp16)[name = string("transpose_426")]; + tensor attn_17_cast_fp16 = matmul(transpose_x = attn_17_transpose_x_1, transpose_y = attn_17_transpose_y_1, x = transpose_350_cast_fp16, y = mh_w_53_cast_fp16)[name = string("attn_17_cast_fp16")]; + tensor var_2597 = const()[name = string("op_2597"), val = tensor([1, 2048, 1, 1])]; + tensor input_65_cast_fp16 = reshape(shape = var_2597, x = attn_17_cast_fp16)[name = string("input_65_cast_fp16")]; + string obj_79_pad_type_0 = const()[name = string("obj_79_pad_type_0"), val = string("valid")]; + tensor obj_79_strides_0 = const()[name = string("obj_79_strides_0"), val = tensor([1, 1])]; + tensor obj_79_pad_0 = const()[name = string("obj_79_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_79_dilations_0 = const()[name = string("obj_79_dilations_0"), val = tensor([1, 1])]; + int32 obj_79_groups_0 = const()[name = string("obj_79_groups_0"), val = int32(1)]; + tensor obj_79_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_79_dilations_0, groups = obj_79_groups_0, pad = obj_79_pad_0, pad_type = obj_79_pad_type_0, strides = obj_79_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_65_cast_fp16)[name = string("obj_79_cast_fp16")]; + tensor inputs_71_cast_fp16 = add(x = inputs_65_cast_fp16, y = obj_79_cast_fp16)[name = string("inputs_71_cast_fp16")]; + tensor inputs_sq_71_cast_fp16 = mul(x = inputs_71_cast_fp16, y = inputs_71_cast_fp16)[name = string("inputs_sq_71_cast_fp16")]; + tensor variance_71_axes_0 = const()[name = string("variance_71_axes_0"), val = tensor([1])]; + bool variance_71_keep_dims_0 = const()[name = string("variance_71_keep_dims_0"), val = bool(true)]; + tensor variance_71_cast_fp16 = reduce_mean(axes = variance_71_axes_0, keep_dims = variance_71_keep_dims_0, x = inputs_sq_71_cast_fp16)[name = string("variance_71_cast_fp16")]; + fp16 var_2615_to_fp16 = const()[name = string("op_2615_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2616_cast_fp16 = add(x = variance_71_cast_fp16, y = var_2615_to_fp16)[name = string("op_2616_cast_fp16")]; + fp32 var_2617_epsilon_0 = const()[name = string("op_2617_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2617_cast_fp16 = rsqrt(epsilon = var_2617_epsilon_0, x = var_2616_cast_fp16)[name = string("op_2617_cast_fp16")]; + tensor hidden_states_87_cast_fp16 = mul(x = inputs_71_cast_fp16, y = var_2617_cast_fp16)[name = string("hidden_states_87_cast_fp16")]; + tensor input_67_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_87_cast_fp16)[name = string("input_67_cast_fp16")]; + string input_69_pad_type_0 = const()[name = string("input_69_pad_type_0"), val = string("valid")]; + tensor input_69_strides_0 = const()[name = string("input_69_strides_0"), val = tensor([1, 1])]; + tensor input_69_pad_0 = const()[name = string("input_69_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_69_dilations_0 = const()[name = string("input_69_dilations_0"), val = tensor([1, 1])]; + int32 input_69_groups_0 = const()[name = string("input_69_groups_0"), val = int32(1)]; + tensor input_69_cast_fp16 = conv(dilations = input_69_dilations_0, groups = input_69_groups_0, pad = input_69_pad_0, pad_type = input_69_pad_type_0, strides = input_69_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_67_cast_fp16)[name = string("input_69_cast_fp16")]; + tensor var_2631_cast_fp16 = silu(x = input_69_cast_fp16)[name = string("op_2631_cast_fp16")]; + string var_2637_pad_type_0 = const()[name = string("op_2637_pad_type_0"), val = string("valid")]; + tensor var_2637_strides_0 = const()[name = string("op_2637_strides_0"), val = tensor([1, 1])]; + tensor var_2637_pad_0 = const()[name = string("op_2637_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2637_dilations_0 = const()[name = string("op_2637_dilations_0"), val = tensor([1, 1])]; + int32 var_2637_groups_0 = const()[name = string("op_2637_groups_0"), val = int32(1)]; + tensor var_2637_cast_fp16 = conv(dilations = var_2637_dilations_0, groups = var_2637_groups_0, pad = var_2637_pad_0, pad_type = var_2637_pad_type_0, strides = var_2637_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_67_cast_fp16)[name = string("op_2637_cast_fp16")]; + tensor input_71_cast_fp16 = mul(x = var_2631_cast_fp16, y = var_2637_cast_fp16)[name = string("input_71_cast_fp16")]; + string hidden_states_89_pad_type_0 = const()[name = string("hidden_states_89_pad_type_0"), val = string("valid")]; + tensor hidden_states_89_strides_0 = const()[name = string("hidden_states_89_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_89_pad_0 = const()[name = string("hidden_states_89_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_89_dilations_0 = const()[name = string("hidden_states_89_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_89_groups_0 = const()[name = string("hidden_states_89_groups_0"), val = int32(1)]; + tensor hidden_states_89_cast_fp16 = conv(dilations = hidden_states_89_dilations_0, groups = hidden_states_89_groups_0, pad = hidden_states_89_pad_0, pad_type = hidden_states_89_pad_type_0, strides = hidden_states_89_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_71_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; + tensor inputs_73_cast_fp16 = add(x = inputs_71_cast_fp16, y = hidden_states_89_cast_fp16)[name = string("inputs_73_cast_fp16")]; + tensor obj_83_begin_0 = const()[name = string("obj_83_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_83_end_0 = const()[name = string("obj_83_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_83_end_mask_0 = const()[name = string("obj_83_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_83_cast_fp16 = slice_by_index(begin = obj_83_begin_0, end = obj_83_end_0, end_mask = obj_83_end_mask_0, x = key_caches_3_cast_fp16)[name = string("obj_83_cast_fp16")]; + tensor obj_85_begin_0 = const()[name = string("obj_85_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_85_end_0 = const()[name = string("obj_85_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_85_end_mask_0 = const()[name = string("obj_85_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_85_cast_fp16 = slice_by_index(begin = obj_85_begin_0, end = obj_85_end_0, end_mask = obj_85_end_mask_0, x = value_caches_3_cast_fp16)[name = string("obj_85_cast_fp16")]; + int32 var_2685 = const()[name = string("op_2685"), val = int32(3)]; + int32 var_2695 = const()[name = string("op_2695"), val = int32(-2)]; + tensor inputs_sq_73_cast_fp16 = mul(x = inputs_73_cast_fp16, y = inputs_73_cast_fp16)[name = string("inputs_sq_73_cast_fp16")]; + tensor variance_73_axes_0 = const()[name = string("variance_73_axes_0"), val = tensor([1])]; + bool variance_73_keep_dims_0 = const()[name = string("variance_73_keep_dims_0"), val = bool(true)]; + tensor variance_73_cast_fp16 = reduce_mean(axes = variance_73_axes_0, keep_dims = variance_73_keep_dims_0, x = inputs_sq_73_cast_fp16)[name = string("variance_73_cast_fp16")]; + fp16 var_2709_to_fp16 = const()[name = string("op_2709_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2710_cast_fp16 = add(x = variance_73_cast_fp16, y = var_2709_to_fp16)[name = string("op_2710_cast_fp16")]; + fp32 var_2711_epsilon_0 = const()[name = string("op_2711_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2711_cast_fp16 = rsqrt(epsilon = var_2711_epsilon_0, x = var_2710_cast_fp16)[name = string("op_2711_cast_fp16")]; + tensor hidden_states_91_cast_fp16 = mul(x = inputs_73_cast_fp16, y = var_2711_cast_fp16)[name = string("hidden_states_91_cast_fp16")]; + tensor obj_81_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_91_cast_fp16)[name = string("obj_81_cast_fp16")]; + string query_55_pad_type_0 = const()[name = string("query_55_pad_type_0"), val = string("valid")]; + tensor query_55_strides_0 = const()[name = string("query_55_strides_0"), val = tensor([1, 1])]; + tensor query_55_pad_0 = const()[name = string("query_55_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_55_dilations_0 = const()[name = string("query_55_dilations_0"), val = tensor([1, 1])]; + int32 query_55_groups_0 = const()[name = string("query_55_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65068608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67165824))))[name = string("layers_4_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor query_55_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_55_dilations_0, groups = query_55_groups_0, pad = query_55_pad_0, pad_type = query_55_pad_type_0, strides = query_55_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_81_cast_fp16)[name = string("query_55_cast_fp16")]; + string current_key_37_pad_type_0 = const()[name = string("current_key_37_pad_type_0"), val = string("valid")]; + tensor current_key_37_strides_0 = const()[name = string("current_key_37_strides_0"), val = tensor([1, 1])]; + tensor current_key_37_pad_0 = const()[name = string("current_key_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_37_dilations_0 = const()[name = string("current_key_37_dilations_0"), val = tensor([1, 1])]; + int32 current_key_37_groups_0 = const()[name = string("current_key_37_groups_0"), val = int32(1)]; + tensor current_key_37_cast_fp16 = conv(dilations = current_key_37_dilations_0, groups = current_key_37_groups_0, pad = current_key_37_pad_0, pad_type = current_key_37_pad_type_0, strides = current_key_37_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_81_cast_fp16)[name = string("current_key_37_cast_fp16")]; + string current_value_19_pad_type_0 = const()[name = string("current_value_19_pad_type_0"), val = string("valid")]; + tensor current_value_19_strides_0 = const()[name = string("current_value_19_strides_0"), val = tensor([1, 1])]; + tensor current_value_19_pad_0 = const()[name = string("current_value_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_19_dilations_0 = const()[name = string("current_value_19_dilations_0"), val = tensor([1, 1])]; + int32 current_value_19_groups_0 = const()[name = string("current_value_19_groups_0"), val = int32(1)]; + tensor current_value_19_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_19_dilations_0, groups = current_value_19_groups_0, pad = current_value_19_pad_0, pad_type = current_value_19_pad_type_0, strides = current_value_19_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_81_cast_fp16)[name = string("current_value_19_cast_fp16")]; + tensor var_2748 = const()[name = string("op_2748"), val = tensor([16, 128, 1, 1])]; + tensor inputs_75_cast_fp16 = reshape(shape = var_2748, x = query_55_cast_fp16)[name = string("inputs_75_cast_fp16")]; + tensor inputs_sq_75_cast_fp16 = mul(x = inputs_75_cast_fp16, y = inputs_75_cast_fp16)[name = string("inputs_sq_75_cast_fp16")]; + tensor variance_75_axes_0 = const()[name = string("variance_75_axes_0"), val = tensor([1])]; + bool variance_75_keep_dims_0 = const()[name = string("variance_75_keep_dims_0"), val = bool(true)]; + tensor variance_75_cast_fp16 = reduce_mean(axes = variance_75_axes_0, keep_dims = variance_75_keep_dims_0, x = inputs_sq_75_cast_fp16)[name = string("variance_75_cast_fp16")]; + fp16 var_2754_to_fp16 = const()[name = string("op_2754_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2755_cast_fp16 = add(x = variance_75_cast_fp16, y = var_2754_to_fp16)[name = string("op_2755_cast_fp16")]; + fp32 var_2756_epsilon_0 = const()[name = string("op_2756_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2756_cast_fp16 = rsqrt(epsilon = var_2756_epsilon_0, x = var_2755_cast_fp16)[name = string("op_2756_cast_fp16")]; + tensor hidden_states_93_cast_fp16 = mul(x = inputs_75_cast_fp16, y = var_2756_cast_fp16)[name = string("hidden_states_93_cast_fp16")]; + tensor w_75_to_fp16 = const()[name = string("w_75_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69264832)))]; + tensor query_normed_19_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_93_cast_fp16)[name = string("query_normed_19_cast_fp16")]; + tensor var_2764 = const()[name = string("op_2764"), val = tensor([8, 128, 1, 1])]; + tensor inputs_77_cast_fp16 = reshape(shape = var_2764, x = current_key_37_cast_fp16)[name = string("inputs_77_cast_fp16")]; + tensor inputs_sq_77_cast_fp16 = mul(x = inputs_77_cast_fp16, y = inputs_77_cast_fp16)[name = string("inputs_sq_77_cast_fp16")]; + tensor variance_77_axes_0 = const()[name = string("variance_77_axes_0"), val = tensor([1])]; + bool variance_77_keep_dims_0 = const()[name = string("variance_77_keep_dims_0"), val = bool(true)]; + tensor variance_77_cast_fp16 = reduce_mean(axes = variance_77_axes_0, keep_dims = variance_77_keep_dims_0, x = inputs_sq_77_cast_fp16)[name = string("variance_77_cast_fp16")]; + fp16 var_2770_to_fp16 = const()[name = string("op_2770_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2771_cast_fp16 = add(x = variance_77_cast_fp16, y = var_2770_to_fp16)[name = string("op_2771_cast_fp16")]; + fp32 var_2772_epsilon_0 = const()[name = string("op_2772_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2772_cast_fp16 = rsqrt(epsilon = var_2772_epsilon_0, x = var_2771_cast_fp16)[name = string("op_2772_cast_fp16")]; + tensor hidden_states_95_cast_fp16 = mul(x = inputs_77_cast_fp16, y = var_2772_cast_fp16)[name = string("hidden_states_95_cast_fp16")]; + tensor current_key_normed_19_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_95_cast_fp16)[name = string("current_key_normed_19_cast_fp16")]; + tensor var_2790 = const()[name = string("op_2790"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_73_cast_fp16 = reshape(shape = var_2790, x = query_normed_19_cast_fp16)[name = string("mh_q_73_cast_fp16")]; + tensor var_2792 = const()[name = string("op_2792"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_73_cast_fp16 = reshape(shape = var_2792, x = current_key_normed_19_cast_fp16)[name = string("mh_k_73_cast_fp16")]; + tensor var_2796_cast_fp16 = mul(x = mh_q_73_cast_fp16, y = cos_11_to_fp16)[name = string("op_2796_cast_fp16")]; + tensor var_2801_begin_0 = const()[name = string("op_2801_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2801_end_0 = const()[name = string("op_2801_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_2801_end_mask_0 = const()[name = string("op_2801_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_2801_cast_fp16 = slice_by_index(begin = var_2801_begin_0, end = var_2801_end_0, end_mask = var_2801_end_mask_0, x = mh_q_73_cast_fp16)[name = string("op_2801_cast_fp16")]; + tensor var_2807_begin_0 = const()[name = string("op_2807_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_2807_end_0 = const()[name = string("op_2807_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_2807_end_mask_0 = const()[name = string("op_2807_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_2807_cast_fp16 = slice_by_index(begin = var_2807_begin_0, end = var_2807_end_0, end_mask = var_2807_end_mask_0, x = mh_q_73_cast_fp16)[name = string("op_2807_cast_fp16")]; + fp16 const_195_promoted_to_fp16 = const()[name = string("const_195_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2809_cast_fp16 = mul(x = var_2807_cast_fp16, y = const_195_promoted_to_fp16)[name = string("op_2809_cast_fp16")]; + bool var_2811_interleave_0 = const()[name = string("op_2811_interleave_0"), val = bool(false)]; + tensor var_2811_cast_fp16 = concat(axis = var_2695, interleave = var_2811_interleave_0, values = (var_2809_cast_fp16, var_2801_cast_fp16))[name = string("op_2811_cast_fp16")]; + tensor var_2812_cast_fp16 = mul(x = var_2811_cast_fp16, y = sin_11_to_fp16)[name = string("op_2812_cast_fp16")]; + tensor mh_q_75_cast_fp16 = add(x = var_2796_cast_fp16, y = var_2812_cast_fp16)[name = string("mh_q_75_cast_fp16")]; + tensor var_2814_cast_fp16 = mul(x = mh_k_73_cast_fp16, y = cos_11_to_fp16)[name = string("op_2814_cast_fp16")]; + tensor var_2819_begin_0 = const()[name = string("op_2819_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2819_end_0 = const()[name = string("op_2819_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_2819_end_mask_0 = const()[name = string("op_2819_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_2819_cast_fp16 = slice_by_index(begin = var_2819_begin_0, end = var_2819_end_0, end_mask = var_2819_end_mask_0, x = mh_k_73_cast_fp16)[name = string("op_2819_cast_fp16")]; + tensor var_2825_begin_0 = const()[name = string("op_2825_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_2825_end_0 = const()[name = string("op_2825_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_2825_end_mask_0 = const()[name = string("op_2825_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_2825_cast_fp16 = slice_by_index(begin = var_2825_begin_0, end = var_2825_end_0, end_mask = var_2825_end_mask_0, x = mh_k_73_cast_fp16)[name = string("op_2825_cast_fp16")]; + fp16 const_198_promoted_to_fp16 = const()[name = string("const_198_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2827_cast_fp16 = mul(x = var_2825_cast_fp16, y = const_198_promoted_to_fp16)[name = string("op_2827_cast_fp16")]; + bool var_2829_interleave_0 = const()[name = string("op_2829_interleave_0"), val = bool(false)]; + tensor var_2829_cast_fp16 = concat(axis = var_2695, interleave = var_2829_interleave_0, values = (var_2827_cast_fp16, var_2819_cast_fp16))[name = string("op_2829_cast_fp16")]; + tensor var_2830_cast_fp16 = mul(x = var_2829_cast_fp16, y = sin_11_to_fp16)[name = string("op_2830_cast_fp16")]; + tensor mh_k_75_cast_fp16 = add(x = var_2814_cast_fp16, y = var_2830_cast_fp16)[name = string("mh_k_75_cast_fp16")]; + tensor var_2834 = const()[name = string("op_2834"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_39_cast_fp16 = reshape(shape = var_2834, x = mh_k_75_cast_fp16)[name = string("current_key_39_cast_fp16")]; + tensor var_2840_to_fp16 = const()[name = string("op_2840_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274688)))]; + tensor var_2841_cast_fp16 = mul(x = obj_83_cast_fp16, y = var_2840_to_fp16)[name = string("op_2841_cast_fp16")]; + tensor var_2838_to_fp16 = const()[name = string("op_2838_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274816)))]; + tensor var_2842_cast_fp16 = mul(x = current_key_39_cast_fp16, y = var_2838_to_fp16)[name = string("op_2842_cast_fp16")]; + tensor key_39_cast_fp16 = add(x = var_2841_cast_fp16, y = var_2842_cast_fp16)[name = string("key_39_cast_fp16")]; + tensor var_2844_to_fp16 = const()[name = string("op_2844_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274688)))]; + tensor var_2845_cast_fp16 = mul(x = obj_85_cast_fp16, y = var_2844_to_fp16)[name = string("op_2845_cast_fp16")]; + tensor var_2846_cast_fp16 = mul(x = current_value_19_cast_fp16, y = var_2838_to_fp16)[name = string("op_2846_cast_fp16")]; + tensor value_19_cast_fp16 = add(x = var_2845_cast_fp16, y = var_2846_cast_fp16)[name = string("value_19_cast_fp16")]; + fp16 var_2853_to_fp16 = const()[name = string("op_2853_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_79_cast_fp16 = mul(x = mh_q_75_cast_fp16, y = var_2853_to_fp16)[name = string("mh_q_79_cast_fp16")]; + tensor var_2855 = const()[name = string("op_2855"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_77_cast_fp16 = reshape(shape = var_2855, x = key_39_cast_fp16)[name = string("mh_k_77_cast_fp16")]; + tensor var_2857 = const()[name = string("op_2857"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_37_cast_fp16 = reshape(shape = var_2857, x = value_19_cast_fp16)[name = string("mh_v_37_cast_fp16")]; + tensor transpose_36_perm_0 = const()[name = string("transpose_36_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_18_reps_0 = const()[name = string("tile_18_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_36_cast_fp16 = transpose(perm = transpose_36_perm_0, x = mh_k_77_cast_fp16)[name = string("transpose_425")]; + tensor tile_18_cast_fp16 = tile(reps = tile_18_reps_0, x = transpose_36_cast_fp16)[name = string("tile_18_cast_fp16")]; + tensor concat_44 = const()[name = string("concat_44"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_36_cast_fp16 = reshape(shape = concat_44, x = tile_18_cast_fp16)[name = string("reshape_36_cast_fp16")]; + tensor transpose_37_perm_0 = const()[name = string("transpose_37_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_45 = const()[name = string("concat_45"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_37_cast_fp16 = transpose(perm = transpose_37_perm_0, x = reshape_36_cast_fp16)[name = string("transpose_424")]; + tensor reshape_37_cast_fp16 = reshape(shape = concat_45, x = transpose_37_cast_fp16)[name = string("reshape_37_cast_fp16")]; + tensor transpose_38_perm_0 = const()[name = string("transpose_38_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_19_reps_0 = const()[name = string("tile_19_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_38_cast_fp16 = transpose(perm = transpose_38_perm_0, x = mh_v_37_cast_fp16)[name = string("transpose_423")]; + tensor tile_19_cast_fp16 = tile(reps = tile_19_reps_0, x = transpose_38_cast_fp16)[name = string("tile_19_cast_fp16")]; + tensor concat_46 = const()[name = string("concat_46"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_38_cast_fp16 = reshape(shape = concat_46, x = tile_19_cast_fp16)[name = string("reshape_38_cast_fp16")]; + tensor transpose_39_perm_0 = const()[name = string("transpose_39_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_47 = const()[name = string("concat_47"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_39_cast_fp16 = transpose(perm = transpose_39_perm_0, x = reshape_38_cast_fp16)[name = string("transpose_422")]; + tensor reshape_39_cast_fp16 = reshape(shape = concat_47, x = transpose_39_cast_fp16)[name = string("reshape_39_cast_fp16")]; + tensor transpose_353_perm_0 = const()[name = string("transpose_353_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_55_transpose_x_1 = const()[name = string("mh_w_55_transpose_x_1"), val = bool(true)]; + bool mh_w_55_transpose_y_1 = const()[name = string("mh_w_55_transpose_y_1"), val = bool(false)]; + tensor transpose_353_cast_fp16 = transpose(perm = transpose_353_perm_0, x = reshape_37_cast_fp16)[name = string("transpose_421")]; + tensor mh_w_55_cast_fp16 = matmul(transpose_x = mh_w_55_transpose_x_1, transpose_y = mh_w_55_transpose_y_1, x = mh_q_79_cast_fp16, y = transpose_353_cast_fp16)[name = string("mh_w_55_cast_fp16")]; + tensor var_2865_to_fp16 = const()[name = string("op_2865_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112274944)))]; + tensor mh_w_57_cast_fp16 = add(x = mh_w_55_cast_fp16, y = var_2865_to_fp16)[name = string("mh_w_57_cast_fp16")]; + tensor mh_w_59_cast_fp16 = softmax(axis = var_2685, x = mh_w_57_cast_fp16)[name = string("mh_w_59_cast_fp16")]; + tensor transpose_354_perm_0 = const()[name = string("transpose_354_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_19_transpose_x_1 = const()[name = string("attn_19_transpose_x_1"), val = bool(false)]; + bool attn_19_transpose_y_1 = const()[name = string("attn_19_transpose_y_1"), val = bool(true)]; + tensor transpose_354_cast_fp16 = transpose(perm = transpose_354_perm_0, x = reshape_39_cast_fp16)[name = string("transpose_420")]; + tensor attn_19_cast_fp16 = matmul(transpose_x = attn_19_transpose_x_1, transpose_y = attn_19_transpose_y_1, x = transpose_354_cast_fp16, y = mh_w_59_cast_fp16)[name = string("attn_19_cast_fp16")]; + tensor var_2871 = const()[name = string("op_2871"), val = tensor([1, 2048, 1, 1])]; + tensor input_73_cast_fp16 = reshape(shape = var_2871, x = attn_19_cast_fp16)[name = string("input_73_cast_fp16")]; + string obj_87_pad_type_0 = const()[name = string("obj_87_pad_type_0"), val = string("valid")]; + tensor obj_87_strides_0 = const()[name = string("obj_87_strides_0"), val = tensor([1, 1])]; + tensor obj_87_pad_0 = const()[name = string("obj_87_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_87_dilations_0 = const()[name = string("obj_87_dilations_0"), val = tensor([1, 1])]; + int32 obj_87_groups_0 = const()[name = string("obj_87_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69265472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(71362688))))[name = string("layers_4_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_87_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_87_dilations_0, groups = obj_87_groups_0, pad = obj_87_pad_0, pad_type = obj_87_pad_type_0, strides = obj_87_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_73_cast_fp16)[name = string("obj_87_cast_fp16")]; + tensor inputs_79_cast_fp16 = add(x = inputs_73_cast_fp16, y = obj_87_cast_fp16)[name = string("inputs_79_cast_fp16")]; + tensor inputs_sq_79_cast_fp16 = mul(x = inputs_79_cast_fp16, y = inputs_79_cast_fp16)[name = string("inputs_sq_79_cast_fp16")]; + tensor variance_79_axes_0 = const()[name = string("variance_79_axes_0"), val = tensor([1])]; + bool variance_79_keep_dims_0 = const()[name = string("variance_79_keep_dims_0"), val = bool(true)]; + tensor variance_79_cast_fp16 = reduce_mean(axes = variance_79_axes_0, keep_dims = variance_79_keep_dims_0, x = inputs_sq_79_cast_fp16)[name = string("variance_79_cast_fp16")]; + fp16 var_2889_to_fp16 = const()[name = string("op_2889_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2890_cast_fp16 = add(x = variance_79_cast_fp16, y = var_2889_to_fp16)[name = string("op_2890_cast_fp16")]; + fp32 var_2891_epsilon_0 = const()[name = string("op_2891_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2891_cast_fp16 = rsqrt(epsilon = var_2891_epsilon_0, x = var_2890_cast_fp16)[name = string("op_2891_cast_fp16")]; + tensor hidden_states_97_cast_fp16 = mul(x = inputs_79_cast_fp16, y = var_2891_cast_fp16)[name = string("hidden_states_97_cast_fp16")]; + tensor w_79_to_fp16 = const()[name = string("w_79_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(71363264)))]; + tensor input_75_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_97_cast_fp16)[name = string("input_75_cast_fp16")]; + string input_77_pad_type_0 = const()[name = string("input_77_pad_type_0"), val = string("valid")]; + tensor input_77_strides_0 = const()[name = string("input_77_strides_0"), val = tensor([1, 1])]; + tensor input_77_pad_0 = const()[name = string("input_77_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_77_dilations_0 = const()[name = string("input_77_dilations_0"), val = tensor([1, 1])]; + int32 input_77_groups_0 = const()[name = string("input_77_groups_0"), val = int32(1)]; + tensor layers_4_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(71365376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(74511168))))[name = string("layers_4_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_77_cast_fp16 = conv(dilations = input_77_dilations_0, groups = input_77_groups_0, pad = input_77_pad_0, pad_type = input_77_pad_type_0, strides = input_77_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_75_cast_fp16)[name = string("input_77_cast_fp16")]; + tensor var_2905_cast_fp16 = silu(x = input_77_cast_fp16)[name = string("op_2905_cast_fp16")]; + string var_2911_pad_type_0 = const()[name = string("op_2911_pad_type_0"), val = string("valid")]; + tensor var_2911_strides_0 = const()[name = string("op_2911_strides_0"), val = tensor([1, 1])]; + tensor var_2911_pad_0 = const()[name = string("op_2911_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2911_dilations_0 = const()[name = string("op_2911_dilations_0"), val = tensor([1, 1])]; + int32 var_2911_groups_0 = const()[name = string("op_2911_groups_0"), val = int32(1)]; + tensor layers_4_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(74511744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(77657536))))[name = string("layers_4_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_2911_cast_fp16 = conv(dilations = var_2911_dilations_0, groups = var_2911_groups_0, pad = var_2911_pad_0, pad_type = var_2911_pad_type_0, strides = var_2911_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_75_cast_fp16)[name = string("op_2911_cast_fp16")]; + tensor input_79_cast_fp16 = mul(x = var_2905_cast_fp16, y = var_2911_cast_fp16)[name = string("input_79_cast_fp16")]; + string hidden_states_99_pad_type_0 = const()[name = string("hidden_states_99_pad_type_0"), val = string("valid")]; + tensor hidden_states_99_strides_0 = const()[name = string("hidden_states_99_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_99_pad_0 = const()[name = string("hidden_states_99_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_99_dilations_0 = const()[name = string("hidden_states_99_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_99_groups_0 = const()[name = string("hidden_states_99_groups_0"), val = int32(1)]; + tensor layers_4_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(77658112))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80803904))))[name = string("layers_4_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_99_cast_fp16 = conv(dilations = hidden_states_99_dilations_0, groups = hidden_states_99_groups_0, pad = hidden_states_99_pad_0, pad_type = hidden_states_99_pad_type_0, strides = hidden_states_99_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_79_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; + tensor inputs_81_cast_fp16 = add(x = inputs_79_cast_fp16, y = hidden_states_99_cast_fp16)[name = string("inputs_81_cast_fp16")]; + int32 var_2939 = const()[name = string("op_2939"), val = int32(1)]; + bool key_caches_5_interleave_0 = const()[name = string("key_caches_5_interleave_0"), val = bool(false)]; + tensor key_caches_5_cast_fp16 = concat(axis = var_2939, interleave = key_caches_5_interleave_0, values = (key_23_cast_fp16, key_27_cast_fp16, key_31_cast_fp16, key_35_cast_fp16, key_39_cast_fp16))[name = string("key_caches_5_cast_fp16")]; + int32 var_2942 = const()[name = string("op_2942"), val = int32(1)]; + bool value_caches_5_interleave_0 = const()[name = string("value_caches_5_interleave_0"), val = bool(false)]; + tensor value_caches_5_cast_fp16 = concat(axis = var_2942, interleave = value_caches_5_interleave_0, values = (value_11_cast_fp16, value_13_cast_fp16, value_15_cast_fp16, value_17_cast_fp16, value_19_cast_fp16))[name = string("value_caches_5_cast_fp16")]; + tensor inputs_sq_81_cast_fp16 = mul(x = inputs_81_cast_fp16, y = inputs_81_cast_fp16)[name = string("inputs_sq_81_cast_fp16")]; + tensor variance_81_axes_0 = const()[name = string("variance_81_axes_0"), val = tensor([1])]; + bool variance_81_keep_dims_0 = const()[name = string("variance_81_keep_dims_0"), val = bool(true)]; + tensor variance_81_cast_fp16 = reduce_mean(axes = variance_81_axes_0, keep_dims = variance_81_keep_dims_0, x = inputs_sq_81_cast_fp16)[name = string("variance_81_cast_fp16")]; + fp16 var_2962_to_fp16 = const()[name = string("op_2962_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2963_cast_fp16 = add(x = variance_81_cast_fp16, y = var_2962_to_fp16)[name = string("op_2963_cast_fp16")]; + fp32 var_2964_epsilon_0 = const()[name = string("op_2964_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2964_cast_fp16 = rsqrt(epsilon = var_2964_epsilon_0, x = var_2963_cast_fp16)[name = string("op_2964_cast_fp16")]; + tensor hidden_states_101_cast_fp16 = mul(x = inputs_81_cast_fp16, y = var_2964_cast_fp16)[name = string("hidden_states_101_cast_fp16")]; + tensor w_81_to_fp16 = const()[name = string("w_81_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80804480)))]; + tensor input_81_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_101_cast_fp16)[name = string("input_81_cast_fp16")]; + string logits_1_pad_type_0 = const()[name = string("logits_1_pad_type_0"), val = string("valid")]; + tensor logits_1_strides_0 = const()[name = string("logits_1_strides_0"), val = tensor([1, 1])]; + tensor logits_1_pad_0 = const()[name = string("logits_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_1_dilations_0 = const()[name = string("logits_1_dilations_0"), val = tensor([1, 1])]; + int32 logits_1_groups_0 = const()[name = string("logits_1_groups_0"), val = int32(1)]; + tensor lm_heads_0_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80806592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82903808))))[name = string("lm_heads_0_weight_to_fp16_palettized")]; + tensor logits_1_cast_fp16 = conv(dilations = logits_1_dilations_0, groups = logits_1_groups_0, pad = logits_1_pad_0, pad_type = logits_1_pad_type_0, strides = logits_1_strides_0, weight = lm_heads_0_weight_to_fp16_palettized, x = input_81_cast_fp16)[name = string("logits_1_cast_fp16")]; + tensor var_2982 = const()[name = string("op_2982"), val = tensor([1, 2048])]; + tensor logits_3_cast_fp16 = reshape(shape = var_2982, x = logits_1_cast_fp16)[name = string("logits_3_cast_fp16")]; + tensor scaled_logits_1_cast_fp16 = real_div(x = logits_3_cast_fp16, y = temperature)[name = string("scaled_logits_1_cast_fp16")]; + int32 var_2992 = const()[name = string("op_2992"), val = int32(100)]; + int32 top_values_1_axis_0 = const()[name = string("top_values_1_axis_0"), val = int32(1)]; + bool top_values_1_ascending_0 = const()[name = string("top_values_1_ascending_0"), val = bool(false)]; + bool top_values_1_sort_0 = const()[name = string("top_values_1_sort_0"), val = bool(true)]; + bool top_values_1_return_indices_0 = const()[name = string("top_values_1_return_indices_0"), val = bool(true)]; + string top_values_1_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_1_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_1_cast_fp16_cast_uint16_0, tensor top_values_1_cast_fp16_cast_uint16_1 = topk(ascending = top_values_1_ascending_0, axis = top_values_1_axis_0, k = var_2992, output_indices_dtype = top_values_1_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_1_return_indices_0, sort = top_values_1_sort_0, x = scaled_logits_1_cast_fp16)[name = string("top_values_1_cast_fp16_cast_uint16")]; + tensor top_k_mask_candidate_ranks_to_fp16 = const()[name = string("top_k_mask_candidate_ranks_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112275072)))]; + string top_k_float_to_fp16_dtype_0 = const()[name = string("top_k_float_to_fp16_dtype_0"), val = string("fp16")]; + tensor top_k_to_fp16 = cast(dtype = top_k_float_to_fp16_dtype_0, x = top_k)[name = string("cast_16")]; + tensor var_2996_cast_fp16 = less(x = top_k_mask_candidate_ranks_to_fp16, y = top_k_to_fp16)[name = string("op_2996_cast_fp16")]; + string candidate_mask_1_to_fp16_dtype_0 = const()[name = string("candidate_mask_1_to_fp16_dtype_0"), val = string("fp16")]; + tensor var_2996_cast_fp16_to_fp16 = cast(dtype = candidate_mask_1_to_fp16_dtype_0, x = var_2996_cast_fp16)[name = string("cast_15")]; + tensor var_2998_cast_fp16 = mul(x = top_values_1_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_2998_cast_fp16")]; + fp16 var_2986_to_fp16 = const()[name = string("op_2986_to_fp16"), val = fp16(0x1p+0)]; + tensor var_2999_cast_fp16 = sub(x = var_2986_to_fp16, y = var_2996_cast_fp16_to_fp16)[name = string("op_2999_cast_fp16")]; + fp16 var_3000_to_fp16 = const()[name = string("op_3000_to_fp16"), val = fp16(0x1.d4cp+14)]; + tensor var_3001_cast_fp16 = mul(x = var_2999_cast_fp16, y = var_3000_to_fp16)[name = string("op_3001_cast_fp16")]; + tensor var_3002_cast_fp16 = add(x = var_2998_cast_fp16, y = var_3001_cast_fp16)[name = string("op_3002_cast_fp16")]; + tensor reduce_min_0_axes_0 = const()[name = string("reduce_min_0_axes_0"), val = tensor([1])]; + bool reduce_min_0_keep_dims_0 = const()[name = string("reduce_min_0_keep_dims_0"), val = bool(true)]; + tensor reduce_min_0_cast_fp16 = reduce_min(axes = reduce_min_0_axes_0, keep_dims = reduce_min_0_keep_dims_0, x = var_3002_cast_fp16)[name = string("reduce_min_0_cast_fp16")]; + tensor var_3005_cast_fp16 = greater_equal(x = scaled_logits_1_cast_fp16, y = reduce_min_0_cast_fp16)[name = string("op_3005_cast_fp16")]; + fp16 var_3006_value_0_to_fp16 = const()[name = string("op_3006_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_3006_cast_fp16 = fill_like(ref_tensor = scaled_logits_1_cast_fp16, value = var_3006_value_0_to_fp16)[name = string("op_3006_cast_fp16")]; + tensor masked_logits_1_cast_fp16 = select(a = scaled_logits_1_cast_fp16, b = var_3006_cast_fp16, cond = var_3005_cast_fp16)[name = string("masked_logits_1_cast_fp16")]; + tensor var_3010_begin_0 = const()[name = string("op_3010_begin_0"), val = tensor([0, 0])]; + tensor var_3010_end_0 = const()[name = string("op_3010_end_0"), val = tensor([1, 2048])]; + tensor var_3010_end_mask_0 = const()[name = string("op_3010_end_mask_0"), val = tensor([false, true])]; + tensor var_3010_squeeze_mask_0 = const()[name = string("op_3010_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_3010_cast_fp16 = slice_by_index(begin = var_3010_begin_0, end = var_3010_end_0, end_mask = var_3010_end_mask_0, squeeze_mask = var_3010_squeeze_mask_0, x = gumbel)[name = string("op_3010_cast_fp16")]; + tensor var_3013 = const()[name = string("op_3013"), val = tensor([1, 2048])]; + tensor var_3014_cast_fp16 = reshape(shape = var_3013, x = var_3010_cast_fp16)[name = string("op_3014_cast_fp16")]; + tensor noisy_logits_1_cast_fp16 = add(x = masked_logits_1_cast_fp16, y = var_3014_cast_fp16)[name = string("noisy_logits_1_cast_fp16")]; + int32 code_1_axis_0 = const()[name = string("code_1_axis_0"), val = int32(1)]; + bool code_1_keep_dims_0 = const()[name = string("code_1_keep_dims_0"), val = bool(false)]; + string code_1_output_dtype_0 = const()[name = string("code_1_output_dtype_0"), val = string("int32")]; + tensor code_1_cast_fp16 = reduce_argmax(axis = code_1_axis_0, keep_dims = code_1_keep_dims_0, output_dtype = code_1_output_dtype_0, x = noisy_logits_1_cast_fp16)[name = string("code_1_cast_fp16")]; + int32 code_embed_1_axis_0 = const()[name = string("code_embed_1_axis_0"), val = int32(0)]; + int32 code_embed_1_batch_dims_0 = const()[name = string("code_embed_1_batch_dims_0"), val = int32(0)]; + bool code_embed_1_validate_indices_0 = const()[name = string("code_embed_1_validate_indices_0"), val = bool(false)]; + tensor codec_embedding_embedding_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112275392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175190016))))[name = string("codec_embedding_embedding_weight_to_fp16_palettized")]; + string code_1_cast_fp16_to_uint16_dtype_0 = const()[name = string("code_1_cast_fp16_to_uint16_dtype_0"), val = string("uint16")]; + tensor code_1_cast_fp16_to_uint16 = cast(dtype = code_1_cast_fp16_to_uint16_dtype_0, x = code_1_cast_fp16)[name = string("cast_14")]; + tensor code_embed_1_cast_fp16_cast_uint16 = gather(axis = code_embed_1_axis_0, batch_dims = code_embed_1_batch_dims_0, indices = code_1_cast_fp16_to_uint16, validate_indices = code_embed_1_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_1_cast_fp16_cast_uint16")]; + tensor var_3029 = const()[name = string("op_3029"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_3_cast_fp16 = reshape(shape = var_3029, x = code_embed_1_cast_fp16_cast_uint16)[name = string("code_embed_3_cast_fp16")]; + string inputs_83_pad_type_0 = const()[name = string("inputs_83_pad_type_0"), val = string("valid")]; + tensor inputs_83_strides_0 = const()[name = string("inputs_83_strides_0"), val = tensor([1, 1])]; + tensor inputs_83_pad_0 = const()[name = string("inputs_83_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_83_dilations_0 = const()[name = string("inputs_83_dilations_0"), val = tensor([1, 1])]; + int32 inputs_83_groups_0 = const()[name = string("inputs_83_groups_0"), val = int32(1)]; + tensor inputs_83_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_83_dilations_0, groups = inputs_83_groups_0, pad = inputs_83_pad_0, pad_type = inputs_83_pad_type_0, strides = inputs_83_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_3_cast_fp16)[name = string("inputs_83_cast_fp16")]; + tensor obj_91_begin_0 = const()[name = string("obj_91_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_91_end_0 = const()[name = string("obj_91_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_91_end_mask_0 = const()[name = string("obj_91_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_91_cast_fp16 = slice_by_index(begin = obj_91_begin_0, end = obj_91_end_0, end_mask = obj_91_end_mask_0, x = key_caches_5_cast_fp16)[name = string("obj_91_cast_fp16")]; + tensor obj_93_begin_0 = const()[name = string("obj_93_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_93_end_0 = const()[name = string("obj_93_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_93_end_mask_0 = const()[name = string("obj_93_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_93_cast_fp16 = slice_by_index(begin = obj_93_begin_0, end = obj_93_end_0, end_mask = obj_93_end_mask_0, x = value_caches_5_cast_fp16)[name = string("obj_93_cast_fp16")]; + int32 var_3134 = const()[name = string("op_3134"), val = int32(3)]; + int32 var_3144 = const()[name = string("op_3144"), val = int32(-2)]; + tensor inputs_sq_83_cast_fp16 = mul(x = inputs_83_cast_fp16, y = inputs_83_cast_fp16)[name = string("inputs_sq_83_cast_fp16")]; + tensor variance_83_axes_0 = const()[name = string("variance_83_axes_0"), val = tensor([1])]; + bool variance_83_keep_dims_0 = const()[name = string("variance_83_keep_dims_0"), val = bool(true)]; + tensor variance_83_cast_fp16 = reduce_mean(axes = variance_83_axes_0, keep_dims = variance_83_keep_dims_0, x = inputs_sq_83_cast_fp16)[name = string("variance_83_cast_fp16")]; + fp16 var_3158_to_fp16 = const()[name = string("op_3158_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3159_cast_fp16 = add(x = variance_83_cast_fp16, y = var_3158_to_fp16)[name = string("op_3159_cast_fp16")]; + fp32 var_3160_epsilon_0 = const()[name = string("op_3160_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3160_cast_fp16 = rsqrt(epsilon = var_3160_epsilon_0, x = var_3159_cast_fp16)[name = string("op_3160_cast_fp16")]; + tensor hidden_states_103_cast_fp16 = mul(x = inputs_83_cast_fp16, y = var_3160_cast_fp16)[name = string("hidden_states_103_cast_fp16")]; + tensor obj_89_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_103_cast_fp16)[name = string("obj_89_cast_fp16")]; + string query_61_pad_type_0 = const()[name = string("query_61_pad_type_0"), val = string("valid")]; + tensor query_61_strides_0 = const()[name = string("query_61_strides_0"), val = tensor([1, 1])]; + tensor query_61_pad_0 = const()[name = string("query_61_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_61_dilations_0 = const()[name = string("query_61_dilations_0"), val = tensor([1, 1])]; + int32 query_61_groups_0 = const()[name = string("query_61_groups_0"), val = int32(1)]; + tensor query_61_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_61_dilations_0, groups = query_61_groups_0, pad = query_61_pad_0, pad_type = query_61_pad_type_0, strides = query_61_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_89_cast_fp16)[name = string("query_61_cast_fp16")]; + string current_key_41_pad_type_0 = const()[name = string("current_key_41_pad_type_0"), val = string("valid")]; + tensor current_key_41_strides_0 = const()[name = string("current_key_41_strides_0"), val = tensor([1, 1])]; + tensor current_key_41_pad_0 = const()[name = string("current_key_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_41_dilations_0 = const()[name = string("current_key_41_dilations_0"), val = tensor([1, 1])]; + int32 current_key_41_groups_0 = const()[name = string("current_key_41_groups_0"), val = int32(1)]; + tensor current_key_41_cast_fp16 = conv(dilations = current_key_41_dilations_0, groups = current_key_41_groups_0, pad = current_key_41_pad_0, pad_type = current_key_41_pad_type_0, strides = current_key_41_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_89_cast_fp16)[name = string("current_key_41_cast_fp16")]; + string current_value_21_pad_type_0 = const()[name = string("current_value_21_pad_type_0"), val = string("valid")]; + tensor current_value_21_strides_0 = const()[name = string("current_value_21_strides_0"), val = tensor([1, 1])]; + tensor current_value_21_pad_0 = const()[name = string("current_value_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_21_dilations_0 = const()[name = string("current_value_21_dilations_0"), val = tensor([1, 1])]; + int32 current_value_21_groups_0 = const()[name = string("current_value_21_groups_0"), val = int32(1)]; + tensor current_value_21_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_21_dilations_0, groups = current_value_21_groups_0, pad = current_value_21_pad_0, pad_type = current_value_21_pad_type_0, strides = current_value_21_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_89_cast_fp16)[name = string("current_value_21_cast_fp16")]; + tensor var_3197 = const()[name = string("op_3197"), val = tensor([16, 128, 1, 1])]; + tensor inputs_85_cast_fp16 = reshape(shape = var_3197, x = query_61_cast_fp16)[name = string("inputs_85_cast_fp16")]; + tensor inputs_sq_85_cast_fp16 = mul(x = inputs_85_cast_fp16, y = inputs_85_cast_fp16)[name = string("inputs_sq_85_cast_fp16")]; + tensor variance_85_axes_0 = const()[name = string("variance_85_axes_0"), val = tensor([1])]; + bool variance_85_keep_dims_0 = const()[name = string("variance_85_keep_dims_0"), val = bool(true)]; + tensor variance_85_cast_fp16 = reduce_mean(axes = variance_85_axes_0, keep_dims = variance_85_keep_dims_0, x = inputs_sq_85_cast_fp16)[name = string("variance_85_cast_fp16")]; + fp16 var_3203_to_fp16 = const()[name = string("op_3203_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3204_cast_fp16 = add(x = variance_85_cast_fp16, y = var_3203_to_fp16)[name = string("op_3204_cast_fp16")]; + fp32 var_3205_epsilon_0 = const()[name = string("op_3205_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3205_cast_fp16 = rsqrt(epsilon = var_3205_epsilon_0, x = var_3204_cast_fp16)[name = string("op_3205_cast_fp16")]; + tensor hidden_states_105_cast_fp16 = mul(x = inputs_85_cast_fp16, y = var_3205_cast_fp16)[name = string("hidden_states_105_cast_fp16")]; + tensor query_normed_21_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_105_cast_fp16)[name = string("query_normed_21_cast_fp16")]; + tensor var_3213 = const()[name = string("op_3213"), val = tensor([8, 128, 1, 1])]; + tensor inputs_87_cast_fp16 = reshape(shape = var_3213, x = current_key_41_cast_fp16)[name = string("inputs_87_cast_fp16")]; + tensor inputs_sq_87_cast_fp16 = mul(x = inputs_87_cast_fp16, y = inputs_87_cast_fp16)[name = string("inputs_sq_87_cast_fp16")]; + tensor variance_87_axes_0 = const()[name = string("variance_87_axes_0"), val = tensor([1])]; + bool variance_87_keep_dims_0 = const()[name = string("variance_87_keep_dims_0"), val = bool(true)]; + tensor variance_87_cast_fp16 = reduce_mean(axes = variance_87_axes_0, keep_dims = variance_87_keep_dims_0, x = inputs_sq_87_cast_fp16)[name = string("variance_87_cast_fp16")]; + fp16 var_3219_to_fp16 = const()[name = string("op_3219_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3220_cast_fp16 = add(x = variance_87_cast_fp16, y = var_3219_to_fp16)[name = string("op_3220_cast_fp16")]; + fp32 var_3221_epsilon_0 = const()[name = string("op_3221_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3221_cast_fp16 = rsqrt(epsilon = var_3221_epsilon_0, x = var_3220_cast_fp16)[name = string("op_3221_cast_fp16")]; + tensor hidden_states_107_cast_fp16 = mul(x = inputs_87_cast_fp16, y = var_3221_cast_fp16)[name = string("hidden_states_107_cast_fp16")]; + tensor current_key_normed_21_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_107_cast_fp16)[name = string("current_key_normed_21_cast_fp16")]; + tensor var_3239 = const()[name = string("op_3239"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_81_cast_fp16 = reshape(shape = var_3239, x = query_normed_21_cast_fp16)[name = string("mh_q_81_cast_fp16")]; + tensor var_3241 = const()[name = string("op_3241"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_81_cast_fp16 = reshape(shape = var_3241, x = current_key_normed_21_cast_fp16)[name = string("mh_k_81_cast_fp16")]; + tensor cos_21_to_fp16 = const()[name = string("cos_21_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175190592)))]; + tensor var_3245_cast_fp16 = mul(x = mh_q_81_cast_fp16, y = cos_21_to_fp16)[name = string("op_3245_cast_fp16")]; + tensor var_3250_begin_0 = const()[name = string("op_3250_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3250_end_0 = const()[name = string("op_3250_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_3250_end_mask_0 = const()[name = string("op_3250_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_3250_cast_fp16 = slice_by_index(begin = var_3250_begin_0, end = var_3250_end_0, end_mask = var_3250_end_mask_0, x = mh_q_81_cast_fp16)[name = string("op_3250_cast_fp16")]; + tensor var_3256_begin_0 = const()[name = string("op_3256_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_3256_end_0 = const()[name = string("op_3256_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_3256_end_mask_0 = const()[name = string("op_3256_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_3256_cast_fp16 = slice_by_index(begin = var_3256_begin_0, end = var_3256_end_0, end_mask = var_3256_end_mask_0, x = mh_q_81_cast_fp16)[name = string("op_3256_cast_fp16")]; + fp16 const_216_promoted_to_fp16 = const()[name = string("const_216_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3258_cast_fp16 = mul(x = var_3256_cast_fp16, y = const_216_promoted_to_fp16)[name = string("op_3258_cast_fp16")]; + bool var_3260_interleave_0 = const()[name = string("op_3260_interleave_0"), val = bool(false)]; + tensor var_3260_cast_fp16 = concat(axis = var_3144, interleave = var_3260_interleave_0, values = (var_3258_cast_fp16, var_3250_cast_fp16))[name = string("op_3260_cast_fp16")]; + tensor sin_21_to_fp16 = const()[name = string("sin_21_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175190912)))]; + tensor var_3261_cast_fp16 = mul(x = var_3260_cast_fp16, y = sin_21_to_fp16)[name = string("op_3261_cast_fp16")]; + tensor mh_q_83_cast_fp16 = add(x = var_3245_cast_fp16, y = var_3261_cast_fp16)[name = string("mh_q_83_cast_fp16")]; + tensor var_3263_cast_fp16 = mul(x = mh_k_81_cast_fp16, y = cos_21_to_fp16)[name = string("op_3263_cast_fp16")]; + tensor var_3268_begin_0 = const()[name = string("op_3268_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3268_end_0 = const()[name = string("op_3268_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_3268_end_mask_0 = const()[name = string("op_3268_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_3268_cast_fp16 = slice_by_index(begin = var_3268_begin_0, end = var_3268_end_0, end_mask = var_3268_end_mask_0, x = mh_k_81_cast_fp16)[name = string("op_3268_cast_fp16")]; + tensor var_3274_begin_0 = const()[name = string("op_3274_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_3274_end_0 = const()[name = string("op_3274_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_3274_end_mask_0 = const()[name = string("op_3274_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_3274_cast_fp16 = slice_by_index(begin = var_3274_begin_0, end = var_3274_end_0, end_mask = var_3274_end_mask_0, x = mh_k_81_cast_fp16)[name = string("op_3274_cast_fp16")]; + fp16 const_219_promoted_to_fp16 = const()[name = string("const_219_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3276_cast_fp16 = mul(x = var_3274_cast_fp16, y = const_219_promoted_to_fp16)[name = string("op_3276_cast_fp16")]; + bool var_3278_interleave_0 = const()[name = string("op_3278_interleave_0"), val = bool(false)]; + tensor var_3278_cast_fp16 = concat(axis = var_3144, interleave = var_3278_interleave_0, values = (var_3276_cast_fp16, var_3268_cast_fp16))[name = string("op_3278_cast_fp16")]; + tensor var_3279_cast_fp16 = mul(x = var_3278_cast_fp16, y = sin_21_to_fp16)[name = string("op_3279_cast_fp16")]; + tensor mh_k_83_cast_fp16 = add(x = var_3263_cast_fp16, y = var_3279_cast_fp16)[name = string("mh_k_83_cast_fp16")]; + tensor var_3283 = const()[name = string("op_3283"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_43_cast_fp16 = reshape(shape = var_3283, x = mh_k_83_cast_fp16)[name = string("current_key_43_cast_fp16")]; + tensor var_3289_to_fp16 = const()[name = string("op_3289_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191232)))]; + tensor var_3290_cast_fp16 = mul(x = obj_91_cast_fp16, y = var_3289_to_fp16)[name = string("op_3290_cast_fp16")]; + tensor var_3287_to_fp16 = const()[name = string("op_3287_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191360)))]; + tensor var_3291_cast_fp16 = mul(x = current_key_43_cast_fp16, y = var_3287_to_fp16)[name = string("op_3291_cast_fp16")]; + tensor key_43_cast_fp16 = add(x = var_3290_cast_fp16, y = var_3291_cast_fp16)[name = string("key_43_cast_fp16")]; + tensor var_3293_to_fp16 = const()[name = string("op_3293_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191232)))]; + tensor var_3294_cast_fp16 = mul(x = obj_93_cast_fp16, y = var_3293_to_fp16)[name = string("op_3294_cast_fp16")]; + tensor var_3295_cast_fp16 = mul(x = current_value_21_cast_fp16, y = var_3287_to_fp16)[name = string("op_3295_cast_fp16")]; + tensor value_21_cast_fp16 = add(x = var_3294_cast_fp16, y = var_3295_cast_fp16)[name = string("value_21_cast_fp16")]; + fp16 var_3302_to_fp16 = const()[name = string("op_3302_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_87_cast_fp16 = mul(x = mh_q_83_cast_fp16, y = var_3302_to_fp16)[name = string("mh_q_87_cast_fp16")]; + tensor var_3304 = const()[name = string("op_3304"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_85_cast_fp16 = reshape(shape = var_3304, x = key_43_cast_fp16)[name = string("mh_k_85_cast_fp16")]; + tensor var_3306 = const()[name = string("op_3306"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_41_cast_fp16 = reshape(shape = var_3306, x = value_21_cast_fp16)[name = string("mh_v_41_cast_fp16")]; + tensor transpose_40_perm_0 = const()[name = string("transpose_40_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_20_reps_0 = const()[name = string("tile_20_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_40_cast_fp16 = transpose(perm = transpose_40_perm_0, x = mh_k_85_cast_fp16)[name = string("transpose_419")]; + tensor tile_20_cast_fp16 = tile(reps = tile_20_reps_0, x = transpose_40_cast_fp16)[name = string("tile_20_cast_fp16")]; + tensor concat_53 = const()[name = string("concat_53"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_40_cast_fp16 = reshape(shape = concat_53, x = tile_20_cast_fp16)[name = string("reshape_40_cast_fp16")]; + tensor transpose_41_perm_0 = const()[name = string("transpose_41_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_54 = const()[name = string("concat_54"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_41_cast_fp16 = transpose(perm = transpose_41_perm_0, x = reshape_40_cast_fp16)[name = string("transpose_418")]; + tensor reshape_41_cast_fp16 = reshape(shape = concat_54, x = transpose_41_cast_fp16)[name = string("reshape_41_cast_fp16")]; + tensor transpose_42_perm_0 = const()[name = string("transpose_42_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_21_reps_0 = const()[name = string("tile_21_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_42_cast_fp16 = transpose(perm = transpose_42_perm_0, x = mh_v_41_cast_fp16)[name = string("transpose_417")]; + tensor tile_21_cast_fp16 = tile(reps = tile_21_reps_0, x = transpose_42_cast_fp16)[name = string("tile_21_cast_fp16")]; + tensor concat_55 = const()[name = string("concat_55"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_42_cast_fp16 = reshape(shape = concat_55, x = tile_21_cast_fp16)[name = string("reshape_42_cast_fp16")]; + tensor transpose_43_perm_0 = const()[name = string("transpose_43_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_56 = const()[name = string("concat_56"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_43_cast_fp16 = transpose(perm = transpose_43_perm_0, x = reshape_42_cast_fp16)[name = string("transpose_416")]; + tensor reshape_43_cast_fp16 = reshape(shape = concat_56, x = transpose_43_cast_fp16)[name = string("reshape_43_cast_fp16")]; + tensor transpose_357_perm_0 = const()[name = string("transpose_357_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_61_transpose_x_1 = const()[name = string("mh_w_61_transpose_x_1"), val = bool(true)]; + bool mh_w_61_transpose_y_1 = const()[name = string("mh_w_61_transpose_y_1"), val = bool(false)]; + tensor transpose_357_cast_fp16 = transpose(perm = transpose_357_perm_0, x = reshape_41_cast_fp16)[name = string("transpose_415")]; + tensor mh_w_61_cast_fp16 = matmul(transpose_x = mh_w_61_transpose_x_1, transpose_y = mh_w_61_transpose_y_1, x = mh_q_87_cast_fp16, y = transpose_357_cast_fp16)[name = string("mh_w_61_cast_fp16")]; + tensor var_3314_to_fp16 = const()[name = string("op_3314_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191488)))]; + tensor mh_w_63_cast_fp16 = add(x = mh_w_61_cast_fp16, y = var_3314_to_fp16)[name = string("mh_w_63_cast_fp16")]; + tensor mh_w_65_cast_fp16 = softmax(axis = var_3134, x = mh_w_63_cast_fp16)[name = string("mh_w_65_cast_fp16")]; + tensor transpose_358_perm_0 = const()[name = string("transpose_358_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_21_transpose_x_1 = const()[name = string("attn_21_transpose_x_1"), val = bool(false)]; + bool attn_21_transpose_y_1 = const()[name = string("attn_21_transpose_y_1"), val = bool(true)]; + tensor transpose_358_cast_fp16 = transpose(perm = transpose_358_perm_0, x = reshape_43_cast_fp16)[name = string("transpose_414")]; + tensor attn_21_cast_fp16 = matmul(transpose_x = attn_21_transpose_x_1, transpose_y = attn_21_transpose_y_1, x = transpose_358_cast_fp16, y = mh_w_65_cast_fp16)[name = string("attn_21_cast_fp16")]; + tensor var_3320 = const()[name = string("op_3320"), val = tensor([1, 2048, 1, 1])]; + tensor input_85_cast_fp16 = reshape(shape = var_3320, x = attn_21_cast_fp16)[name = string("input_85_cast_fp16")]; + string obj_99_pad_type_0 = const()[name = string("obj_99_pad_type_0"), val = string("valid")]; + tensor obj_99_strides_0 = const()[name = string("obj_99_strides_0"), val = tensor([1, 1])]; + tensor obj_99_pad_0 = const()[name = string("obj_99_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_99_dilations_0 = const()[name = string("obj_99_dilations_0"), val = tensor([1, 1])]; + int32 obj_99_groups_0 = const()[name = string("obj_99_groups_0"), val = int32(1)]; + tensor obj_99_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_99_dilations_0, groups = obj_99_groups_0, pad = obj_99_pad_0, pad_type = obj_99_pad_type_0, strides = obj_99_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_85_cast_fp16)[name = string("obj_99_cast_fp16")]; + tensor inputs_89_cast_fp16 = add(x = inputs_83_cast_fp16, y = obj_99_cast_fp16)[name = string("inputs_89_cast_fp16")]; + tensor inputs_sq_89_cast_fp16 = mul(x = inputs_89_cast_fp16, y = inputs_89_cast_fp16)[name = string("inputs_sq_89_cast_fp16")]; + tensor variance_89_axes_0 = const()[name = string("variance_89_axes_0"), val = tensor([1])]; + bool variance_89_keep_dims_0 = const()[name = string("variance_89_keep_dims_0"), val = bool(true)]; + tensor variance_89_cast_fp16 = reduce_mean(axes = variance_89_axes_0, keep_dims = variance_89_keep_dims_0, x = inputs_sq_89_cast_fp16)[name = string("variance_89_cast_fp16")]; + fp16 var_3338_to_fp16 = const()[name = string("op_3338_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3339_cast_fp16 = add(x = variance_89_cast_fp16, y = var_3338_to_fp16)[name = string("op_3339_cast_fp16")]; + fp32 var_3340_epsilon_0 = const()[name = string("op_3340_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3340_cast_fp16 = rsqrt(epsilon = var_3340_epsilon_0, x = var_3339_cast_fp16)[name = string("op_3340_cast_fp16")]; + tensor hidden_states_109_cast_fp16 = mul(x = inputs_89_cast_fp16, y = var_3340_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; + tensor input_87_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_109_cast_fp16)[name = string("input_87_cast_fp16")]; + string input_89_pad_type_0 = const()[name = string("input_89_pad_type_0"), val = string("valid")]; + tensor input_89_strides_0 = const()[name = string("input_89_strides_0"), val = tensor([1, 1])]; + tensor input_89_pad_0 = const()[name = string("input_89_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_89_dilations_0 = const()[name = string("input_89_dilations_0"), val = tensor([1, 1])]; + int32 input_89_groups_0 = const()[name = string("input_89_groups_0"), val = int32(1)]; + tensor input_89_cast_fp16 = conv(dilations = input_89_dilations_0, groups = input_89_groups_0, pad = input_89_pad_0, pad_type = input_89_pad_type_0, strides = input_89_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_87_cast_fp16)[name = string("input_89_cast_fp16")]; + tensor var_3354_cast_fp16 = silu(x = input_89_cast_fp16)[name = string("op_3354_cast_fp16")]; + string var_3360_pad_type_0 = const()[name = string("op_3360_pad_type_0"), val = string("valid")]; + tensor var_3360_strides_0 = const()[name = string("op_3360_strides_0"), val = tensor([1, 1])]; + tensor var_3360_pad_0 = const()[name = string("op_3360_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3360_dilations_0 = const()[name = string("op_3360_dilations_0"), val = tensor([1, 1])]; + int32 var_3360_groups_0 = const()[name = string("op_3360_groups_0"), val = int32(1)]; + tensor var_3360_cast_fp16 = conv(dilations = var_3360_dilations_0, groups = var_3360_groups_0, pad = var_3360_pad_0, pad_type = var_3360_pad_type_0, strides = var_3360_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_87_cast_fp16)[name = string("op_3360_cast_fp16")]; + tensor input_91_cast_fp16 = mul(x = var_3354_cast_fp16, y = var_3360_cast_fp16)[name = string("input_91_cast_fp16")]; + string hidden_states_111_pad_type_0 = const()[name = string("hidden_states_111_pad_type_0"), val = string("valid")]; + tensor hidden_states_111_strides_0 = const()[name = string("hidden_states_111_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_111_pad_0 = const()[name = string("hidden_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_111_dilations_0 = const()[name = string("hidden_states_111_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_111_groups_0 = const()[name = string("hidden_states_111_groups_0"), val = int32(1)]; + tensor hidden_states_111_cast_fp16 = conv(dilations = hidden_states_111_dilations_0, groups = hidden_states_111_groups_0, pad = hidden_states_111_pad_0, pad_type = hidden_states_111_pad_type_0, strides = hidden_states_111_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_91_cast_fp16)[name = string("hidden_states_111_cast_fp16")]; + tensor inputs_91_cast_fp16 = add(x = inputs_89_cast_fp16, y = hidden_states_111_cast_fp16)[name = string("inputs_91_cast_fp16")]; + tensor obj_103_begin_0 = const()[name = string("obj_103_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_103_end_0 = const()[name = string("obj_103_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_103_end_mask_0 = const()[name = string("obj_103_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_103_cast_fp16 = slice_by_index(begin = obj_103_begin_0, end = obj_103_end_0, end_mask = obj_103_end_mask_0, x = key_caches_5_cast_fp16)[name = string("obj_103_cast_fp16")]; + tensor obj_105_begin_0 = const()[name = string("obj_105_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_105_end_0 = const()[name = string("obj_105_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_105_end_mask_0 = const()[name = string("obj_105_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_105_cast_fp16 = slice_by_index(begin = obj_105_begin_0, end = obj_105_end_0, end_mask = obj_105_end_mask_0, x = value_caches_5_cast_fp16)[name = string("obj_105_cast_fp16")]; + int32 var_3408 = const()[name = string("op_3408"), val = int32(3)]; + int32 var_3418 = const()[name = string("op_3418"), val = int32(-2)]; + tensor inputs_sq_91_cast_fp16 = mul(x = inputs_91_cast_fp16, y = inputs_91_cast_fp16)[name = string("inputs_sq_91_cast_fp16")]; + tensor variance_91_axes_0 = const()[name = string("variance_91_axes_0"), val = tensor([1])]; + bool variance_91_keep_dims_0 = const()[name = string("variance_91_keep_dims_0"), val = bool(true)]; + tensor variance_91_cast_fp16 = reduce_mean(axes = variance_91_axes_0, keep_dims = variance_91_keep_dims_0, x = inputs_sq_91_cast_fp16)[name = string("variance_91_cast_fp16")]; + fp16 var_3432_to_fp16 = const()[name = string("op_3432_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3433_cast_fp16 = add(x = variance_91_cast_fp16, y = var_3432_to_fp16)[name = string("op_3433_cast_fp16")]; + fp32 var_3434_epsilon_0 = const()[name = string("op_3434_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3434_cast_fp16 = rsqrt(epsilon = var_3434_epsilon_0, x = var_3433_cast_fp16)[name = string("op_3434_cast_fp16")]; + tensor hidden_states_113_cast_fp16 = mul(x = inputs_91_cast_fp16, y = var_3434_cast_fp16)[name = string("hidden_states_113_cast_fp16")]; + tensor obj_101_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_113_cast_fp16)[name = string("obj_101_cast_fp16")]; + string query_67_pad_type_0 = const()[name = string("query_67_pad_type_0"), val = string("valid")]; + tensor query_67_strides_0 = const()[name = string("query_67_strides_0"), val = tensor([1, 1])]; + tensor query_67_pad_0 = const()[name = string("query_67_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_67_dilations_0 = const()[name = string("query_67_dilations_0"), val = tensor([1, 1])]; + int32 query_67_groups_0 = const()[name = string("query_67_groups_0"), val = int32(1)]; + tensor query_67_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_67_dilations_0, groups = query_67_groups_0, pad = query_67_pad_0, pad_type = query_67_pad_type_0, strides = query_67_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_101_cast_fp16)[name = string("query_67_cast_fp16")]; + string current_key_45_pad_type_0 = const()[name = string("current_key_45_pad_type_0"), val = string("valid")]; + tensor current_key_45_strides_0 = const()[name = string("current_key_45_strides_0"), val = tensor([1, 1])]; + tensor current_key_45_pad_0 = const()[name = string("current_key_45_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_45_dilations_0 = const()[name = string("current_key_45_dilations_0"), val = tensor([1, 1])]; + int32 current_key_45_groups_0 = const()[name = string("current_key_45_groups_0"), val = int32(1)]; + tensor current_key_45_cast_fp16 = conv(dilations = current_key_45_dilations_0, groups = current_key_45_groups_0, pad = current_key_45_pad_0, pad_type = current_key_45_pad_type_0, strides = current_key_45_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_101_cast_fp16)[name = string("current_key_45_cast_fp16")]; + string current_value_23_pad_type_0 = const()[name = string("current_value_23_pad_type_0"), val = string("valid")]; + tensor current_value_23_strides_0 = const()[name = string("current_value_23_strides_0"), val = tensor([1, 1])]; + tensor current_value_23_pad_0 = const()[name = string("current_value_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_23_dilations_0 = const()[name = string("current_value_23_dilations_0"), val = tensor([1, 1])]; + int32 current_value_23_groups_0 = const()[name = string("current_value_23_groups_0"), val = int32(1)]; + tensor current_value_23_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_23_dilations_0, groups = current_value_23_groups_0, pad = current_value_23_pad_0, pad_type = current_value_23_pad_type_0, strides = current_value_23_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_101_cast_fp16)[name = string("current_value_23_cast_fp16")]; + tensor var_3471 = const()[name = string("op_3471"), val = tensor([16, 128, 1, 1])]; + tensor inputs_93_cast_fp16 = reshape(shape = var_3471, x = query_67_cast_fp16)[name = string("inputs_93_cast_fp16")]; + tensor inputs_sq_93_cast_fp16 = mul(x = inputs_93_cast_fp16, y = inputs_93_cast_fp16)[name = string("inputs_sq_93_cast_fp16")]; + tensor variance_93_axes_0 = const()[name = string("variance_93_axes_0"), val = tensor([1])]; + bool variance_93_keep_dims_0 = const()[name = string("variance_93_keep_dims_0"), val = bool(true)]; + tensor variance_93_cast_fp16 = reduce_mean(axes = variance_93_axes_0, keep_dims = variance_93_keep_dims_0, x = inputs_sq_93_cast_fp16)[name = string("variance_93_cast_fp16")]; + fp16 var_3477_to_fp16 = const()[name = string("op_3477_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3478_cast_fp16 = add(x = variance_93_cast_fp16, y = var_3477_to_fp16)[name = string("op_3478_cast_fp16")]; + fp32 var_3479_epsilon_0 = const()[name = string("op_3479_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3479_cast_fp16 = rsqrt(epsilon = var_3479_epsilon_0, x = var_3478_cast_fp16)[name = string("op_3479_cast_fp16")]; + tensor hidden_states_115_cast_fp16 = mul(x = inputs_93_cast_fp16, y = var_3479_cast_fp16)[name = string("hidden_states_115_cast_fp16")]; + tensor query_normed_23_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_115_cast_fp16)[name = string("query_normed_23_cast_fp16")]; + tensor var_3487 = const()[name = string("op_3487"), val = tensor([8, 128, 1, 1])]; + tensor inputs_95_cast_fp16 = reshape(shape = var_3487, x = current_key_45_cast_fp16)[name = string("inputs_95_cast_fp16")]; + tensor inputs_sq_95_cast_fp16 = mul(x = inputs_95_cast_fp16, y = inputs_95_cast_fp16)[name = string("inputs_sq_95_cast_fp16")]; + tensor variance_95_axes_0 = const()[name = string("variance_95_axes_0"), val = tensor([1])]; + bool variance_95_keep_dims_0 = const()[name = string("variance_95_keep_dims_0"), val = bool(true)]; + tensor variance_95_cast_fp16 = reduce_mean(axes = variance_95_axes_0, keep_dims = variance_95_keep_dims_0, x = inputs_sq_95_cast_fp16)[name = string("variance_95_cast_fp16")]; + fp16 var_3493_to_fp16 = const()[name = string("op_3493_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3494_cast_fp16 = add(x = variance_95_cast_fp16, y = var_3493_to_fp16)[name = string("op_3494_cast_fp16")]; + fp32 var_3495_epsilon_0 = const()[name = string("op_3495_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3495_cast_fp16 = rsqrt(epsilon = var_3495_epsilon_0, x = var_3494_cast_fp16)[name = string("op_3495_cast_fp16")]; + tensor hidden_states_117_cast_fp16 = mul(x = inputs_95_cast_fp16, y = var_3495_cast_fp16)[name = string("hidden_states_117_cast_fp16")]; + tensor current_key_normed_23_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_117_cast_fp16)[name = string("current_key_normed_23_cast_fp16")]; + tensor var_3513 = const()[name = string("op_3513"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_89_cast_fp16 = reshape(shape = var_3513, x = query_normed_23_cast_fp16)[name = string("mh_q_89_cast_fp16")]; + tensor var_3515 = const()[name = string("op_3515"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_89_cast_fp16 = reshape(shape = var_3515, x = current_key_normed_23_cast_fp16)[name = string("mh_k_89_cast_fp16")]; + tensor var_3519_cast_fp16 = mul(x = mh_q_89_cast_fp16, y = cos_21_to_fp16)[name = string("op_3519_cast_fp16")]; + tensor var_3524_begin_0 = const()[name = string("op_3524_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3524_end_0 = const()[name = string("op_3524_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_3524_end_mask_0 = const()[name = string("op_3524_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_3524_cast_fp16 = slice_by_index(begin = var_3524_begin_0, end = var_3524_end_0, end_mask = var_3524_end_mask_0, x = mh_q_89_cast_fp16)[name = string("op_3524_cast_fp16")]; + tensor var_3530_begin_0 = const()[name = string("op_3530_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_3530_end_0 = const()[name = string("op_3530_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_3530_end_mask_0 = const()[name = string("op_3530_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_3530_cast_fp16 = slice_by_index(begin = var_3530_begin_0, end = var_3530_end_0, end_mask = var_3530_end_mask_0, x = mh_q_89_cast_fp16)[name = string("op_3530_cast_fp16")]; + fp16 const_236_promoted_to_fp16 = const()[name = string("const_236_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3532_cast_fp16 = mul(x = var_3530_cast_fp16, y = const_236_promoted_to_fp16)[name = string("op_3532_cast_fp16")]; + bool var_3534_interleave_0 = const()[name = string("op_3534_interleave_0"), val = bool(false)]; + tensor var_3534_cast_fp16 = concat(axis = var_3418, interleave = var_3534_interleave_0, values = (var_3532_cast_fp16, var_3524_cast_fp16))[name = string("op_3534_cast_fp16")]; + tensor var_3535_cast_fp16 = mul(x = var_3534_cast_fp16, y = sin_21_to_fp16)[name = string("op_3535_cast_fp16")]; + tensor mh_q_91_cast_fp16 = add(x = var_3519_cast_fp16, y = var_3535_cast_fp16)[name = string("mh_q_91_cast_fp16")]; + tensor var_3537_cast_fp16 = mul(x = mh_k_89_cast_fp16, y = cos_21_to_fp16)[name = string("op_3537_cast_fp16")]; + tensor var_3542_begin_0 = const()[name = string("op_3542_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3542_end_0 = const()[name = string("op_3542_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_3542_end_mask_0 = const()[name = string("op_3542_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_3542_cast_fp16 = slice_by_index(begin = var_3542_begin_0, end = var_3542_end_0, end_mask = var_3542_end_mask_0, x = mh_k_89_cast_fp16)[name = string("op_3542_cast_fp16")]; + tensor var_3548_begin_0 = const()[name = string("op_3548_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_3548_end_0 = const()[name = string("op_3548_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_3548_end_mask_0 = const()[name = string("op_3548_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_3548_cast_fp16 = slice_by_index(begin = var_3548_begin_0, end = var_3548_end_0, end_mask = var_3548_end_mask_0, x = mh_k_89_cast_fp16)[name = string("op_3548_cast_fp16")]; + fp16 const_239_promoted_to_fp16 = const()[name = string("const_239_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3550_cast_fp16 = mul(x = var_3548_cast_fp16, y = const_239_promoted_to_fp16)[name = string("op_3550_cast_fp16")]; + bool var_3552_interleave_0 = const()[name = string("op_3552_interleave_0"), val = bool(false)]; + tensor var_3552_cast_fp16 = concat(axis = var_3418, interleave = var_3552_interleave_0, values = (var_3550_cast_fp16, var_3542_cast_fp16))[name = string("op_3552_cast_fp16")]; + tensor var_3553_cast_fp16 = mul(x = var_3552_cast_fp16, y = sin_21_to_fp16)[name = string("op_3553_cast_fp16")]; + tensor mh_k_91_cast_fp16 = add(x = var_3537_cast_fp16, y = var_3553_cast_fp16)[name = string("mh_k_91_cast_fp16")]; + tensor var_3557 = const()[name = string("op_3557"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_47_cast_fp16 = reshape(shape = var_3557, x = mh_k_91_cast_fp16)[name = string("current_key_47_cast_fp16")]; + tensor var_3563_to_fp16 = const()[name = string("op_3563_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191232)))]; + tensor var_3564_cast_fp16 = mul(x = obj_103_cast_fp16, y = var_3563_to_fp16)[name = string("op_3564_cast_fp16")]; + tensor var_3561_to_fp16 = const()[name = string("op_3561_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191360)))]; + tensor var_3565_cast_fp16 = mul(x = current_key_47_cast_fp16, y = var_3561_to_fp16)[name = string("op_3565_cast_fp16")]; + tensor key_47_cast_fp16 = add(x = var_3564_cast_fp16, y = var_3565_cast_fp16)[name = string("key_47_cast_fp16")]; + tensor var_3567_to_fp16 = const()[name = string("op_3567_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191232)))]; + tensor var_3568_cast_fp16 = mul(x = obj_105_cast_fp16, y = var_3567_to_fp16)[name = string("op_3568_cast_fp16")]; + tensor var_3569_cast_fp16 = mul(x = current_value_23_cast_fp16, y = var_3561_to_fp16)[name = string("op_3569_cast_fp16")]; + tensor value_23_cast_fp16 = add(x = var_3568_cast_fp16, y = var_3569_cast_fp16)[name = string("value_23_cast_fp16")]; + fp16 var_3576_to_fp16 = const()[name = string("op_3576_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_95_cast_fp16 = mul(x = mh_q_91_cast_fp16, y = var_3576_to_fp16)[name = string("mh_q_95_cast_fp16")]; + tensor var_3578 = const()[name = string("op_3578"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_93_cast_fp16 = reshape(shape = var_3578, x = key_47_cast_fp16)[name = string("mh_k_93_cast_fp16")]; + tensor var_3580 = const()[name = string("op_3580"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_45_cast_fp16 = reshape(shape = var_3580, x = value_23_cast_fp16)[name = string("mh_v_45_cast_fp16")]; + tensor transpose_44_perm_0 = const()[name = string("transpose_44_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_22_reps_0 = const()[name = string("tile_22_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_44_cast_fp16 = transpose(perm = transpose_44_perm_0, x = mh_k_93_cast_fp16)[name = string("transpose_413")]; + tensor tile_22_cast_fp16 = tile(reps = tile_22_reps_0, x = transpose_44_cast_fp16)[name = string("tile_22_cast_fp16")]; + tensor concat_57 = const()[name = string("concat_57"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_44_cast_fp16 = reshape(shape = concat_57, x = tile_22_cast_fp16)[name = string("reshape_44_cast_fp16")]; + tensor transpose_45_perm_0 = const()[name = string("transpose_45_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_58 = const()[name = string("concat_58"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_45_cast_fp16 = transpose(perm = transpose_45_perm_0, x = reshape_44_cast_fp16)[name = string("transpose_412")]; + tensor reshape_45_cast_fp16 = reshape(shape = concat_58, x = transpose_45_cast_fp16)[name = string("reshape_45_cast_fp16")]; + tensor transpose_46_perm_0 = const()[name = string("transpose_46_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_23_reps_0 = const()[name = string("tile_23_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_46_cast_fp16 = transpose(perm = transpose_46_perm_0, x = mh_v_45_cast_fp16)[name = string("transpose_411")]; + tensor tile_23_cast_fp16 = tile(reps = tile_23_reps_0, x = transpose_46_cast_fp16)[name = string("tile_23_cast_fp16")]; + tensor concat_59 = const()[name = string("concat_59"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_46_cast_fp16 = reshape(shape = concat_59, x = tile_23_cast_fp16)[name = string("reshape_46_cast_fp16")]; + tensor transpose_47_perm_0 = const()[name = string("transpose_47_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_60 = const()[name = string("concat_60"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_47_cast_fp16 = transpose(perm = transpose_47_perm_0, x = reshape_46_cast_fp16)[name = string("transpose_410")]; + tensor reshape_47_cast_fp16 = reshape(shape = concat_60, x = transpose_47_cast_fp16)[name = string("reshape_47_cast_fp16")]; + tensor transpose_361_perm_0 = const()[name = string("transpose_361_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_67_transpose_x_1 = const()[name = string("mh_w_67_transpose_x_1"), val = bool(true)]; + bool mh_w_67_transpose_y_1 = const()[name = string("mh_w_67_transpose_y_1"), val = bool(false)]; + tensor transpose_361_cast_fp16 = transpose(perm = transpose_361_perm_0, x = reshape_45_cast_fp16)[name = string("transpose_409")]; + tensor mh_w_67_cast_fp16 = matmul(transpose_x = mh_w_67_transpose_x_1, transpose_y = mh_w_67_transpose_y_1, x = mh_q_95_cast_fp16, y = transpose_361_cast_fp16)[name = string("mh_w_67_cast_fp16")]; + tensor var_3588_to_fp16 = const()[name = string("op_3588_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191488)))]; + tensor mh_w_69_cast_fp16 = add(x = mh_w_67_cast_fp16, y = var_3588_to_fp16)[name = string("mh_w_69_cast_fp16")]; + tensor mh_w_71_cast_fp16 = softmax(axis = var_3408, x = mh_w_69_cast_fp16)[name = string("mh_w_71_cast_fp16")]; + tensor transpose_362_perm_0 = const()[name = string("transpose_362_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_23_transpose_x_1 = const()[name = string("attn_23_transpose_x_1"), val = bool(false)]; + bool attn_23_transpose_y_1 = const()[name = string("attn_23_transpose_y_1"), val = bool(true)]; + tensor transpose_362_cast_fp16 = transpose(perm = transpose_362_perm_0, x = reshape_47_cast_fp16)[name = string("transpose_408")]; + tensor attn_23_cast_fp16 = matmul(transpose_x = attn_23_transpose_x_1, transpose_y = attn_23_transpose_y_1, x = transpose_362_cast_fp16, y = mh_w_71_cast_fp16)[name = string("attn_23_cast_fp16")]; + tensor var_3594 = const()[name = string("op_3594"), val = tensor([1, 2048, 1, 1])]; + tensor input_93_cast_fp16 = reshape(shape = var_3594, x = attn_23_cast_fp16)[name = string("input_93_cast_fp16")]; + string obj_107_pad_type_0 = const()[name = string("obj_107_pad_type_0"), val = string("valid")]; + tensor obj_107_strides_0 = const()[name = string("obj_107_strides_0"), val = tensor([1, 1])]; + tensor obj_107_pad_0 = const()[name = string("obj_107_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_107_dilations_0 = const()[name = string("obj_107_dilations_0"), val = tensor([1, 1])]; + int32 obj_107_groups_0 = const()[name = string("obj_107_groups_0"), val = int32(1)]; + tensor obj_107_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_107_dilations_0, groups = obj_107_groups_0, pad = obj_107_pad_0, pad_type = obj_107_pad_type_0, strides = obj_107_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_93_cast_fp16)[name = string("obj_107_cast_fp16")]; + tensor inputs_97_cast_fp16 = add(x = inputs_91_cast_fp16, y = obj_107_cast_fp16)[name = string("inputs_97_cast_fp16")]; + tensor inputs_sq_97_cast_fp16 = mul(x = inputs_97_cast_fp16, y = inputs_97_cast_fp16)[name = string("inputs_sq_97_cast_fp16")]; + tensor variance_97_axes_0 = const()[name = string("variance_97_axes_0"), val = tensor([1])]; + bool variance_97_keep_dims_0 = const()[name = string("variance_97_keep_dims_0"), val = bool(true)]; + tensor variance_97_cast_fp16 = reduce_mean(axes = variance_97_axes_0, keep_dims = variance_97_keep_dims_0, x = inputs_sq_97_cast_fp16)[name = string("variance_97_cast_fp16")]; + fp16 var_3612_to_fp16 = const()[name = string("op_3612_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3613_cast_fp16 = add(x = variance_97_cast_fp16, y = var_3612_to_fp16)[name = string("op_3613_cast_fp16")]; + fp32 var_3614_epsilon_0 = const()[name = string("op_3614_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3614_cast_fp16 = rsqrt(epsilon = var_3614_epsilon_0, x = var_3613_cast_fp16)[name = string("op_3614_cast_fp16")]; + tensor hidden_states_119_cast_fp16 = mul(x = inputs_97_cast_fp16, y = var_3614_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; + tensor input_95_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_119_cast_fp16)[name = string("input_95_cast_fp16")]; + string input_97_pad_type_0 = const()[name = string("input_97_pad_type_0"), val = string("valid")]; + tensor input_97_strides_0 = const()[name = string("input_97_strides_0"), val = tensor([1, 1])]; + tensor input_97_pad_0 = const()[name = string("input_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_97_dilations_0 = const()[name = string("input_97_dilations_0"), val = tensor([1, 1])]; + int32 input_97_groups_0 = const()[name = string("input_97_groups_0"), val = int32(1)]; + tensor input_97_cast_fp16 = conv(dilations = input_97_dilations_0, groups = input_97_groups_0, pad = input_97_pad_0, pad_type = input_97_pad_type_0, strides = input_97_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_95_cast_fp16)[name = string("input_97_cast_fp16")]; + tensor var_3628_cast_fp16 = silu(x = input_97_cast_fp16)[name = string("op_3628_cast_fp16")]; + string var_3634_pad_type_0 = const()[name = string("op_3634_pad_type_0"), val = string("valid")]; + tensor var_3634_strides_0 = const()[name = string("op_3634_strides_0"), val = tensor([1, 1])]; + tensor var_3634_pad_0 = const()[name = string("op_3634_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3634_dilations_0 = const()[name = string("op_3634_dilations_0"), val = tensor([1, 1])]; + int32 var_3634_groups_0 = const()[name = string("op_3634_groups_0"), val = int32(1)]; + tensor var_3634_cast_fp16 = conv(dilations = var_3634_dilations_0, groups = var_3634_groups_0, pad = var_3634_pad_0, pad_type = var_3634_pad_type_0, strides = var_3634_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_95_cast_fp16)[name = string("op_3634_cast_fp16")]; + tensor input_99_cast_fp16 = mul(x = var_3628_cast_fp16, y = var_3634_cast_fp16)[name = string("input_99_cast_fp16")]; + string hidden_states_121_pad_type_0 = const()[name = string("hidden_states_121_pad_type_0"), val = string("valid")]; + tensor hidden_states_121_strides_0 = const()[name = string("hidden_states_121_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_121_pad_0 = const()[name = string("hidden_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_121_dilations_0 = const()[name = string("hidden_states_121_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_121_groups_0 = const()[name = string("hidden_states_121_groups_0"), val = int32(1)]; + tensor hidden_states_121_cast_fp16 = conv(dilations = hidden_states_121_dilations_0, groups = hidden_states_121_groups_0, pad = hidden_states_121_pad_0, pad_type = hidden_states_121_pad_type_0, strides = hidden_states_121_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_99_cast_fp16)[name = string("hidden_states_121_cast_fp16")]; + tensor inputs_99_cast_fp16 = add(x = inputs_97_cast_fp16, y = hidden_states_121_cast_fp16)[name = string("inputs_99_cast_fp16")]; + tensor obj_111_begin_0 = const()[name = string("obj_111_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_111_end_0 = const()[name = string("obj_111_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_111_end_mask_0 = const()[name = string("obj_111_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_111_cast_fp16 = slice_by_index(begin = obj_111_begin_0, end = obj_111_end_0, end_mask = obj_111_end_mask_0, x = key_caches_5_cast_fp16)[name = string("obj_111_cast_fp16")]; + tensor obj_113_begin_0 = const()[name = string("obj_113_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_113_end_0 = const()[name = string("obj_113_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_113_end_mask_0 = const()[name = string("obj_113_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_113_cast_fp16 = slice_by_index(begin = obj_113_begin_0, end = obj_113_end_0, end_mask = obj_113_end_mask_0, x = value_caches_5_cast_fp16)[name = string("obj_113_cast_fp16")]; + int32 var_3682 = const()[name = string("op_3682"), val = int32(3)]; + int32 var_3692 = const()[name = string("op_3692"), val = int32(-2)]; + tensor inputs_sq_99_cast_fp16 = mul(x = inputs_99_cast_fp16, y = inputs_99_cast_fp16)[name = string("inputs_sq_99_cast_fp16")]; + tensor variance_99_axes_0 = const()[name = string("variance_99_axes_0"), val = tensor([1])]; + bool variance_99_keep_dims_0 = const()[name = string("variance_99_keep_dims_0"), val = bool(true)]; + tensor variance_99_cast_fp16 = reduce_mean(axes = variance_99_axes_0, keep_dims = variance_99_keep_dims_0, x = inputs_sq_99_cast_fp16)[name = string("variance_99_cast_fp16")]; + fp16 var_3706_to_fp16 = const()[name = string("op_3706_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3707_cast_fp16 = add(x = variance_99_cast_fp16, y = var_3706_to_fp16)[name = string("op_3707_cast_fp16")]; + fp32 var_3708_epsilon_0 = const()[name = string("op_3708_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3708_cast_fp16 = rsqrt(epsilon = var_3708_epsilon_0, x = var_3707_cast_fp16)[name = string("op_3708_cast_fp16")]; + tensor hidden_states_123_cast_fp16 = mul(x = inputs_99_cast_fp16, y = var_3708_cast_fp16)[name = string("hidden_states_123_cast_fp16")]; + tensor obj_109_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_123_cast_fp16)[name = string("obj_109_cast_fp16")]; + string query_73_pad_type_0 = const()[name = string("query_73_pad_type_0"), val = string("valid")]; + tensor query_73_strides_0 = const()[name = string("query_73_strides_0"), val = tensor([1, 1])]; + tensor query_73_pad_0 = const()[name = string("query_73_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_73_dilations_0 = const()[name = string("query_73_dilations_0"), val = tensor([1, 1])]; + int32 query_73_groups_0 = const()[name = string("query_73_groups_0"), val = int32(1)]; + tensor query_73_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_73_dilations_0, groups = query_73_groups_0, pad = query_73_pad_0, pad_type = query_73_pad_type_0, strides = query_73_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_109_cast_fp16)[name = string("query_73_cast_fp16")]; + string current_key_49_pad_type_0 = const()[name = string("current_key_49_pad_type_0"), val = string("valid")]; + tensor current_key_49_strides_0 = const()[name = string("current_key_49_strides_0"), val = tensor([1, 1])]; + tensor current_key_49_pad_0 = const()[name = string("current_key_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_49_dilations_0 = const()[name = string("current_key_49_dilations_0"), val = tensor([1, 1])]; + int32 current_key_49_groups_0 = const()[name = string("current_key_49_groups_0"), val = int32(1)]; + tensor current_key_49_cast_fp16 = conv(dilations = current_key_49_dilations_0, groups = current_key_49_groups_0, pad = current_key_49_pad_0, pad_type = current_key_49_pad_type_0, strides = current_key_49_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_109_cast_fp16)[name = string("current_key_49_cast_fp16")]; + string current_value_25_pad_type_0 = const()[name = string("current_value_25_pad_type_0"), val = string("valid")]; + tensor current_value_25_strides_0 = const()[name = string("current_value_25_strides_0"), val = tensor([1, 1])]; + tensor current_value_25_pad_0 = const()[name = string("current_value_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_25_dilations_0 = const()[name = string("current_value_25_dilations_0"), val = tensor([1, 1])]; + int32 current_value_25_groups_0 = const()[name = string("current_value_25_groups_0"), val = int32(1)]; + tensor current_value_25_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_25_dilations_0, groups = current_value_25_groups_0, pad = current_value_25_pad_0, pad_type = current_value_25_pad_type_0, strides = current_value_25_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_109_cast_fp16)[name = string("current_value_25_cast_fp16")]; + tensor var_3745 = const()[name = string("op_3745"), val = tensor([16, 128, 1, 1])]; + tensor inputs_101_cast_fp16 = reshape(shape = var_3745, x = query_73_cast_fp16)[name = string("inputs_101_cast_fp16")]; + tensor inputs_sq_101_cast_fp16 = mul(x = inputs_101_cast_fp16, y = inputs_101_cast_fp16)[name = string("inputs_sq_101_cast_fp16")]; + tensor variance_101_axes_0 = const()[name = string("variance_101_axes_0"), val = tensor([1])]; + bool variance_101_keep_dims_0 = const()[name = string("variance_101_keep_dims_0"), val = bool(true)]; + tensor variance_101_cast_fp16 = reduce_mean(axes = variance_101_axes_0, keep_dims = variance_101_keep_dims_0, x = inputs_sq_101_cast_fp16)[name = string("variance_101_cast_fp16")]; + fp16 var_3751_to_fp16 = const()[name = string("op_3751_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3752_cast_fp16 = add(x = variance_101_cast_fp16, y = var_3751_to_fp16)[name = string("op_3752_cast_fp16")]; + fp32 var_3753_epsilon_0 = const()[name = string("op_3753_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3753_cast_fp16 = rsqrt(epsilon = var_3753_epsilon_0, x = var_3752_cast_fp16)[name = string("op_3753_cast_fp16")]; + tensor hidden_states_125_cast_fp16 = mul(x = inputs_101_cast_fp16, y = var_3753_cast_fp16)[name = string("hidden_states_125_cast_fp16")]; + tensor query_normed_25_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_125_cast_fp16)[name = string("query_normed_25_cast_fp16")]; + tensor var_3761 = const()[name = string("op_3761"), val = tensor([8, 128, 1, 1])]; + tensor inputs_103_cast_fp16 = reshape(shape = var_3761, x = current_key_49_cast_fp16)[name = string("inputs_103_cast_fp16")]; + tensor inputs_sq_103_cast_fp16 = mul(x = inputs_103_cast_fp16, y = inputs_103_cast_fp16)[name = string("inputs_sq_103_cast_fp16")]; + tensor variance_103_axes_0 = const()[name = string("variance_103_axes_0"), val = tensor([1])]; + bool variance_103_keep_dims_0 = const()[name = string("variance_103_keep_dims_0"), val = bool(true)]; + tensor variance_103_cast_fp16 = reduce_mean(axes = variance_103_axes_0, keep_dims = variance_103_keep_dims_0, x = inputs_sq_103_cast_fp16)[name = string("variance_103_cast_fp16")]; + fp16 var_3767_to_fp16 = const()[name = string("op_3767_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3768_cast_fp16 = add(x = variance_103_cast_fp16, y = var_3767_to_fp16)[name = string("op_3768_cast_fp16")]; + fp32 var_3769_epsilon_0 = const()[name = string("op_3769_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3769_cast_fp16 = rsqrt(epsilon = var_3769_epsilon_0, x = var_3768_cast_fp16)[name = string("op_3769_cast_fp16")]; + tensor hidden_states_127_cast_fp16 = mul(x = inputs_103_cast_fp16, y = var_3769_cast_fp16)[name = string("hidden_states_127_cast_fp16")]; + tensor current_key_normed_25_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_127_cast_fp16)[name = string("current_key_normed_25_cast_fp16")]; + tensor var_3787 = const()[name = string("op_3787"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_97_cast_fp16 = reshape(shape = var_3787, x = query_normed_25_cast_fp16)[name = string("mh_q_97_cast_fp16")]; + tensor var_3789 = const()[name = string("op_3789"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_97_cast_fp16 = reshape(shape = var_3789, x = current_key_normed_25_cast_fp16)[name = string("mh_k_97_cast_fp16")]; + tensor var_3793_cast_fp16 = mul(x = mh_q_97_cast_fp16, y = cos_21_to_fp16)[name = string("op_3793_cast_fp16")]; + tensor var_3798_begin_0 = const()[name = string("op_3798_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3798_end_0 = const()[name = string("op_3798_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_3798_end_mask_0 = const()[name = string("op_3798_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_3798_cast_fp16 = slice_by_index(begin = var_3798_begin_0, end = var_3798_end_0, end_mask = var_3798_end_mask_0, x = mh_q_97_cast_fp16)[name = string("op_3798_cast_fp16")]; + tensor var_3804_begin_0 = const()[name = string("op_3804_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_3804_end_0 = const()[name = string("op_3804_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_3804_end_mask_0 = const()[name = string("op_3804_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_3804_cast_fp16 = slice_by_index(begin = var_3804_begin_0, end = var_3804_end_0, end_mask = var_3804_end_mask_0, x = mh_q_97_cast_fp16)[name = string("op_3804_cast_fp16")]; + fp16 const_256_promoted_to_fp16 = const()[name = string("const_256_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3806_cast_fp16 = mul(x = var_3804_cast_fp16, y = const_256_promoted_to_fp16)[name = string("op_3806_cast_fp16")]; + bool var_3808_interleave_0 = const()[name = string("op_3808_interleave_0"), val = bool(false)]; + tensor var_3808_cast_fp16 = concat(axis = var_3692, interleave = var_3808_interleave_0, values = (var_3806_cast_fp16, var_3798_cast_fp16))[name = string("op_3808_cast_fp16")]; + tensor var_3809_cast_fp16 = mul(x = var_3808_cast_fp16, y = sin_21_to_fp16)[name = string("op_3809_cast_fp16")]; + tensor mh_q_99_cast_fp16 = add(x = var_3793_cast_fp16, y = var_3809_cast_fp16)[name = string("mh_q_99_cast_fp16")]; + tensor var_3811_cast_fp16 = mul(x = mh_k_97_cast_fp16, y = cos_21_to_fp16)[name = string("op_3811_cast_fp16")]; + tensor var_3816_begin_0 = const()[name = string("op_3816_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3816_end_0 = const()[name = string("op_3816_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_3816_end_mask_0 = const()[name = string("op_3816_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_3816_cast_fp16 = slice_by_index(begin = var_3816_begin_0, end = var_3816_end_0, end_mask = var_3816_end_mask_0, x = mh_k_97_cast_fp16)[name = string("op_3816_cast_fp16")]; + tensor var_3822_begin_0 = const()[name = string("op_3822_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_3822_end_0 = const()[name = string("op_3822_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_3822_end_mask_0 = const()[name = string("op_3822_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_3822_cast_fp16 = slice_by_index(begin = var_3822_begin_0, end = var_3822_end_0, end_mask = var_3822_end_mask_0, x = mh_k_97_cast_fp16)[name = string("op_3822_cast_fp16")]; + fp16 const_259_promoted_to_fp16 = const()[name = string("const_259_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3824_cast_fp16 = mul(x = var_3822_cast_fp16, y = const_259_promoted_to_fp16)[name = string("op_3824_cast_fp16")]; + bool var_3826_interleave_0 = const()[name = string("op_3826_interleave_0"), val = bool(false)]; + tensor var_3826_cast_fp16 = concat(axis = var_3692, interleave = var_3826_interleave_0, values = (var_3824_cast_fp16, var_3816_cast_fp16))[name = string("op_3826_cast_fp16")]; + tensor var_3827_cast_fp16 = mul(x = var_3826_cast_fp16, y = sin_21_to_fp16)[name = string("op_3827_cast_fp16")]; + tensor mh_k_99_cast_fp16 = add(x = var_3811_cast_fp16, y = var_3827_cast_fp16)[name = string("mh_k_99_cast_fp16")]; + tensor var_3831 = const()[name = string("op_3831"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_51_cast_fp16 = reshape(shape = var_3831, x = mh_k_99_cast_fp16)[name = string("current_key_51_cast_fp16")]; + tensor var_3837_to_fp16 = const()[name = string("op_3837_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191232)))]; + tensor var_3838_cast_fp16 = mul(x = obj_111_cast_fp16, y = var_3837_to_fp16)[name = string("op_3838_cast_fp16")]; + tensor var_3835_to_fp16 = const()[name = string("op_3835_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191360)))]; + tensor var_3839_cast_fp16 = mul(x = current_key_51_cast_fp16, y = var_3835_to_fp16)[name = string("op_3839_cast_fp16")]; + tensor key_51_cast_fp16 = add(x = var_3838_cast_fp16, y = var_3839_cast_fp16)[name = string("key_51_cast_fp16")]; + tensor var_3841_to_fp16 = const()[name = string("op_3841_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191232)))]; + tensor var_3842_cast_fp16 = mul(x = obj_113_cast_fp16, y = var_3841_to_fp16)[name = string("op_3842_cast_fp16")]; + tensor var_3843_cast_fp16 = mul(x = current_value_25_cast_fp16, y = var_3835_to_fp16)[name = string("op_3843_cast_fp16")]; + tensor value_25_cast_fp16 = add(x = var_3842_cast_fp16, y = var_3843_cast_fp16)[name = string("value_25_cast_fp16")]; + fp16 var_3850_to_fp16 = const()[name = string("op_3850_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_103_cast_fp16 = mul(x = mh_q_99_cast_fp16, y = var_3850_to_fp16)[name = string("mh_q_103_cast_fp16")]; + tensor var_3852 = const()[name = string("op_3852"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_101_cast_fp16 = reshape(shape = var_3852, x = key_51_cast_fp16)[name = string("mh_k_101_cast_fp16")]; + tensor var_3854 = const()[name = string("op_3854"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_49_cast_fp16 = reshape(shape = var_3854, x = value_25_cast_fp16)[name = string("mh_v_49_cast_fp16")]; + tensor transpose_48_perm_0 = const()[name = string("transpose_48_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_24_reps_0 = const()[name = string("tile_24_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_48_cast_fp16 = transpose(perm = transpose_48_perm_0, x = mh_k_101_cast_fp16)[name = string("transpose_407")]; + tensor tile_24_cast_fp16 = tile(reps = tile_24_reps_0, x = transpose_48_cast_fp16)[name = string("tile_24_cast_fp16")]; + tensor concat_61 = const()[name = string("concat_61"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_48_cast_fp16 = reshape(shape = concat_61, x = tile_24_cast_fp16)[name = string("reshape_48_cast_fp16")]; + tensor transpose_49_perm_0 = const()[name = string("transpose_49_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_62 = const()[name = string("concat_62"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_49_cast_fp16 = transpose(perm = transpose_49_perm_0, x = reshape_48_cast_fp16)[name = string("transpose_406")]; + tensor reshape_49_cast_fp16 = reshape(shape = concat_62, x = transpose_49_cast_fp16)[name = string("reshape_49_cast_fp16")]; + tensor transpose_50_perm_0 = const()[name = string("transpose_50_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_25_reps_0 = const()[name = string("tile_25_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_50_cast_fp16 = transpose(perm = transpose_50_perm_0, x = mh_v_49_cast_fp16)[name = string("transpose_405")]; + tensor tile_25_cast_fp16 = tile(reps = tile_25_reps_0, x = transpose_50_cast_fp16)[name = string("tile_25_cast_fp16")]; + tensor concat_63 = const()[name = string("concat_63"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_50_cast_fp16 = reshape(shape = concat_63, x = tile_25_cast_fp16)[name = string("reshape_50_cast_fp16")]; + tensor transpose_51_perm_0 = const()[name = string("transpose_51_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_64 = const()[name = string("concat_64"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_51_cast_fp16 = transpose(perm = transpose_51_perm_0, x = reshape_50_cast_fp16)[name = string("transpose_404")]; + tensor reshape_51_cast_fp16 = reshape(shape = concat_64, x = transpose_51_cast_fp16)[name = string("reshape_51_cast_fp16")]; + tensor transpose_365_perm_0 = const()[name = string("transpose_365_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_73_transpose_x_1 = const()[name = string("mh_w_73_transpose_x_1"), val = bool(true)]; + bool mh_w_73_transpose_y_1 = const()[name = string("mh_w_73_transpose_y_1"), val = bool(false)]; + tensor transpose_365_cast_fp16 = transpose(perm = transpose_365_perm_0, x = reshape_49_cast_fp16)[name = string("transpose_403")]; + tensor mh_w_73_cast_fp16 = matmul(transpose_x = mh_w_73_transpose_x_1, transpose_y = mh_w_73_transpose_y_1, x = mh_q_103_cast_fp16, y = transpose_365_cast_fp16)[name = string("mh_w_73_cast_fp16")]; + tensor var_3862_to_fp16 = const()[name = string("op_3862_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191488)))]; + tensor mh_w_75_cast_fp16 = add(x = mh_w_73_cast_fp16, y = var_3862_to_fp16)[name = string("mh_w_75_cast_fp16")]; + tensor mh_w_77_cast_fp16 = softmax(axis = var_3682, x = mh_w_75_cast_fp16)[name = string("mh_w_77_cast_fp16")]; + tensor transpose_366_perm_0 = const()[name = string("transpose_366_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_25_transpose_x_1 = const()[name = string("attn_25_transpose_x_1"), val = bool(false)]; + bool attn_25_transpose_y_1 = const()[name = string("attn_25_transpose_y_1"), val = bool(true)]; + tensor transpose_366_cast_fp16 = transpose(perm = transpose_366_perm_0, x = reshape_51_cast_fp16)[name = string("transpose_402")]; + tensor attn_25_cast_fp16 = matmul(transpose_x = attn_25_transpose_x_1, transpose_y = attn_25_transpose_y_1, x = transpose_366_cast_fp16, y = mh_w_77_cast_fp16)[name = string("attn_25_cast_fp16")]; + tensor var_3868 = const()[name = string("op_3868"), val = tensor([1, 2048, 1, 1])]; + tensor input_101_cast_fp16 = reshape(shape = var_3868, x = attn_25_cast_fp16)[name = string("input_101_cast_fp16")]; + string obj_115_pad_type_0 = const()[name = string("obj_115_pad_type_0"), val = string("valid")]; + tensor obj_115_strides_0 = const()[name = string("obj_115_strides_0"), val = tensor([1, 1])]; + tensor obj_115_pad_0 = const()[name = string("obj_115_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_115_dilations_0 = const()[name = string("obj_115_dilations_0"), val = tensor([1, 1])]; + int32 obj_115_groups_0 = const()[name = string("obj_115_groups_0"), val = int32(1)]; + tensor obj_115_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_115_dilations_0, groups = obj_115_groups_0, pad = obj_115_pad_0, pad_type = obj_115_pad_type_0, strides = obj_115_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_101_cast_fp16)[name = string("obj_115_cast_fp16")]; + tensor inputs_105_cast_fp16 = add(x = inputs_99_cast_fp16, y = obj_115_cast_fp16)[name = string("inputs_105_cast_fp16")]; + tensor inputs_sq_105_cast_fp16 = mul(x = inputs_105_cast_fp16, y = inputs_105_cast_fp16)[name = string("inputs_sq_105_cast_fp16")]; + tensor variance_105_axes_0 = const()[name = string("variance_105_axes_0"), val = tensor([1])]; + bool variance_105_keep_dims_0 = const()[name = string("variance_105_keep_dims_0"), val = bool(true)]; + tensor variance_105_cast_fp16 = reduce_mean(axes = variance_105_axes_0, keep_dims = variance_105_keep_dims_0, x = inputs_sq_105_cast_fp16)[name = string("variance_105_cast_fp16")]; + fp16 var_3886_to_fp16 = const()[name = string("op_3886_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3887_cast_fp16 = add(x = variance_105_cast_fp16, y = var_3886_to_fp16)[name = string("op_3887_cast_fp16")]; + fp32 var_3888_epsilon_0 = const()[name = string("op_3888_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3888_cast_fp16 = rsqrt(epsilon = var_3888_epsilon_0, x = var_3887_cast_fp16)[name = string("op_3888_cast_fp16")]; + tensor hidden_states_129_cast_fp16 = mul(x = inputs_105_cast_fp16, y = var_3888_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; + tensor input_103_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_129_cast_fp16)[name = string("input_103_cast_fp16")]; + string input_105_pad_type_0 = const()[name = string("input_105_pad_type_0"), val = string("valid")]; + tensor input_105_strides_0 = const()[name = string("input_105_strides_0"), val = tensor([1, 1])]; + tensor input_105_pad_0 = const()[name = string("input_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_105_dilations_0 = const()[name = string("input_105_dilations_0"), val = tensor([1, 1])]; + int32 input_105_groups_0 = const()[name = string("input_105_groups_0"), val = int32(1)]; + tensor input_105_cast_fp16 = conv(dilations = input_105_dilations_0, groups = input_105_groups_0, pad = input_105_pad_0, pad_type = input_105_pad_type_0, strides = input_105_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_103_cast_fp16)[name = string("input_105_cast_fp16")]; + tensor var_3902_cast_fp16 = silu(x = input_105_cast_fp16)[name = string("op_3902_cast_fp16")]; + string var_3908_pad_type_0 = const()[name = string("op_3908_pad_type_0"), val = string("valid")]; + tensor var_3908_strides_0 = const()[name = string("op_3908_strides_0"), val = tensor([1, 1])]; + tensor var_3908_pad_0 = const()[name = string("op_3908_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3908_dilations_0 = const()[name = string("op_3908_dilations_0"), val = tensor([1, 1])]; + int32 var_3908_groups_0 = const()[name = string("op_3908_groups_0"), val = int32(1)]; + tensor var_3908_cast_fp16 = conv(dilations = var_3908_dilations_0, groups = var_3908_groups_0, pad = var_3908_pad_0, pad_type = var_3908_pad_type_0, strides = var_3908_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_103_cast_fp16)[name = string("op_3908_cast_fp16")]; + tensor input_107_cast_fp16 = mul(x = var_3902_cast_fp16, y = var_3908_cast_fp16)[name = string("input_107_cast_fp16")]; + string hidden_states_131_pad_type_0 = const()[name = string("hidden_states_131_pad_type_0"), val = string("valid")]; + tensor hidden_states_131_strides_0 = const()[name = string("hidden_states_131_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_131_pad_0 = const()[name = string("hidden_states_131_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_131_dilations_0 = const()[name = string("hidden_states_131_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_131_groups_0 = const()[name = string("hidden_states_131_groups_0"), val = int32(1)]; + tensor hidden_states_131_cast_fp16 = conv(dilations = hidden_states_131_dilations_0, groups = hidden_states_131_groups_0, pad = hidden_states_131_pad_0, pad_type = hidden_states_131_pad_type_0, strides = hidden_states_131_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_107_cast_fp16)[name = string("hidden_states_131_cast_fp16")]; + tensor inputs_107_cast_fp16 = add(x = inputs_105_cast_fp16, y = hidden_states_131_cast_fp16)[name = string("inputs_107_cast_fp16")]; + tensor obj_119_begin_0 = const()[name = string("obj_119_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_119_end_0 = const()[name = string("obj_119_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_119_end_mask_0 = const()[name = string("obj_119_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_119_cast_fp16 = slice_by_index(begin = obj_119_begin_0, end = obj_119_end_0, end_mask = obj_119_end_mask_0, x = key_caches_5_cast_fp16)[name = string("obj_119_cast_fp16")]; + tensor obj_121_begin_0 = const()[name = string("obj_121_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_121_end_0 = const()[name = string("obj_121_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_121_end_mask_0 = const()[name = string("obj_121_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_121_cast_fp16 = slice_by_index(begin = obj_121_begin_0, end = obj_121_end_0, end_mask = obj_121_end_mask_0, x = value_caches_5_cast_fp16)[name = string("obj_121_cast_fp16")]; + int32 var_3956 = const()[name = string("op_3956"), val = int32(3)]; + int32 var_3966 = const()[name = string("op_3966"), val = int32(-2)]; + tensor inputs_sq_107_cast_fp16 = mul(x = inputs_107_cast_fp16, y = inputs_107_cast_fp16)[name = string("inputs_sq_107_cast_fp16")]; + tensor variance_107_axes_0 = const()[name = string("variance_107_axes_0"), val = tensor([1])]; + bool variance_107_keep_dims_0 = const()[name = string("variance_107_keep_dims_0"), val = bool(true)]; + tensor variance_107_cast_fp16 = reduce_mean(axes = variance_107_axes_0, keep_dims = variance_107_keep_dims_0, x = inputs_sq_107_cast_fp16)[name = string("variance_107_cast_fp16")]; + fp16 var_3980_to_fp16 = const()[name = string("op_3980_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_3981_cast_fp16 = add(x = variance_107_cast_fp16, y = var_3980_to_fp16)[name = string("op_3981_cast_fp16")]; + fp32 var_3982_epsilon_0 = const()[name = string("op_3982_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_3982_cast_fp16 = rsqrt(epsilon = var_3982_epsilon_0, x = var_3981_cast_fp16)[name = string("op_3982_cast_fp16")]; + tensor hidden_states_133_cast_fp16 = mul(x = inputs_107_cast_fp16, y = var_3982_cast_fp16)[name = string("hidden_states_133_cast_fp16")]; + tensor obj_117_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_133_cast_fp16)[name = string("obj_117_cast_fp16")]; + string query_79_pad_type_0 = const()[name = string("query_79_pad_type_0"), val = string("valid")]; + tensor query_79_strides_0 = const()[name = string("query_79_strides_0"), val = tensor([1, 1])]; + tensor query_79_pad_0 = const()[name = string("query_79_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_79_dilations_0 = const()[name = string("query_79_dilations_0"), val = tensor([1, 1])]; + int32 query_79_groups_0 = const()[name = string("query_79_groups_0"), val = int32(1)]; + tensor query_79_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_79_dilations_0, groups = query_79_groups_0, pad = query_79_pad_0, pad_type = query_79_pad_type_0, strides = query_79_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_117_cast_fp16)[name = string("query_79_cast_fp16")]; + string current_key_53_pad_type_0 = const()[name = string("current_key_53_pad_type_0"), val = string("valid")]; + tensor current_key_53_strides_0 = const()[name = string("current_key_53_strides_0"), val = tensor([1, 1])]; + tensor current_key_53_pad_0 = const()[name = string("current_key_53_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_53_dilations_0 = const()[name = string("current_key_53_dilations_0"), val = tensor([1, 1])]; + int32 current_key_53_groups_0 = const()[name = string("current_key_53_groups_0"), val = int32(1)]; + tensor current_key_53_cast_fp16 = conv(dilations = current_key_53_dilations_0, groups = current_key_53_groups_0, pad = current_key_53_pad_0, pad_type = current_key_53_pad_type_0, strides = current_key_53_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_117_cast_fp16)[name = string("current_key_53_cast_fp16")]; + string current_value_27_pad_type_0 = const()[name = string("current_value_27_pad_type_0"), val = string("valid")]; + tensor current_value_27_strides_0 = const()[name = string("current_value_27_strides_0"), val = tensor([1, 1])]; + tensor current_value_27_pad_0 = const()[name = string("current_value_27_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_27_dilations_0 = const()[name = string("current_value_27_dilations_0"), val = tensor([1, 1])]; + int32 current_value_27_groups_0 = const()[name = string("current_value_27_groups_0"), val = int32(1)]; + tensor current_value_27_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_27_dilations_0, groups = current_value_27_groups_0, pad = current_value_27_pad_0, pad_type = current_value_27_pad_type_0, strides = current_value_27_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_117_cast_fp16)[name = string("current_value_27_cast_fp16")]; + tensor var_4019 = const()[name = string("op_4019"), val = tensor([16, 128, 1, 1])]; + tensor inputs_109_cast_fp16 = reshape(shape = var_4019, x = query_79_cast_fp16)[name = string("inputs_109_cast_fp16")]; + tensor inputs_sq_109_cast_fp16 = mul(x = inputs_109_cast_fp16, y = inputs_109_cast_fp16)[name = string("inputs_sq_109_cast_fp16")]; + tensor variance_109_axes_0 = const()[name = string("variance_109_axes_0"), val = tensor([1])]; + bool variance_109_keep_dims_0 = const()[name = string("variance_109_keep_dims_0"), val = bool(true)]; + tensor variance_109_cast_fp16 = reduce_mean(axes = variance_109_axes_0, keep_dims = variance_109_keep_dims_0, x = inputs_sq_109_cast_fp16)[name = string("variance_109_cast_fp16")]; + fp16 var_4025_to_fp16 = const()[name = string("op_4025_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4026_cast_fp16 = add(x = variance_109_cast_fp16, y = var_4025_to_fp16)[name = string("op_4026_cast_fp16")]; + fp32 var_4027_epsilon_0 = const()[name = string("op_4027_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4027_cast_fp16 = rsqrt(epsilon = var_4027_epsilon_0, x = var_4026_cast_fp16)[name = string("op_4027_cast_fp16")]; + tensor hidden_states_135_cast_fp16 = mul(x = inputs_109_cast_fp16, y = var_4027_cast_fp16)[name = string("hidden_states_135_cast_fp16")]; + tensor query_normed_27_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_135_cast_fp16)[name = string("query_normed_27_cast_fp16")]; + tensor var_4035 = const()[name = string("op_4035"), val = tensor([8, 128, 1, 1])]; + tensor inputs_111_cast_fp16 = reshape(shape = var_4035, x = current_key_53_cast_fp16)[name = string("inputs_111_cast_fp16")]; + tensor inputs_sq_111_cast_fp16 = mul(x = inputs_111_cast_fp16, y = inputs_111_cast_fp16)[name = string("inputs_sq_111_cast_fp16")]; + tensor variance_111_axes_0 = const()[name = string("variance_111_axes_0"), val = tensor([1])]; + bool variance_111_keep_dims_0 = const()[name = string("variance_111_keep_dims_0"), val = bool(true)]; + tensor variance_111_cast_fp16 = reduce_mean(axes = variance_111_axes_0, keep_dims = variance_111_keep_dims_0, x = inputs_sq_111_cast_fp16)[name = string("variance_111_cast_fp16")]; + fp16 var_4041_to_fp16 = const()[name = string("op_4041_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4042_cast_fp16 = add(x = variance_111_cast_fp16, y = var_4041_to_fp16)[name = string("op_4042_cast_fp16")]; + fp32 var_4043_epsilon_0 = const()[name = string("op_4043_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4043_cast_fp16 = rsqrt(epsilon = var_4043_epsilon_0, x = var_4042_cast_fp16)[name = string("op_4043_cast_fp16")]; + tensor hidden_states_137_cast_fp16 = mul(x = inputs_111_cast_fp16, y = var_4043_cast_fp16)[name = string("hidden_states_137_cast_fp16")]; + tensor current_key_normed_27_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_137_cast_fp16)[name = string("current_key_normed_27_cast_fp16")]; + tensor var_4061 = const()[name = string("op_4061"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_105_cast_fp16 = reshape(shape = var_4061, x = query_normed_27_cast_fp16)[name = string("mh_q_105_cast_fp16")]; + tensor var_4063 = const()[name = string("op_4063"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_105_cast_fp16 = reshape(shape = var_4063, x = current_key_normed_27_cast_fp16)[name = string("mh_k_105_cast_fp16")]; + tensor var_4067_cast_fp16 = mul(x = mh_q_105_cast_fp16, y = cos_21_to_fp16)[name = string("op_4067_cast_fp16")]; + tensor var_4072_begin_0 = const()[name = string("op_4072_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4072_end_0 = const()[name = string("op_4072_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_4072_end_mask_0 = const()[name = string("op_4072_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_4072_cast_fp16 = slice_by_index(begin = var_4072_begin_0, end = var_4072_end_0, end_mask = var_4072_end_mask_0, x = mh_q_105_cast_fp16)[name = string("op_4072_cast_fp16")]; + tensor var_4078_begin_0 = const()[name = string("op_4078_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_4078_end_0 = const()[name = string("op_4078_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_4078_end_mask_0 = const()[name = string("op_4078_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_4078_cast_fp16 = slice_by_index(begin = var_4078_begin_0, end = var_4078_end_0, end_mask = var_4078_end_mask_0, x = mh_q_105_cast_fp16)[name = string("op_4078_cast_fp16")]; + fp16 const_276_promoted_to_fp16 = const()[name = string("const_276_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4080_cast_fp16 = mul(x = var_4078_cast_fp16, y = const_276_promoted_to_fp16)[name = string("op_4080_cast_fp16")]; + bool var_4082_interleave_0 = const()[name = string("op_4082_interleave_0"), val = bool(false)]; + tensor var_4082_cast_fp16 = concat(axis = var_3966, interleave = var_4082_interleave_0, values = (var_4080_cast_fp16, var_4072_cast_fp16))[name = string("op_4082_cast_fp16")]; + tensor var_4083_cast_fp16 = mul(x = var_4082_cast_fp16, y = sin_21_to_fp16)[name = string("op_4083_cast_fp16")]; + tensor mh_q_107_cast_fp16 = add(x = var_4067_cast_fp16, y = var_4083_cast_fp16)[name = string("mh_q_107_cast_fp16")]; + tensor var_4085_cast_fp16 = mul(x = mh_k_105_cast_fp16, y = cos_21_to_fp16)[name = string("op_4085_cast_fp16")]; + tensor var_4090_begin_0 = const()[name = string("op_4090_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4090_end_0 = const()[name = string("op_4090_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_4090_end_mask_0 = const()[name = string("op_4090_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_4090_cast_fp16 = slice_by_index(begin = var_4090_begin_0, end = var_4090_end_0, end_mask = var_4090_end_mask_0, x = mh_k_105_cast_fp16)[name = string("op_4090_cast_fp16")]; + tensor var_4096_begin_0 = const()[name = string("op_4096_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_4096_end_0 = const()[name = string("op_4096_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_4096_end_mask_0 = const()[name = string("op_4096_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_4096_cast_fp16 = slice_by_index(begin = var_4096_begin_0, end = var_4096_end_0, end_mask = var_4096_end_mask_0, x = mh_k_105_cast_fp16)[name = string("op_4096_cast_fp16")]; + fp16 const_279_promoted_to_fp16 = const()[name = string("const_279_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4098_cast_fp16 = mul(x = var_4096_cast_fp16, y = const_279_promoted_to_fp16)[name = string("op_4098_cast_fp16")]; + bool var_4100_interleave_0 = const()[name = string("op_4100_interleave_0"), val = bool(false)]; + tensor var_4100_cast_fp16 = concat(axis = var_3966, interleave = var_4100_interleave_0, values = (var_4098_cast_fp16, var_4090_cast_fp16))[name = string("op_4100_cast_fp16")]; + tensor var_4101_cast_fp16 = mul(x = var_4100_cast_fp16, y = sin_21_to_fp16)[name = string("op_4101_cast_fp16")]; + tensor mh_k_107_cast_fp16 = add(x = var_4085_cast_fp16, y = var_4101_cast_fp16)[name = string("mh_k_107_cast_fp16")]; + tensor var_4105 = const()[name = string("op_4105"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_55_cast_fp16 = reshape(shape = var_4105, x = mh_k_107_cast_fp16)[name = string("current_key_55_cast_fp16")]; + tensor var_4111_to_fp16 = const()[name = string("op_4111_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191232)))]; + tensor var_4112_cast_fp16 = mul(x = obj_119_cast_fp16, y = var_4111_to_fp16)[name = string("op_4112_cast_fp16")]; + tensor var_4109_to_fp16 = const()[name = string("op_4109_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191360)))]; + tensor var_4113_cast_fp16 = mul(x = current_key_55_cast_fp16, y = var_4109_to_fp16)[name = string("op_4113_cast_fp16")]; + tensor key_55_cast_fp16 = add(x = var_4112_cast_fp16, y = var_4113_cast_fp16)[name = string("key_55_cast_fp16")]; + tensor var_4115_to_fp16 = const()[name = string("op_4115_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191232)))]; + tensor var_4116_cast_fp16 = mul(x = obj_121_cast_fp16, y = var_4115_to_fp16)[name = string("op_4116_cast_fp16")]; + tensor var_4117_cast_fp16 = mul(x = current_value_27_cast_fp16, y = var_4109_to_fp16)[name = string("op_4117_cast_fp16")]; + tensor value_27_cast_fp16 = add(x = var_4116_cast_fp16, y = var_4117_cast_fp16)[name = string("value_27_cast_fp16")]; + fp16 var_4124_to_fp16 = const()[name = string("op_4124_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_111_cast_fp16 = mul(x = mh_q_107_cast_fp16, y = var_4124_to_fp16)[name = string("mh_q_111_cast_fp16")]; + tensor var_4126 = const()[name = string("op_4126"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_109_cast_fp16 = reshape(shape = var_4126, x = key_55_cast_fp16)[name = string("mh_k_109_cast_fp16")]; + tensor var_4128 = const()[name = string("op_4128"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_53_cast_fp16 = reshape(shape = var_4128, x = value_27_cast_fp16)[name = string("mh_v_53_cast_fp16")]; + tensor transpose_52_perm_0 = const()[name = string("transpose_52_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_26_reps_0 = const()[name = string("tile_26_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_52_cast_fp16 = transpose(perm = transpose_52_perm_0, x = mh_k_109_cast_fp16)[name = string("transpose_401")]; + tensor tile_26_cast_fp16 = tile(reps = tile_26_reps_0, x = transpose_52_cast_fp16)[name = string("tile_26_cast_fp16")]; + tensor concat_65 = const()[name = string("concat_65"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_52_cast_fp16 = reshape(shape = concat_65, x = tile_26_cast_fp16)[name = string("reshape_52_cast_fp16")]; + tensor transpose_53_perm_0 = const()[name = string("transpose_53_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_66 = const()[name = string("concat_66"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_53_cast_fp16 = transpose(perm = transpose_53_perm_0, x = reshape_52_cast_fp16)[name = string("transpose_400")]; + tensor reshape_53_cast_fp16 = reshape(shape = concat_66, x = transpose_53_cast_fp16)[name = string("reshape_53_cast_fp16")]; + tensor transpose_54_perm_0 = const()[name = string("transpose_54_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_27_reps_0 = const()[name = string("tile_27_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_54_cast_fp16 = transpose(perm = transpose_54_perm_0, x = mh_v_53_cast_fp16)[name = string("transpose_399")]; + tensor tile_27_cast_fp16 = tile(reps = tile_27_reps_0, x = transpose_54_cast_fp16)[name = string("tile_27_cast_fp16")]; + tensor concat_67 = const()[name = string("concat_67"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_54_cast_fp16 = reshape(shape = concat_67, x = tile_27_cast_fp16)[name = string("reshape_54_cast_fp16")]; + tensor transpose_55_perm_0 = const()[name = string("transpose_55_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_68 = const()[name = string("concat_68"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_55_cast_fp16 = transpose(perm = transpose_55_perm_0, x = reshape_54_cast_fp16)[name = string("transpose_398")]; + tensor reshape_55_cast_fp16 = reshape(shape = concat_68, x = transpose_55_cast_fp16)[name = string("reshape_55_cast_fp16")]; + tensor transpose_369_perm_0 = const()[name = string("transpose_369_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_79_transpose_x_1 = const()[name = string("mh_w_79_transpose_x_1"), val = bool(true)]; + bool mh_w_79_transpose_y_1 = const()[name = string("mh_w_79_transpose_y_1"), val = bool(false)]; + tensor transpose_369_cast_fp16 = transpose(perm = transpose_369_perm_0, x = reshape_53_cast_fp16)[name = string("transpose_397")]; + tensor mh_w_79_cast_fp16 = matmul(transpose_x = mh_w_79_transpose_x_1, transpose_y = mh_w_79_transpose_y_1, x = mh_q_111_cast_fp16, y = transpose_369_cast_fp16)[name = string("mh_w_79_cast_fp16")]; + tensor var_4136_to_fp16 = const()[name = string("op_4136_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191488)))]; + tensor mh_w_81_cast_fp16 = add(x = mh_w_79_cast_fp16, y = var_4136_to_fp16)[name = string("mh_w_81_cast_fp16")]; + tensor mh_w_83_cast_fp16 = softmax(axis = var_3956, x = mh_w_81_cast_fp16)[name = string("mh_w_83_cast_fp16")]; + tensor transpose_370_perm_0 = const()[name = string("transpose_370_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_27_transpose_x_1 = const()[name = string("attn_27_transpose_x_1"), val = bool(false)]; + bool attn_27_transpose_y_1 = const()[name = string("attn_27_transpose_y_1"), val = bool(true)]; + tensor transpose_370_cast_fp16 = transpose(perm = transpose_370_perm_0, x = reshape_55_cast_fp16)[name = string("transpose_396")]; + tensor attn_27_cast_fp16 = matmul(transpose_x = attn_27_transpose_x_1, transpose_y = attn_27_transpose_y_1, x = transpose_370_cast_fp16, y = mh_w_83_cast_fp16)[name = string("attn_27_cast_fp16")]; + tensor var_4142 = const()[name = string("op_4142"), val = tensor([1, 2048, 1, 1])]; + tensor input_109_cast_fp16 = reshape(shape = var_4142, x = attn_27_cast_fp16)[name = string("input_109_cast_fp16")]; + string obj_123_pad_type_0 = const()[name = string("obj_123_pad_type_0"), val = string("valid")]; + tensor obj_123_strides_0 = const()[name = string("obj_123_strides_0"), val = tensor([1, 1])]; + tensor obj_123_pad_0 = const()[name = string("obj_123_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_123_dilations_0 = const()[name = string("obj_123_dilations_0"), val = tensor([1, 1])]; + int32 obj_123_groups_0 = const()[name = string("obj_123_groups_0"), val = int32(1)]; + tensor obj_123_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_123_dilations_0, groups = obj_123_groups_0, pad = obj_123_pad_0, pad_type = obj_123_pad_type_0, strides = obj_123_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_109_cast_fp16)[name = string("obj_123_cast_fp16")]; + tensor inputs_113_cast_fp16 = add(x = inputs_107_cast_fp16, y = obj_123_cast_fp16)[name = string("inputs_113_cast_fp16")]; + tensor inputs_sq_113_cast_fp16 = mul(x = inputs_113_cast_fp16, y = inputs_113_cast_fp16)[name = string("inputs_sq_113_cast_fp16")]; + tensor variance_113_axes_0 = const()[name = string("variance_113_axes_0"), val = tensor([1])]; + bool variance_113_keep_dims_0 = const()[name = string("variance_113_keep_dims_0"), val = bool(true)]; + tensor variance_113_cast_fp16 = reduce_mean(axes = variance_113_axes_0, keep_dims = variance_113_keep_dims_0, x = inputs_sq_113_cast_fp16)[name = string("variance_113_cast_fp16")]; + fp16 var_4160_to_fp16 = const()[name = string("op_4160_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4161_cast_fp16 = add(x = variance_113_cast_fp16, y = var_4160_to_fp16)[name = string("op_4161_cast_fp16")]; + fp32 var_4162_epsilon_0 = const()[name = string("op_4162_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4162_cast_fp16 = rsqrt(epsilon = var_4162_epsilon_0, x = var_4161_cast_fp16)[name = string("op_4162_cast_fp16")]; + tensor hidden_states_139_cast_fp16 = mul(x = inputs_113_cast_fp16, y = var_4162_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; + tensor input_111_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_139_cast_fp16)[name = string("input_111_cast_fp16")]; + string input_113_pad_type_0 = const()[name = string("input_113_pad_type_0"), val = string("valid")]; + tensor input_113_strides_0 = const()[name = string("input_113_strides_0"), val = tensor([1, 1])]; + tensor input_113_pad_0 = const()[name = string("input_113_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_113_dilations_0 = const()[name = string("input_113_dilations_0"), val = tensor([1, 1])]; + int32 input_113_groups_0 = const()[name = string("input_113_groups_0"), val = int32(1)]; + tensor input_113_cast_fp16 = conv(dilations = input_113_dilations_0, groups = input_113_groups_0, pad = input_113_pad_0, pad_type = input_113_pad_type_0, strides = input_113_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_111_cast_fp16)[name = string("input_113_cast_fp16")]; + tensor var_4176_cast_fp16 = silu(x = input_113_cast_fp16)[name = string("op_4176_cast_fp16")]; + string var_4182_pad_type_0 = const()[name = string("op_4182_pad_type_0"), val = string("valid")]; + tensor var_4182_strides_0 = const()[name = string("op_4182_strides_0"), val = tensor([1, 1])]; + tensor var_4182_pad_0 = const()[name = string("op_4182_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4182_dilations_0 = const()[name = string("op_4182_dilations_0"), val = tensor([1, 1])]; + int32 var_4182_groups_0 = const()[name = string("op_4182_groups_0"), val = int32(1)]; + tensor var_4182_cast_fp16 = conv(dilations = var_4182_dilations_0, groups = var_4182_groups_0, pad = var_4182_pad_0, pad_type = var_4182_pad_type_0, strides = var_4182_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_111_cast_fp16)[name = string("op_4182_cast_fp16")]; + tensor input_115_cast_fp16 = mul(x = var_4176_cast_fp16, y = var_4182_cast_fp16)[name = string("input_115_cast_fp16")]; + string hidden_states_141_pad_type_0 = const()[name = string("hidden_states_141_pad_type_0"), val = string("valid")]; + tensor hidden_states_141_strides_0 = const()[name = string("hidden_states_141_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_141_pad_0 = const()[name = string("hidden_states_141_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_141_dilations_0 = const()[name = string("hidden_states_141_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_141_groups_0 = const()[name = string("hidden_states_141_groups_0"), val = int32(1)]; + tensor hidden_states_141_cast_fp16 = conv(dilations = hidden_states_141_dilations_0, groups = hidden_states_141_groups_0, pad = hidden_states_141_pad_0, pad_type = hidden_states_141_pad_type_0, strides = hidden_states_141_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_115_cast_fp16)[name = string("hidden_states_141_cast_fp16")]; + tensor inputs_115_cast_fp16 = add(x = inputs_113_cast_fp16, y = hidden_states_141_cast_fp16)[name = string("inputs_115_cast_fp16")]; + tensor obj_127_begin_0 = const()[name = string("obj_127_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_127_end_0 = const()[name = string("obj_127_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_127_end_mask_0 = const()[name = string("obj_127_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_127_cast_fp16 = slice_by_index(begin = obj_127_begin_0, end = obj_127_end_0, end_mask = obj_127_end_mask_0, x = key_caches_5_cast_fp16)[name = string("obj_127_cast_fp16")]; + tensor obj_129_begin_0 = const()[name = string("obj_129_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_129_end_0 = const()[name = string("obj_129_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_129_end_mask_0 = const()[name = string("obj_129_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_129_cast_fp16 = slice_by_index(begin = obj_129_begin_0, end = obj_129_end_0, end_mask = obj_129_end_mask_0, x = value_caches_5_cast_fp16)[name = string("obj_129_cast_fp16")]; + int32 var_4230 = const()[name = string("op_4230"), val = int32(3)]; + int32 var_4240 = const()[name = string("op_4240"), val = int32(-2)]; + tensor inputs_sq_115_cast_fp16 = mul(x = inputs_115_cast_fp16, y = inputs_115_cast_fp16)[name = string("inputs_sq_115_cast_fp16")]; + tensor variance_115_axes_0 = const()[name = string("variance_115_axes_0"), val = tensor([1])]; + bool variance_115_keep_dims_0 = const()[name = string("variance_115_keep_dims_0"), val = bool(true)]; + tensor variance_115_cast_fp16 = reduce_mean(axes = variance_115_axes_0, keep_dims = variance_115_keep_dims_0, x = inputs_sq_115_cast_fp16)[name = string("variance_115_cast_fp16")]; + fp16 var_4254_to_fp16 = const()[name = string("op_4254_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4255_cast_fp16 = add(x = variance_115_cast_fp16, y = var_4254_to_fp16)[name = string("op_4255_cast_fp16")]; + fp32 var_4256_epsilon_0 = const()[name = string("op_4256_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4256_cast_fp16 = rsqrt(epsilon = var_4256_epsilon_0, x = var_4255_cast_fp16)[name = string("op_4256_cast_fp16")]; + tensor hidden_states_143_cast_fp16 = mul(x = inputs_115_cast_fp16, y = var_4256_cast_fp16)[name = string("hidden_states_143_cast_fp16")]; + tensor obj_125_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_143_cast_fp16)[name = string("obj_125_cast_fp16")]; + string query_85_pad_type_0 = const()[name = string("query_85_pad_type_0"), val = string("valid")]; + tensor query_85_strides_0 = const()[name = string("query_85_strides_0"), val = tensor([1, 1])]; + tensor query_85_pad_0 = const()[name = string("query_85_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_85_dilations_0 = const()[name = string("query_85_dilations_0"), val = tensor([1, 1])]; + int32 query_85_groups_0 = const()[name = string("query_85_groups_0"), val = int32(1)]; + tensor query_85_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_85_dilations_0, groups = query_85_groups_0, pad = query_85_pad_0, pad_type = query_85_pad_type_0, strides = query_85_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_125_cast_fp16)[name = string("query_85_cast_fp16")]; + string current_key_57_pad_type_0 = const()[name = string("current_key_57_pad_type_0"), val = string("valid")]; + tensor current_key_57_strides_0 = const()[name = string("current_key_57_strides_0"), val = tensor([1, 1])]; + tensor current_key_57_pad_0 = const()[name = string("current_key_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_57_dilations_0 = const()[name = string("current_key_57_dilations_0"), val = tensor([1, 1])]; + int32 current_key_57_groups_0 = const()[name = string("current_key_57_groups_0"), val = int32(1)]; + tensor current_key_57_cast_fp16 = conv(dilations = current_key_57_dilations_0, groups = current_key_57_groups_0, pad = current_key_57_pad_0, pad_type = current_key_57_pad_type_0, strides = current_key_57_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_125_cast_fp16)[name = string("current_key_57_cast_fp16")]; + string current_value_29_pad_type_0 = const()[name = string("current_value_29_pad_type_0"), val = string("valid")]; + tensor current_value_29_strides_0 = const()[name = string("current_value_29_strides_0"), val = tensor([1, 1])]; + tensor current_value_29_pad_0 = const()[name = string("current_value_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_29_dilations_0 = const()[name = string("current_value_29_dilations_0"), val = tensor([1, 1])]; + int32 current_value_29_groups_0 = const()[name = string("current_value_29_groups_0"), val = int32(1)]; + tensor current_value_29_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_29_dilations_0, groups = current_value_29_groups_0, pad = current_value_29_pad_0, pad_type = current_value_29_pad_type_0, strides = current_value_29_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_125_cast_fp16)[name = string("current_value_29_cast_fp16")]; + tensor var_4293 = const()[name = string("op_4293"), val = tensor([16, 128, 1, 1])]; + tensor inputs_117_cast_fp16 = reshape(shape = var_4293, x = query_85_cast_fp16)[name = string("inputs_117_cast_fp16")]; + tensor inputs_sq_117_cast_fp16 = mul(x = inputs_117_cast_fp16, y = inputs_117_cast_fp16)[name = string("inputs_sq_117_cast_fp16")]; + tensor variance_117_axes_0 = const()[name = string("variance_117_axes_0"), val = tensor([1])]; + bool variance_117_keep_dims_0 = const()[name = string("variance_117_keep_dims_0"), val = bool(true)]; + tensor variance_117_cast_fp16 = reduce_mean(axes = variance_117_axes_0, keep_dims = variance_117_keep_dims_0, x = inputs_sq_117_cast_fp16)[name = string("variance_117_cast_fp16")]; + fp16 var_4299_to_fp16 = const()[name = string("op_4299_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4300_cast_fp16 = add(x = variance_117_cast_fp16, y = var_4299_to_fp16)[name = string("op_4300_cast_fp16")]; + fp32 var_4301_epsilon_0 = const()[name = string("op_4301_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4301_cast_fp16 = rsqrt(epsilon = var_4301_epsilon_0, x = var_4300_cast_fp16)[name = string("op_4301_cast_fp16")]; + tensor hidden_states_145_cast_fp16 = mul(x = inputs_117_cast_fp16, y = var_4301_cast_fp16)[name = string("hidden_states_145_cast_fp16")]; + tensor query_normed_29_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_145_cast_fp16)[name = string("query_normed_29_cast_fp16")]; + tensor var_4309 = const()[name = string("op_4309"), val = tensor([8, 128, 1, 1])]; + tensor inputs_119_cast_fp16 = reshape(shape = var_4309, x = current_key_57_cast_fp16)[name = string("inputs_119_cast_fp16")]; + tensor inputs_sq_119_cast_fp16 = mul(x = inputs_119_cast_fp16, y = inputs_119_cast_fp16)[name = string("inputs_sq_119_cast_fp16")]; + tensor variance_119_axes_0 = const()[name = string("variance_119_axes_0"), val = tensor([1])]; + bool variance_119_keep_dims_0 = const()[name = string("variance_119_keep_dims_0"), val = bool(true)]; + tensor variance_119_cast_fp16 = reduce_mean(axes = variance_119_axes_0, keep_dims = variance_119_keep_dims_0, x = inputs_sq_119_cast_fp16)[name = string("variance_119_cast_fp16")]; + fp16 var_4315_to_fp16 = const()[name = string("op_4315_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4316_cast_fp16 = add(x = variance_119_cast_fp16, y = var_4315_to_fp16)[name = string("op_4316_cast_fp16")]; + fp32 var_4317_epsilon_0 = const()[name = string("op_4317_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4317_cast_fp16 = rsqrt(epsilon = var_4317_epsilon_0, x = var_4316_cast_fp16)[name = string("op_4317_cast_fp16")]; + tensor hidden_states_147_cast_fp16 = mul(x = inputs_119_cast_fp16, y = var_4317_cast_fp16)[name = string("hidden_states_147_cast_fp16")]; + tensor current_key_normed_29_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_147_cast_fp16)[name = string("current_key_normed_29_cast_fp16")]; + tensor var_4335 = const()[name = string("op_4335"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_113_cast_fp16 = reshape(shape = var_4335, x = query_normed_29_cast_fp16)[name = string("mh_q_113_cast_fp16")]; + tensor var_4337 = const()[name = string("op_4337"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_113_cast_fp16 = reshape(shape = var_4337, x = current_key_normed_29_cast_fp16)[name = string("mh_k_113_cast_fp16")]; + tensor var_4341_cast_fp16 = mul(x = mh_q_113_cast_fp16, y = cos_21_to_fp16)[name = string("op_4341_cast_fp16")]; + tensor var_4346_begin_0 = const()[name = string("op_4346_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4346_end_0 = const()[name = string("op_4346_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_4346_end_mask_0 = const()[name = string("op_4346_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_4346_cast_fp16 = slice_by_index(begin = var_4346_begin_0, end = var_4346_end_0, end_mask = var_4346_end_mask_0, x = mh_q_113_cast_fp16)[name = string("op_4346_cast_fp16")]; + tensor var_4352_begin_0 = const()[name = string("op_4352_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_4352_end_0 = const()[name = string("op_4352_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_4352_end_mask_0 = const()[name = string("op_4352_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_4352_cast_fp16 = slice_by_index(begin = var_4352_begin_0, end = var_4352_end_0, end_mask = var_4352_end_mask_0, x = mh_q_113_cast_fp16)[name = string("op_4352_cast_fp16")]; + fp16 const_296_promoted_to_fp16 = const()[name = string("const_296_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4354_cast_fp16 = mul(x = var_4352_cast_fp16, y = const_296_promoted_to_fp16)[name = string("op_4354_cast_fp16")]; + bool var_4356_interleave_0 = const()[name = string("op_4356_interleave_0"), val = bool(false)]; + tensor var_4356_cast_fp16 = concat(axis = var_4240, interleave = var_4356_interleave_0, values = (var_4354_cast_fp16, var_4346_cast_fp16))[name = string("op_4356_cast_fp16")]; + tensor var_4357_cast_fp16 = mul(x = var_4356_cast_fp16, y = sin_21_to_fp16)[name = string("op_4357_cast_fp16")]; + tensor mh_q_115_cast_fp16 = add(x = var_4341_cast_fp16, y = var_4357_cast_fp16)[name = string("mh_q_115_cast_fp16")]; + tensor var_4359_cast_fp16 = mul(x = mh_k_113_cast_fp16, y = cos_21_to_fp16)[name = string("op_4359_cast_fp16")]; + tensor var_4364_begin_0 = const()[name = string("op_4364_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4364_end_0 = const()[name = string("op_4364_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_4364_end_mask_0 = const()[name = string("op_4364_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_4364_cast_fp16 = slice_by_index(begin = var_4364_begin_0, end = var_4364_end_0, end_mask = var_4364_end_mask_0, x = mh_k_113_cast_fp16)[name = string("op_4364_cast_fp16")]; + tensor var_4370_begin_0 = const()[name = string("op_4370_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_4370_end_0 = const()[name = string("op_4370_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_4370_end_mask_0 = const()[name = string("op_4370_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_4370_cast_fp16 = slice_by_index(begin = var_4370_begin_0, end = var_4370_end_0, end_mask = var_4370_end_mask_0, x = mh_k_113_cast_fp16)[name = string("op_4370_cast_fp16")]; + fp16 const_299_promoted_to_fp16 = const()[name = string("const_299_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4372_cast_fp16 = mul(x = var_4370_cast_fp16, y = const_299_promoted_to_fp16)[name = string("op_4372_cast_fp16")]; + bool var_4374_interleave_0 = const()[name = string("op_4374_interleave_0"), val = bool(false)]; + tensor var_4374_cast_fp16 = concat(axis = var_4240, interleave = var_4374_interleave_0, values = (var_4372_cast_fp16, var_4364_cast_fp16))[name = string("op_4374_cast_fp16")]; + tensor var_4375_cast_fp16 = mul(x = var_4374_cast_fp16, y = sin_21_to_fp16)[name = string("op_4375_cast_fp16")]; + tensor mh_k_115_cast_fp16 = add(x = var_4359_cast_fp16, y = var_4375_cast_fp16)[name = string("mh_k_115_cast_fp16")]; + tensor var_4379 = const()[name = string("op_4379"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_59_cast_fp16 = reshape(shape = var_4379, x = mh_k_115_cast_fp16)[name = string("current_key_59_cast_fp16")]; + tensor var_4385_to_fp16 = const()[name = string("op_4385_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191232)))]; + tensor var_4386_cast_fp16 = mul(x = obj_127_cast_fp16, y = var_4385_to_fp16)[name = string("op_4386_cast_fp16")]; + tensor var_4383_to_fp16 = const()[name = string("op_4383_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191360)))]; + tensor var_4387_cast_fp16 = mul(x = current_key_59_cast_fp16, y = var_4383_to_fp16)[name = string("op_4387_cast_fp16")]; + tensor key_59_cast_fp16 = add(x = var_4386_cast_fp16, y = var_4387_cast_fp16)[name = string("key_59_cast_fp16")]; + tensor var_4389_to_fp16 = const()[name = string("op_4389_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191232)))]; + tensor var_4390_cast_fp16 = mul(x = obj_129_cast_fp16, y = var_4389_to_fp16)[name = string("op_4390_cast_fp16")]; + tensor var_4391_cast_fp16 = mul(x = current_value_29_cast_fp16, y = var_4383_to_fp16)[name = string("op_4391_cast_fp16")]; + tensor value_29_cast_fp16 = add(x = var_4390_cast_fp16, y = var_4391_cast_fp16)[name = string("value_29_cast_fp16")]; + fp16 var_4398_to_fp16 = const()[name = string("op_4398_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_119_cast_fp16 = mul(x = mh_q_115_cast_fp16, y = var_4398_to_fp16)[name = string("mh_q_119_cast_fp16")]; + tensor var_4400 = const()[name = string("op_4400"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_117_cast_fp16 = reshape(shape = var_4400, x = key_59_cast_fp16)[name = string("mh_k_117_cast_fp16")]; + tensor var_4402 = const()[name = string("op_4402"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_57_cast_fp16 = reshape(shape = var_4402, x = value_29_cast_fp16)[name = string("mh_v_57_cast_fp16")]; + tensor transpose_56_perm_0 = const()[name = string("transpose_56_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_28_reps_0 = const()[name = string("tile_28_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_56_cast_fp16 = transpose(perm = transpose_56_perm_0, x = mh_k_117_cast_fp16)[name = string("transpose_395")]; + tensor tile_28_cast_fp16 = tile(reps = tile_28_reps_0, x = transpose_56_cast_fp16)[name = string("tile_28_cast_fp16")]; + tensor concat_69 = const()[name = string("concat_69"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_56_cast_fp16 = reshape(shape = concat_69, x = tile_28_cast_fp16)[name = string("reshape_56_cast_fp16")]; + tensor transpose_57_perm_0 = const()[name = string("transpose_57_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_70 = const()[name = string("concat_70"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_57_cast_fp16 = transpose(perm = transpose_57_perm_0, x = reshape_56_cast_fp16)[name = string("transpose_394")]; + tensor reshape_57_cast_fp16 = reshape(shape = concat_70, x = transpose_57_cast_fp16)[name = string("reshape_57_cast_fp16")]; + tensor transpose_58_perm_0 = const()[name = string("transpose_58_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_29_reps_0 = const()[name = string("tile_29_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_58_cast_fp16 = transpose(perm = transpose_58_perm_0, x = mh_v_57_cast_fp16)[name = string("transpose_393")]; + tensor tile_29_cast_fp16 = tile(reps = tile_29_reps_0, x = transpose_58_cast_fp16)[name = string("tile_29_cast_fp16")]; + tensor concat_71 = const()[name = string("concat_71"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_58_cast_fp16 = reshape(shape = concat_71, x = tile_29_cast_fp16)[name = string("reshape_58_cast_fp16")]; + tensor transpose_59_perm_0 = const()[name = string("transpose_59_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_72 = const()[name = string("concat_72"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_59_cast_fp16 = transpose(perm = transpose_59_perm_0, x = reshape_58_cast_fp16)[name = string("transpose_392")]; + tensor reshape_59_cast_fp16 = reshape(shape = concat_72, x = transpose_59_cast_fp16)[name = string("reshape_59_cast_fp16")]; + tensor transpose_373_perm_0 = const()[name = string("transpose_373_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_85_transpose_x_1 = const()[name = string("mh_w_85_transpose_x_1"), val = bool(true)]; + bool mh_w_85_transpose_y_1 = const()[name = string("mh_w_85_transpose_y_1"), val = bool(false)]; + tensor transpose_373_cast_fp16 = transpose(perm = transpose_373_perm_0, x = reshape_57_cast_fp16)[name = string("transpose_391")]; + tensor mh_w_85_cast_fp16 = matmul(transpose_x = mh_w_85_transpose_x_1, transpose_y = mh_w_85_transpose_y_1, x = mh_q_119_cast_fp16, y = transpose_373_cast_fp16)[name = string("mh_w_85_cast_fp16")]; + tensor var_4410_to_fp16 = const()[name = string("op_4410_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191488)))]; + tensor mh_w_87_cast_fp16 = add(x = mh_w_85_cast_fp16, y = var_4410_to_fp16)[name = string("mh_w_87_cast_fp16")]; + tensor mh_w_89_cast_fp16 = softmax(axis = var_4230, x = mh_w_87_cast_fp16)[name = string("mh_w_89_cast_fp16")]; + tensor transpose_374_perm_0 = const()[name = string("transpose_374_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_29_transpose_x_1 = const()[name = string("attn_29_transpose_x_1"), val = bool(false)]; + bool attn_29_transpose_y_1 = const()[name = string("attn_29_transpose_y_1"), val = bool(true)]; + tensor transpose_374_cast_fp16 = transpose(perm = transpose_374_perm_0, x = reshape_59_cast_fp16)[name = string("transpose_390")]; + tensor attn_29_cast_fp16 = matmul(transpose_x = attn_29_transpose_x_1, transpose_y = attn_29_transpose_y_1, x = transpose_374_cast_fp16, y = mh_w_89_cast_fp16)[name = string("attn_29_cast_fp16")]; + tensor var_4416 = const()[name = string("op_4416"), val = tensor([1, 2048, 1, 1])]; + tensor input_117_cast_fp16 = reshape(shape = var_4416, x = attn_29_cast_fp16)[name = string("input_117_cast_fp16")]; + string obj_131_pad_type_0 = const()[name = string("obj_131_pad_type_0"), val = string("valid")]; + tensor obj_131_strides_0 = const()[name = string("obj_131_strides_0"), val = tensor([1, 1])]; + tensor obj_131_pad_0 = const()[name = string("obj_131_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_131_dilations_0 = const()[name = string("obj_131_dilations_0"), val = tensor([1, 1])]; + int32 obj_131_groups_0 = const()[name = string("obj_131_groups_0"), val = int32(1)]; + tensor obj_131_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_131_dilations_0, groups = obj_131_groups_0, pad = obj_131_pad_0, pad_type = obj_131_pad_type_0, strides = obj_131_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_117_cast_fp16)[name = string("obj_131_cast_fp16")]; + tensor inputs_121_cast_fp16 = add(x = inputs_115_cast_fp16, y = obj_131_cast_fp16)[name = string("inputs_121_cast_fp16")]; + tensor inputs_sq_121_cast_fp16 = mul(x = inputs_121_cast_fp16, y = inputs_121_cast_fp16)[name = string("inputs_sq_121_cast_fp16")]; + tensor variance_121_axes_0 = const()[name = string("variance_121_axes_0"), val = tensor([1])]; + bool variance_121_keep_dims_0 = const()[name = string("variance_121_keep_dims_0"), val = bool(true)]; + tensor variance_121_cast_fp16 = reduce_mean(axes = variance_121_axes_0, keep_dims = variance_121_keep_dims_0, x = inputs_sq_121_cast_fp16)[name = string("variance_121_cast_fp16")]; + fp16 var_4434_to_fp16 = const()[name = string("op_4434_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4435_cast_fp16 = add(x = variance_121_cast_fp16, y = var_4434_to_fp16)[name = string("op_4435_cast_fp16")]; + fp32 var_4436_epsilon_0 = const()[name = string("op_4436_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4436_cast_fp16 = rsqrt(epsilon = var_4436_epsilon_0, x = var_4435_cast_fp16)[name = string("op_4436_cast_fp16")]; + tensor hidden_states_149_cast_fp16 = mul(x = inputs_121_cast_fp16, y = var_4436_cast_fp16)[name = string("hidden_states_149_cast_fp16")]; + tensor input_119_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_149_cast_fp16)[name = string("input_119_cast_fp16")]; + string input_121_pad_type_0 = const()[name = string("input_121_pad_type_0"), val = string("valid")]; + tensor input_121_strides_0 = const()[name = string("input_121_strides_0"), val = tensor([1, 1])]; + tensor input_121_pad_0 = const()[name = string("input_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_121_dilations_0 = const()[name = string("input_121_dilations_0"), val = tensor([1, 1])]; + int32 input_121_groups_0 = const()[name = string("input_121_groups_0"), val = int32(1)]; + tensor input_121_cast_fp16 = conv(dilations = input_121_dilations_0, groups = input_121_groups_0, pad = input_121_pad_0, pad_type = input_121_pad_type_0, strides = input_121_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_119_cast_fp16)[name = string("input_121_cast_fp16")]; + tensor var_4450_cast_fp16 = silu(x = input_121_cast_fp16)[name = string("op_4450_cast_fp16")]; + string var_4456_pad_type_0 = const()[name = string("op_4456_pad_type_0"), val = string("valid")]; + tensor var_4456_strides_0 = const()[name = string("op_4456_strides_0"), val = tensor([1, 1])]; + tensor var_4456_pad_0 = const()[name = string("op_4456_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4456_dilations_0 = const()[name = string("op_4456_dilations_0"), val = tensor([1, 1])]; + int32 var_4456_groups_0 = const()[name = string("op_4456_groups_0"), val = int32(1)]; + tensor var_4456_cast_fp16 = conv(dilations = var_4456_dilations_0, groups = var_4456_groups_0, pad = var_4456_pad_0, pad_type = var_4456_pad_type_0, strides = var_4456_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_119_cast_fp16)[name = string("op_4456_cast_fp16")]; + tensor input_123_cast_fp16 = mul(x = var_4450_cast_fp16, y = var_4456_cast_fp16)[name = string("input_123_cast_fp16")]; + string hidden_states_151_pad_type_0 = const()[name = string("hidden_states_151_pad_type_0"), val = string("valid")]; + tensor hidden_states_151_strides_0 = const()[name = string("hidden_states_151_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_151_pad_0 = const()[name = string("hidden_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_151_dilations_0 = const()[name = string("hidden_states_151_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_151_groups_0 = const()[name = string("hidden_states_151_groups_0"), val = int32(1)]; + tensor hidden_states_151_cast_fp16 = conv(dilations = hidden_states_151_dilations_0, groups = hidden_states_151_groups_0, pad = hidden_states_151_pad_0, pad_type = hidden_states_151_pad_type_0, strides = hidden_states_151_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_123_cast_fp16)[name = string("hidden_states_151_cast_fp16")]; + tensor inputs_123_cast_fp16 = add(x = inputs_121_cast_fp16, y = hidden_states_151_cast_fp16)[name = string("inputs_123_cast_fp16")]; + int32 var_4484 = const()[name = string("op_4484"), val = int32(1)]; + bool key_caches_7_interleave_0 = const()[name = string("key_caches_7_interleave_0"), val = bool(false)]; + tensor key_caches_7_cast_fp16 = concat(axis = var_4484, interleave = key_caches_7_interleave_0, values = (key_43_cast_fp16, key_47_cast_fp16, key_51_cast_fp16, key_55_cast_fp16, key_59_cast_fp16))[name = string("key_caches_7_cast_fp16")]; + int32 var_4487 = const()[name = string("op_4487"), val = int32(1)]; + bool value_caches_7_interleave_0 = const()[name = string("value_caches_7_interleave_0"), val = bool(false)]; + tensor value_caches_7_cast_fp16 = concat(axis = var_4487, interleave = value_caches_7_interleave_0, values = (value_21_cast_fp16, value_23_cast_fp16, value_25_cast_fp16, value_27_cast_fp16, value_29_cast_fp16))[name = string("value_caches_7_cast_fp16")]; + tensor inputs_sq_123_cast_fp16 = mul(x = inputs_123_cast_fp16, y = inputs_123_cast_fp16)[name = string("inputs_sq_123_cast_fp16")]; + tensor variance_123_axes_0 = const()[name = string("variance_123_axes_0"), val = tensor([1])]; + bool variance_123_keep_dims_0 = const()[name = string("variance_123_keep_dims_0"), val = bool(true)]; + tensor variance_123_cast_fp16 = reduce_mean(axes = variance_123_axes_0, keep_dims = variance_123_keep_dims_0, x = inputs_sq_123_cast_fp16)[name = string("variance_123_cast_fp16")]; + fp16 var_4497_to_fp16 = const()[name = string("op_4497_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4498_cast_fp16 = add(x = variance_123_cast_fp16, y = var_4497_to_fp16)[name = string("op_4498_cast_fp16")]; + fp32 var_4499_epsilon_0 = const()[name = string("op_4499_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4499_cast_fp16 = rsqrt(epsilon = var_4499_epsilon_0, x = var_4498_cast_fp16)[name = string("op_4499_cast_fp16")]; + tensor hidden_states_153_cast_fp16 = mul(x = inputs_123_cast_fp16, y = var_4499_cast_fp16)[name = string("hidden_states_153_cast_fp16")]; + tensor input_125_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_153_cast_fp16)[name = string("input_125_cast_fp16")]; + string logits_5_pad_type_0 = const()[name = string("logits_5_pad_type_0"), val = string("valid")]; + tensor logits_5_strides_0 = const()[name = string("logits_5_strides_0"), val = tensor([1, 1])]; + tensor logits_5_pad_0 = const()[name = string("logits_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_5_dilations_0 = const()[name = string("logits_5_dilations_0"), val = tensor([1, 1])]; + int32 logits_5_groups_0 = const()[name = string("logits_5_groups_0"), val = int32(1)]; + tensor lm_heads_1_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82904384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85001600))))[name = string("lm_heads_1_weight_to_fp16_palettized")]; + tensor logits_5_cast_fp16 = conv(dilations = logits_5_dilations_0, groups = logits_5_groups_0, pad = logits_5_pad_0, pad_type = logits_5_pad_type_0, strides = logits_5_strides_0, weight = lm_heads_1_weight_to_fp16_palettized, x = input_125_cast_fp16)[name = string("logits_5_cast_fp16")]; + tensor var_4517 = const()[name = string("op_4517"), val = tensor([1, 2048])]; + tensor logits_7_cast_fp16 = reshape(shape = var_4517, x = logits_5_cast_fp16)[name = string("logits_7_cast_fp16")]; + tensor scaled_logits_3_cast_fp16 = real_div(x = logits_7_cast_fp16, y = temperature)[name = string("scaled_logits_3_cast_fp16")]; + int32 var_4527 = const()[name = string("op_4527"), val = int32(100)]; + int32 top_values_3_axis_0 = const()[name = string("top_values_3_axis_0"), val = int32(1)]; + bool top_values_3_ascending_0 = const()[name = string("top_values_3_ascending_0"), val = bool(false)]; + bool top_values_3_sort_0 = const()[name = string("top_values_3_sort_0"), val = bool(true)]; + bool top_values_3_return_indices_0 = const()[name = string("top_values_3_return_indices_0"), val = bool(true)]; + string top_values_3_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_3_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_3_cast_fp16_cast_uint16_0, tensor top_values_3_cast_fp16_cast_uint16_1 = topk(ascending = top_values_3_ascending_0, axis = top_values_3_axis_0, k = var_4527, output_indices_dtype = top_values_3_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_3_return_indices_0, sort = top_values_3_sort_0, x = scaled_logits_3_cast_fp16)[name = string("top_values_3_cast_fp16_cast_uint16")]; + tensor var_4533_cast_fp16 = mul(x = top_values_3_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_4533_cast_fp16")]; + tensor var_4537_cast_fp16 = add(x = var_4533_cast_fp16, y = var_3001_cast_fp16)[name = string("op_4537_cast_fp16")]; + tensor reduce_min_1_axes_0 = const()[name = string("reduce_min_1_axes_0"), val = tensor([1])]; + bool reduce_min_1_keep_dims_0 = const()[name = string("reduce_min_1_keep_dims_0"), val = bool(true)]; + tensor reduce_min_1_cast_fp16 = reduce_min(axes = reduce_min_1_axes_0, keep_dims = reduce_min_1_keep_dims_0, x = var_4537_cast_fp16)[name = string("reduce_min_1_cast_fp16")]; + tensor var_4540_cast_fp16 = greater_equal(x = scaled_logits_3_cast_fp16, y = reduce_min_1_cast_fp16)[name = string("op_4540_cast_fp16")]; + fp16 var_4541_value_0_to_fp16 = const()[name = string("op_4541_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_4541_cast_fp16 = fill_like(ref_tensor = scaled_logits_3_cast_fp16, value = var_4541_value_0_to_fp16)[name = string("op_4541_cast_fp16")]; + tensor masked_logits_3_cast_fp16 = select(a = scaled_logits_3_cast_fp16, b = var_4541_cast_fp16, cond = var_4540_cast_fp16)[name = string("masked_logits_3_cast_fp16")]; + tensor var_4545_begin_0 = const()[name = string("op_4545_begin_0"), val = tensor([1, 0])]; + tensor var_4545_end_0 = const()[name = string("op_4545_end_0"), val = tensor([2, 2048])]; + tensor var_4545_end_mask_0 = const()[name = string("op_4545_end_mask_0"), val = tensor([false, true])]; + tensor var_4545_squeeze_mask_0 = const()[name = string("op_4545_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_4545_cast_fp16 = slice_by_index(begin = var_4545_begin_0, end = var_4545_end_0, end_mask = var_4545_end_mask_0, squeeze_mask = var_4545_squeeze_mask_0, x = gumbel)[name = string("op_4545_cast_fp16")]; + tensor var_4548 = const()[name = string("op_4548"), val = tensor([1, 2048])]; + tensor var_4549_cast_fp16 = reshape(shape = var_4548, x = var_4545_cast_fp16)[name = string("op_4549_cast_fp16")]; + tensor noisy_logits_3_cast_fp16 = add(x = masked_logits_3_cast_fp16, y = var_4549_cast_fp16)[name = string("noisy_logits_3_cast_fp16")]; + int32 code_3_axis_0 = const()[name = string("code_3_axis_0"), val = int32(1)]; + bool code_3_keep_dims_0 = const()[name = string("code_3_keep_dims_0"), val = bool(false)]; + string code_3_output_dtype_0 = const()[name = string("code_3_output_dtype_0"), val = string("int32")]; + tensor code_3_cast_fp16 = reduce_argmax(axis = code_3_axis_0, keep_dims = code_3_keep_dims_0, output_dtype = code_3_output_dtype_0, x = noisy_logits_3_cast_fp16)[name = string("code_3_cast_fp16")]; + int32 var_4560 = const()[name = string("op_4560"), val = int32(2048)]; + tensor input_127 = add(x = code_3_cast_fp16, y = var_4560)[name = string("input_127")]; + int32 code_embed_5_axis_0 = const()[name = string("code_embed_5_axis_0"), val = int32(0)]; + int32 code_embed_5_batch_dims_0 = const()[name = string("code_embed_5_batch_dims_0"), val = int32(0)]; + bool code_embed_5_validate_indices_0 = const()[name = string("code_embed_5_validate_indices_0"), val = bool(false)]; + string input_127_to_uint16_dtype_0 = const()[name = string("input_127_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_127_to_uint16 = cast(dtype = input_127_to_uint16_dtype_0, x = input_127)[name = string("cast_13")]; + tensor code_embed_5_cast_fp16_cast_uint16 = gather(axis = code_embed_5_axis_0, batch_dims = code_embed_5_batch_dims_0, indices = input_127_to_uint16, validate_indices = code_embed_5_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_5_cast_fp16_cast_uint16")]; + tensor var_4564 = const()[name = string("op_4564"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_7_cast_fp16 = reshape(shape = var_4564, x = code_embed_5_cast_fp16_cast_uint16)[name = string("code_embed_7_cast_fp16")]; + tensor embed_sum_5_cast_fp16 = add(x = code_embed_3_cast_fp16, y = code_embed_7_cast_fp16)[name = string("embed_sum_5_cast_fp16")]; + string inputs_125_pad_type_0 = const()[name = string("inputs_125_pad_type_0"), val = string("valid")]; + tensor inputs_125_strides_0 = const()[name = string("inputs_125_strides_0"), val = tensor([1, 1])]; + tensor inputs_125_pad_0 = const()[name = string("inputs_125_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_125_dilations_0 = const()[name = string("inputs_125_dilations_0"), val = tensor([1, 1])]; + int32 inputs_125_groups_0 = const()[name = string("inputs_125_groups_0"), val = int32(1)]; + tensor inputs_125_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_125_dilations_0, groups = inputs_125_groups_0, pad = inputs_125_pad_0, pad_type = inputs_125_pad_type_0, strides = inputs_125_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_7_cast_fp16)[name = string("inputs_125_cast_fp16")]; + tensor obj_135_begin_0 = const()[name = string("obj_135_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_135_end_0 = const()[name = string("obj_135_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_135_end_mask_0 = const()[name = string("obj_135_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_135_cast_fp16 = slice_by_index(begin = obj_135_begin_0, end = obj_135_end_0, end_mask = obj_135_end_mask_0, x = key_caches_7_cast_fp16)[name = string("obj_135_cast_fp16")]; + tensor obj_137_begin_0 = const()[name = string("obj_137_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_137_end_0 = const()[name = string("obj_137_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_137_end_mask_0 = const()[name = string("obj_137_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_137_cast_fp16 = slice_by_index(begin = obj_137_begin_0, end = obj_137_end_0, end_mask = obj_137_end_mask_0, x = value_caches_7_cast_fp16)[name = string("obj_137_cast_fp16")]; + int32 var_4669 = const()[name = string("op_4669"), val = int32(3)]; + int32 var_4679 = const()[name = string("op_4679"), val = int32(-2)]; + tensor inputs_sq_125_cast_fp16 = mul(x = inputs_125_cast_fp16, y = inputs_125_cast_fp16)[name = string("inputs_sq_125_cast_fp16")]; + tensor variance_125_axes_0 = const()[name = string("variance_125_axes_0"), val = tensor([1])]; + bool variance_125_keep_dims_0 = const()[name = string("variance_125_keep_dims_0"), val = bool(true)]; + tensor variance_125_cast_fp16 = reduce_mean(axes = variance_125_axes_0, keep_dims = variance_125_keep_dims_0, x = inputs_sq_125_cast_fp16)[name = string("variance_125_cast_fp16")]; + fp16 var_4693_to_fp16 = const()[name = string("op_4693_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4694_cast_fp16 = add(x = variance_125_cast_fp16, y = var_4693_to_fp16)[name = string("op_4694_cast_fp16")]; + fp32 var_4695_epsilon_0 = const()[name = string("op_4695_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4695_cast_fp16 = rsqrt(epsilon = var_4695_epsilon_0, x = var_4694_cast_fp16)[name = string("op_4695_cast_fp16")]; + tensor hidden_states_155_cast_fp16 = mul(x = inputs_125_cast_fp16, y = var_4695_cast_fp16)[name = string("hidden_states_155_cast_fp16")]; + tensor obj_133_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_155_cast_fp16)[name = string("obj_133_cast_fp16")]; + string query_91_pad_type_0 = const()[name = string("query_91_pad_type_0"), val = string("valid")]; + tensor query_91_strides_0 = const()[name = string("query_91_strides_0"), val = tensor([1, 1])]; + tensor query_91_pad_0 = const()[name = string("query_91_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_91_dilations_0 = const()[name = string("query_91_dilations_0"), val = tensor([1, 1])]; + int32 query_91_groups_0 = const()[name = string("query_91_groups_0"), val = int32(1)]; + tensor query_91_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_91_dilations_0, groups = query_91_groups_0, pad = query_91_pad_0, pad_type = query_91_pad_type_0, strides = query_91_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_133_cast_fp16)[name = string("query_91_cast_fp16")]; + string current_key_61_pad_type_0 = const()[name = string("current_key_61_pad_type_0"), val = string("valid")]; + tensor current_key_61_strides_0 = const()[name = string("current_key_61_strides_0"), val = tensor([1, 1])]; + tensor current_key_61_pad_0 = const()[name = string("current_key_61_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_61_dilations_0 = const()[name = string("current_key_61_dilations_0"), val = tensor([1, 1])]; + int32 current_key_61_groups_0 = const()[name = string("current_key_61_groups_0"), val = int32(1)]; + tensor current_key_61_cast_fp16 = conv(dilations = current_key_61_dilations_0, groups = current_key_61_groups_0, pad = current_key_61_pad_0, pad_type = current_key_61_pad_type_0, strides = current_key_61_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_133_cast_fp16)[name = string("current_key_61_cast_fp16")]; + string current_value_31_pad_type_0 = const()[name = string("current_value_31_pad_type_0"), val = string("valid")]; + tensor current_value_31_strides_0 = const()[name = string("current_value_31_strides_0"), val = tensor([1, 1])]; + tensor current_value_31_pad_0 = const()[name = string("current_value_31_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_31_dilations_0 = const()[name = string("current_value_31_dilations_0"), val = tensor([1, 1])]; + int32 current_value_31_groups_0 = const()[name = string("current_value_31_groups_0"), val = int32(1)]; + tensor current_value_31_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_31_dilations_0, groups = current_value_31_groups_0, pad = current_value_31_pad_0, pad_type = current_value_31_pad_type_0, strides = current_value_31_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_133_cast_fp16)[name = string("current_value_31_cast_fp16")]; + tensor var_4732 = const()[name = string("op_4732"), val = tensor([16, 128, 1, 1])]; + tensor inputs_127_cast_fp16 = reshape(shape = var_4732, x = query_91_cast_fp16)[name = string("inputs_127_cast_fp16")]; + tensor inputs_sq_127_cast_fp16 = mul(x = inputs_127_cast_fp16, y = inputs_127_cast_fp16)[name = string("inputs_sq_127_cast_fp16")]; + tensor variance_127_axes_0 = const()[name = string("variance_127_axes_0"), val = tensor([1])]; + bool variance_127_keep_dims_0 = const()[name = string("variance_127_keep_dims_0"), val = bool(true)]; + tensor variance_127_cast_fp16 = reduce_mean(axes = variance_127_axes_0, keep_dims = variance_127_keep_dims_0, x = inputs_sq_127_cast_fp16)[name = string("variance_127_cast_fp16")]; + fp16 var_4738_to_fp16 = const()[name = string("op_4738_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4739_cast_fp16 = add(x = variance_127_cast_fp16, y = var_4738_to_fp16)[name = string("op_4739_cast_fp16")]; + fp32 var_4740_epsilon_0 = const()[name = string("op_4740_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4740_cast_fp16 = rsqrt(epsilon = var_4740_epsilon_0, x = var_4739_cast_fp16)[name = string("op_4740_cast_fp16")]; + tensor hidden_states_157_cast_fp16 = mul(x = inputs_127_cast_fp16, y = var_4740_cast_fp16)[name = string("hidden_states_157_cast_fp16")]; + tensor query_normed_31_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_157_cast_fp16)[name = string("query_normed_31_cast_fp16")]; + tensor var_4748 = const()[name = string("op_4748"), val = tensor([8, 128, 1, 1])]; + tensor inputs_129_cast_fp16 = reshape(shape = var_4748, x = current_key_61_cast_fp16)[name = string("inputs_129_cast_fp16")]; + tensor inputs_sq_129_cast_fp16 = mul(x = inputs_129_cast_fp16, y = inputs_129_cast_fp16)[name = string("inputs_sq_129_cast_fp16")]; + tensor variance_129_axes_0 = const()[name = string("variance_129_axes_0"), val = tensor([1])]; + bool variance_129_keep_dims_0 = const()[name = string("variance_129_keep_dims_0"), val = bool(true)]; + tensor variance_129_cast_fp16 = reduce_mean(axes = variance_129_axes_0, keep_dims = variance_129_keep_dims_0, x = inputs_sq_129_cast_fp16)[name = string("variance_129_cast_fp16")]; + fp16 var_4754_to_fp16 = const()[name = string("op_4754_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4755_cast_fp16 = add(x = variance_129_cast_fp16, y = var_4754_to_fp16)[name = string("op_4755_cast_fp16")]; + fp32 var_4756_epsilon_0 = const()[name = string("op_4756_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4756_cast_fp16 = rsqrt(epsilon = var_4756_epsilon_0, x = var_4755_cast_fp16)[name = string("op_4756_cast_fp16")]; + tensor hidden_states_159_cast_fp16 = mul(x = inputs_129_cast_fp16, y = var_4756_cast_fp16)[name = string("hidden_states_159_cast_fp16")]; + tensor current_key_normed_31_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_159_cast_fp16)[name = string("current_key_normed_31_cast_fp16")]; + tensor var_4774 = const()[name = string("op_4774"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_121_cast_fp16 = reshape(shape = var_4774, x = query_normed_31_cast_fp16)[name = string("mh_q_121_cast_fp16")]; + tensor var_4776 = const()[name = string("op_4776"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_121_cast_fp16 = reshape(shape = var_4776, x = current_key_normed_31_cast_fp16)[name = string("mh_k_121_cast_fp16")]; + tensor cos_31_to_fp16 = const()[name = string("cos_31_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191616)))]; + tensor var_4780_cast_fp16 = mul(x = mh_q_121_cast_fp16, y = cos_31_to_fp16)[name = string("op_4780_cast_fp16")]; + tensor var_4785_begin_0 = const()[name = string("op_4785_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4785_end_0 = const()[name = string("op_4785_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_4785_end_mask_0 = const()[name = string("op_4785_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_4785_cast_fp16 = slice_by_index(begin = var_4785_begin_0, end = var_4785_end_0, end_mask = var_4785_end_mask_0, x = mh_q_121_cast_fp16)[name = string("op_4785_cast_fp16")]; + tensor var_4791_begin_0 = const()[name = string("op_4791_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_4791_end_0 = const()[name = string("op_4791_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_4791_end_mask_0 = const()[name = string("op_4791_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_4791_cast_fp16 = slice_by_index(begin = var_4791_begin_0, end = var_4791_end_0, end_mask = var_4791_end_mask_0, x = mh_q_121_cast_fp16)[name = string("op_4791_cast_fp16")]; + fp16 const_317_promoted_to_fp16 = const()[name = string("const_317_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4793_cast_fp16 = mul(x = var_4791_cast_fp16, y = const_317_promoted_to_fp16)[name = string("op_4793_cast_fp16")]; + bool var_4795_interleave_0 = const()[name = string("op_4795_interleave_0"), val = bool(false)]; + tensor var_4795_cast_fp16 = concat(axis = var_4679, interleave = var_4795_interleave_0, values = (var_4793_cast_fp16, var_4785_cast_fp16))[name = string("op_4795_cast_fp16")]; + tensor sin_31_to_fp16 = const()[name = string("sin_31_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175191936)))]; + tensor var_4796_cast_fp16 = mul(x = var_4795_cast_fp16, y = sin_31_to_fp16)[name = string("op_4796_cast_fp16")]; + tensor mh_q_123_cast_fp16 = add(x = var_4780_cast_fp16, y = var_4796_cast_fp16)[name = string("mh_q_123_cast_fp16")]; + tensor var_4798_cast_fp16 = mul(x = mh_k_121_cast_fp16, y = cos_31_to_fp16)[name = string("op_4798_cast_fp16")]; + tensor var_4803_begin_0 = const()[name = string("op_4803_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4803_end_0 = const()[name = string("op_4803_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_4803_end_mask_0 = const()[name = string("op_4803_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_4803_cast_fp16 = slice_by_index(begin = var_4803_begin_0, end = var_4803_end_0, end_mask = var_4803_end_mask_0, x = mh_k_121_cast_fp16)[name = string("op_4803_cast_fp16")]; + tensor var_4809_begin_0 = const()[name = string("op_4809_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_4809_end_0 = const()[name = string("op_4809_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_4809_end_mask_0 = const()[name = string("op_4809_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_4809_cast_fp16 = slice_by_index(begin = var_4809_begin_0, end = var_4809_end_0, end_mask = var_4809_end_mask_0, x = mh_k_121_cast_fp16)[name = string("op_4809_cast_fp16")]; + fp16 const_320_promoted_to_fp16 = const()[name = string("const_320_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4811_cast_fp16 = mul(x = var_4809_cast_fp16, y = const_320_promoted_to_fp16)[name = string("op_4811_cast_fp16")]; + bool var_4813_interleave_0 = const()[name = string("op_4813_interleave_0"), val = bool(false)]; + tensor var_4813_cast_fp16 = concat(axis = var_4679, interleave = var_4813_interleave_0, values = (var_4811_cast_fp16, var_4803_cast_fp16))[name = string("op_4813_cast_fp16")]; + tensor var_4814_cast_fp16 = mul(x = var_4813_cast_fp16, y = sin_31_to_fp16)[name = string("op_4814_cast_fp16")]; + tensor mh_k_123_cast_fp16 = add(x = var_4798_cast_fp16, y = var_4814_cast_fp16)[name = string("mh_k_123_cast_fp16")]; + tensor var_4818 = const()[name = string("op_4818"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_63_cast_fp16 = reshape(shape = var_4818, x = mh_k_123_cast_fp16)[name = string("current_key_63_cast_fp16")]; + tensor var_4824_to_fp16 = const()[name = string("op_4824_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192256)))]; + tensor var_4825_cast_fp16 = mul(x = obj_135_cast_fp16, y = var_4824_to_fp16)[name = string("op_4825_cast_fp16")]; + tensor var_4822_to_fp16 = const()[name = string("op_4822_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192384)))]; + tensor var_4826_cast_fp16 = mul(x = current_key_63_cast_fp16, y = var_4822_to_fp16)[name = string("op_4826_cast_fp16")]; + tensor key_63_cast_fp16 = add(x = var_4825_cast_fp16, y = var_4826_cast_fp16)[name = string("key_63_cast_fp16")]; + tensor var_4828_to_fp16 = const()[name = string("op_4828_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192256)))]; + tensor var_4829_cast_fp16 = mul(x = obj_137_cast_fp16, y = var_4828_to_fp16)[name = string("op_4829_cast_fp16")]; + tensor var_4830_cast_fp16 = mul(x = current_value_31_cast_fp16, y = var_4822_to_fp16)[name = string("op_4830_cast_fp16")]; + tensor value_31_cast_fp16 = add(x = var_4829_cast_fp16, y = var_4830_cast_fp16)[name = string("value_31_cast_fp16")]; + fp16 var_4837_to_fp16 = const()[name = string("op_4837_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_127_cast_fp16 = mul(x = mh_q_123_cast_fp16, y = var_4837_to_fp16)[name = string("mh_q_127_cast_fp16")]; + tensor var_4839 = const()[name = string("op_4839"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_125_cast_fp16 = reshape(shape = var_4839, x = key_63_cast_fp16)[name = string("mh_k_125_cast_fp16")]; + tensor var_4841 = const()[name = string("op_4841"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_61_cast_fp16 = reshape(shape = var_4841, x = value_31_cast_fp16)[name = string("mh_v_61_cast_fp16")]; + tensor transpose_60_perm_0 = const()[name = string("transpose_60_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_30_reps_0 = const()[name = string("tile_30_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_60_cast_fp16 = transpose(perm = transpose_60_perm_0, x = mh_k_125_cast_fp16)[name = string("transpose_389")]; + tensor tile_30_cast_fp16 = tile(reps = tile_30_reps_0, x = transpose_60_cast_fp16)[name = string("tile_30_cast_fp16")]; + tensor concat_78 = const()[name = string("concat_78"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_60_cast_fp16 = reshape(shape = concat_78, x = tile_30_cast_fp16)[name = string("reshape_60_cast_fp16")]; + tensor transpose_61_perm_0 = const()[name = string("transpose_61_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_79 = const()[name = string("concat_79"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_61_cast_fp16 = transpose(perm = transpose_61_perm_0, x = reshape_60_cast_fp16)[name = string("transpose_388")]; + tensor reshape_61_cast_fp16 = reshape(shape = concat_79, x = transpose_61_cast_fp16)[name = string("reshape_61_cast_fp16")]; + tensor transpose_62_perm_0 = const()[name = string("transpose_62_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_31_reps_0 = const()[name = string("tile_31_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_62_cast_fp16 = transpose(perm = transpose_62_perm_0, x = mh_v_61_cast_fp16)[name = string("transpose_387")]; + tensor tile_31_cast_fp16 = tile(reps = tile_31_reps_0, x = transpose_62_cast_fp16)[name = string("tile_31_cast_fp16")]; + tensor concat_80 = const()[name = string("concat_80"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_62_cast_fp16 = reshape(shape = concat_80, x = tile_31_cast_fp16)[name = string("reshape_62_cast_fp16")]; + tensor transpose_63_perm_0 = const()[name = string("transpose_63_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_81 = const()[name = string("concat_81"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_63_cast_fp16 = transpose(perm = transpose_63_perm_0, x = reshape_62_cast_fp16)[name = string("transpose_386")]; + tensor reshape_63_cast_fp16 = reshape(shape = concat_81, x = transpose_63_cast_fp16)[name = string("reshape_63_cast_fp16")]; + tensor transpose_377_perm_0 = const()[name = string("transpose_377_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_91_transpose_x_1 = const()[name = string("mh_w_91_transpose_x_1"), val = bool(true)]; + bool mh_w_91_transpose_y_1 = const()[name = string("mh_w_91_transpose_y_1"), val = bool(false)]; + tensor transpose_377_cast_fp16 = transpose(perm = transpose_377_perm_0, x = reshape_61_cast_fp16)[name = string("transpose_385")]; + tensor mh_w_91_cast_fp16 = matmul(transpose_x = mh_w_91_transpose_x_1, transpose_y = mh_w_91_transpose_y_1, x = mh_q_127_cast_fp16, y = transpose_377_cast_fp16)[name = string("mh_w_91_cast_fp16")]; + tensor var_4849_to_fp16 = const()[name = string("op_4849_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192512)))]; + tensor mh_w_93_cast_fp16 = add(x = mh_w_91_cast_fp16, y = var_4849_to_fp16)[name = string("mh_w_93_cast_fp16")]; + tensor mh_w_95_cast_fp16 = softmax(axis = var_4669, x = mh_w_93_cast_fp16)[name = string("mh_w_95_cast_fp16")]; + tensor transpose_378_perm_0 = const()[name = string("transpose_378_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_31_transpose_x_1 = const()[name = string("attn_31_transpose_x_1"), val = bool(false)]; + bool attn_31_transpose_y_1 = const()[name = string("attn_31_transpose_y_1"), val = bool(true)]; + tensor transpose_378_cast_fp16 = transpose(perm = transpose_378_perm_0, x = reshape_63_cast_fp16)[name = string("transpose_384")]; + tensor attn_31_cast_fp16 = matmul(transpose_x = attn_31_transpose_x_1, transpose_y = attn_31_transpose_y_1, x = transpose_378_cast_fp16, y = mh_w_95_cast_fp16)[name = string("attn_31_cast_fp16")]; + tensor var_4855 = const()[name = string("op_4855"), val = tensor([1, 2048, 1, 1])]; + tensor input_129_cast_fp16 = reshape(shape = var_4855, x = attn_31_cast_fp16)[name = string("input_129_cast_fp16")]; + string obj_143_pad_type_0 = const()[name = string("obj_143_pad_type_0"), val = string("valid")]; + tensor obj_143_strides_0 = const()[name = string("obj_143_strides_0"), val = tensor([1, 1])]; + tensor obj_143_pad_0 = const()[name = string("obj_143_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_143_dilations_0 = const()[name = string("obj_143_dilations_0"), val = tensor([1, 1])]; + int32 obj_143_groups_0 = const()[name = string("obj_143_groups_0"), val = int32(1)]; + tensor obj_143_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_143_dilations_0, groups = obj_143_groups_0, pad = obj_143_pad_0, pad_type = obj_143_pad_type_0, strides = obj_143_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_129_cast_fp16)[name = string("obj_143_cast_fp16")]; + tensor inputs_131_cast_fp16 = add(x = inputs_125_cast_fp16, y = obj_143_cast_fp16)[name = string("inputs_131_cast_fp16")]; + tensor inputs_sq_131_cast_fp16 = mul(x = inputs_131_cast_fp16, y = inputs_131_cast_fp16)[name = string("inputs_sq_131_cast_fp16")]; + tensor variance_131_axes_0 = const()[name = string("variance_131_axes_0"), val = tensor([1])]; + bool variance_131_keep_dims_0 = const()[name = string("variance_131_keep_dims_0"), val = bool(true)]; + tensor variance_131_cast_fp16 = reduce_mean(axes = variance_131_axes_0, keep_dims = variance_131_keep_dims_0, x = inputs_sq_131_cast_fp16)[name = string("variance_131_cast_fp16")]; + fp16 var_4873_to_fp16 = const()[name = string("op_4873_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4874_cast_fp16 = add(x = variance_131_cast_fp16, y = var_4873_to_fp16)[name = string("op_4874_cast_fp16")]; + fp32 var_4875_epsilon_0 = const()[name = string("op_4875_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4875_cast_fp16 = rsqrt(epsilon = var_4875_epsilon_0, x = var_4874_cast_fp16)[name = string("op_4875_cast_fp16")]; + tensor hidden_states_161_cast_fp16 = mul(x = inputs_131_cast_fp16, y = var_4875_cast_fp16)[name = string("hidden_states_161_cast_fp16")]; + tensor input_131_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_161_cast_fp16)[name = string("input_131_cast_fp16")]; + string input_133_pad_type_0 = const()[name = string("input_133_pad_type_0"), val = string("valid")]; + tensor input_133_strides_0 = const()[name = string("input_133_strides_0"), val = tensor([1, 1])]; + tensor input_133_pad_0 = const()[name = string("input_133_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_133_dilations_0 = const()[name = string("input_133_dilations_0"), val = tensor([1, 1])]; + int32 input_133_groups_0 = const()[name = string("input_133_groups_0"), val = int32(1)]; + tensor input_133_cast_fp16 = conv(dilations = input_133_dilations_0, groups = input_133_groups_0, pad = input_133_pad_0, pad_type = input_133_pad_type_0, strides = input_133_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_131_cast_fp16)[name = string("input_133_cast_fp16")]; + tensor var_4889_cast_fp16 = silu(x = input_133_cast_fp16)[name = string("op_4889_cast_fp16")]; + string var_4895_pad_type_0 = const()[name = string("op_4895_pad_type_0"), val = string("valid")]; + tensor var_4895_strides_0 = const()[name = string("op_4895_strides_0"), val = tensor([1, 1])]; + tensor var_4895_pad_0 = const()[name = string("op_4895_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4895_dilations_0 = const()[name = string("op_4895_dilations_0"), val = tensor([1, 1])]; + int32 var_4895_groups_0 = const()[name = string("op_4895_groups_0"), val = int32(1)]; + tensor var_4895_cast_fp16 = conv(dilations = var_4895_dilations_0, groups = var_4895_groups_0, pad = var_4895_pad_0, pad_type = var_4895_pad_type_0, strides = var_4895_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_131_cast_fp16)[name = string("op_4895_cast_fp16")]; + tensor input_135_cast_fp16 = mul(x = var_4889_cast_fp16, y = var_4895_cast_fp16)[name = string("input_135_cast_fp16")]; + string hidden_states_163_pad_type_0 = const()[name = string("hidden_states_163_pad_type_0"), val = string("valid")]; + tensor hidden_states_163_strides_0 = const()[name = string("hidden_states_163_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_163_pad_0 = const()[name = string("hidden_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_163_dilations_0 = const()[name = string("hidden_states_163_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_163_groups_0 = const()[name = string("hidden_states_163_groups_0"), val = int32(1)]; + tensor hidden_states_163_cast_fp16 = conv(dilations = hidden_states_163_dilations_0, groups = hidden_states_163_groups_0, pad = hidden_states_163_pad_0, pad_type = hidden_states_163_pad_type_0, strides = hidden_states_163_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_135_cast_fp16)[name = string("hidden_states_163_cast_fp16")]; + tensor inputs_133_cast_fp16 = add(x = inputs_131_cast_fp16, y = hidden_states_163_cast_fp16)[name = string("inputs_133_cast_fp16")]; + tensor obj_147_begin_0 = const()[name = string("obj_147_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_147_end_0 = const()[name = string("obj_147_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_147_end_mask_0 = const()[name = string("obj_147_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_147_cast_fp16 = slice_by_index(begin = obj_147_begin_0, end = obj_147_end_0, end_mask = obj_147_end_mask_0, x = key_caches_7_cast_fp16)[name = string("obj_147_cast_fp16")]; + tensor obj_149_begin_0 = const()[name = string("obj_149_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_149_end_0 = const()[name = string("obj_149_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_149_end_mask_0 = const()[name = string("obj_149_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_149_cast_fp16 = slice_by_index(begin = obj_149_begin_0, end = obj_149_end_0, end_mask = obj_149_end_mask_0, x = value_caches_7_cast_fp16)[name = string("obj_149_cast_fp16")]; + int32 var_4943 = const()[name = string("op_4943"), val = int32(3)]; + int32 var_4953 = const()[name = string("op_4953"), val = int32(-2)]; + tensor inputs_sq_133_cast_fp16 = mul(x = inputs_133_cast_fp16, y = inputs_133_cast_fp16)[name = string("inputs_sq_133_cast_fp16")]; + tensor variance_133_axes_0 = const()[name = string("variance_133_axes_0"), val = tensor([1])]; + bool variance_133_keep_dims_0 = const()[name = string("variance_133_keep_dims_0"), val = bool(true)]; + tensor variance_133_cast_fp16 = reduce_mean(axes = variance_133_axes_0, keep_dims = variance_133_keep_dims_0, x = inputs_sq_133_cast_fp16)[name = string("variance_133_cast_fp16")]; + fp16 var_4967_to_fp16 = const()[name = string("op_4967_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_4968_cast_fp16 = add(x = variance_133_cast_fp16, y = var_4967_to_fp16)[name = string("op_4968_cast_fp16")]; + fp32 var_4969_epsilon_0 = const()[name = string("op_4969_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_4969_cast_fp16 = rsqrt(epsilon = var_4969_epsilon_0, x = var_4968_cast_fp16)[name = string("op_4969_cast_fp16")]; + tensor hidden_states_165_cast_fp16 = mul(x = inputs_133_cast_fp16, y = var_4969_cast_fp16)[name = string("hidden_states_165_cast_fp16")]; + tensor obj_145_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_165_cast_fp16)[name = string("obj_145_cast_fp16")]; + string query_97_pad_type_0 = const()[name = string("query_97_pad_type_0"), val = string("valid")]; + tensor query_97_strides_0 = const()[name = string("query_97_strides_0"), val = tensor([1, 1])]; + tensor query_97_pad_0 = const()[name = string("query_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_97_dilations_0 = const()[name = string("query_97_dilations_0"), val = tensor([1, 1])]; + int32 query_97_groups_0 = const()[name = string("query_97_groups_0"), val = int32(1)]; + tensor query_97_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_97_dilations_0, groups = query_97_groups_0, pad = query_97_pad_0, pad_type = query_97_pad_type_0, strides = query_97_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_145_cast_fp16)[name = string("query_97_cast_fp16")]; + string current_key_65_pad_type_0 = const()[name = string("current_key_65_pad_type_0"), val = string("valid")]; + tensor current_key_65_strides_0 = const()[name = string("current_key_65_strides_0"), val = tensor([1, 1])]; + tensor current_key_65_pad_0 = const()[name = string("current_key_65_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_65_dilations_0 = const()[name = string("current_key_65_dilations_0"), val = tensor([1, 1])]; + int32 current_key_65_groups_0 = const()[name = string("current_key_65_groups_0"), val = int32(1)]; + tensor current_key_65_cast_fp16 = conv(dilations = current_key_65_dilations_0, groups = current_key_65_groups_0, pad = current_key_65_pad_0, pad_type = current_key_65_pad_type_0, strides = current_key_65_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_145_cast_fp16)[name = string("current_key_65_cast_fp16")]; + string current_value_33_pad_type_0 = const()[name = string("current_value_33_pad_type_0"), val = string("valid")]; + tensor current_value_33_strides_0 = const()[name = string("current_value_33_strides_0"), val = tensor([1, 1])]; + tensor current_value_33_pad_0 = const()[name = string("current_value_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_33_dilations_0 = const()[name = string("current_value_33_dilations_0"), val = tensor([1, 1])]; + int32 current_value_33_groups_0 = const()[name = string("current_value_33_groups_0"), val = int32(1)]; + tensor current_value_33_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_33_dilations_0, groups = current_value_33_groups_0, pad = current_value_33_pad_0, pad_type = current_value_33_pad_type_0, strides = current_value_33_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_145_cast_fp16)[name = string("current_value_33_cast_fp16")]; + tensor var_5006 = const()[name = string("op_5006"), val = tensor([16, 128, 1, 1])]; + tensor inputs_135_cast_fp16 = reshape(shape = var_5006, x = query_97_cast_fp16)[name = string("inputs_135_cast_fp16")]; + tensor inputs_sq_135_cast_fp16 = mul(x = inputs_135_cast_fp16, y = inputs_135_cast_fp16)[name = string("inputs_sq_135_cast_fp16")]; + tensor variance_135_axes_0 = const()[name = string("variance_135_axes_0"), val = tensor([1])]; + bool variance_135_keep_dims_0 = const()[name = string("variance_135_keep_dims_0"), val = bool(true)]; + tensor variance_135_cast_fp16 = reduce_mean(axes = variance_135_axes_0, keep_dims = variance_135_keep_dims_0, x = inputs_sq_135_cast_fp16)[name = string("variance_135_cast_fp16")]; + fp16 var_5012_to_fp16 = const()[name = string("op_5012_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5013_cast_fp16 = add(x = variance_135_cast_fp16, y = var_5012_to_fp16)[name = string("op_5013_cast_fp16")]; + fp32 var_5014_epsilon_0 = const()[name = string("op_5014_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5014_cast_fp16 = rsqrt(epsilon = var_5014_epsilon_0, x = var_5013_cast_fp16)[name = string("op_5014_cast_fp16")]; + tensor hidden_states_167_cast_fp16 = mul(x = inputs_135_cast_fp16, y = var_5014_cast_fp16)[name = string("hidden_states_167_cast_fp16")]; + tensor query_normed_33_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_167_cast_fp16)[name = string("query_normed_33_cast_fp16")]; + tensor var_5022 = const()[name = string("op_5022"), val = tensor([8, 128, 1, 1])]; + tensor inputs_137_cast_fp16 = reshape(shape = var_5022, x = current_key_65_cast_fp16)[name = string("inputs_137_cast_fp16")]; + tensor inputs_sq_137_cast_fp16 = mul(x = inputs_137_cast_fp16, y = inputs_137_cast_fp16)[name = string("inputs_sq_137_cast_fp16")]; + tensor variance_137_axes_0 = const()[name = string("variance_137_axes_0"), val = tensor([1])]; + bool variance_137_keep_dims_0 = const()[name = string("variance_137_keep_dims_0"), val = bool(true)]; + tensor variance_137_cast_fp16 = reduce_mean(axes = variance_137_axes_0, keep_dims = variance_137_keep_dims_0, x = inputs_sq_137_cast_fp16)[name = string("variance_137_cast_fp16")]; + fp16 var_5028_to_fp16 = const()[name = string("op_5028_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5029_cast_fp16 = add(x = variance_137_cast_fp16, y = var_5028_to_fp16)[name = string("op_5029_cast_fp16")]; + fp32 var_5030_epsilon_0 = const()[name = string("op_5030_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5030_cast_fp16 = rsqrt(epsilon = var_5030_epsilon_0, x = var_5029_cast_fp16)[name = string("op_5030_cast_fp16")]; + tensor hidden_states_169_cast_fp16 = mul(x = inputs_137_cast_fp16, y = var_5030_cast_fp16)[name = string("hidden_states_169_cast_fp16")]; + tensor current_key_normed_33_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_169_cast_fp16)[name = string("current_key_normed_33_cast_fp16")]; + tensor var_5048 = const()[name = string("op_5048"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_129_cast_fp16 = reshape(shape = var_5048, x = query_normed_33_cast_fp16)[name = string("mh_q_129_cast_fp16")]; + tensor var_5050 = const()[name = string("op_5050"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_129_cast_fp16 = reshape(shape = var_5050, x = current_key_normed_33_cast_fp16)[name = string("mh_k_129_cast_fp16")]; + tensor var_5054_cast_fp16 = mul(x = mh_q_129_cast_fp16, y = cos_31_to_fp16)[name = string("op_5054_cast_fp16")]; + tensor var_5059_begin_0 = const()[name = string("op_5059_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5059_end_0 = const()[name = string("op_5059_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_5059_end_mask_0 = const()[name = string("op_5059_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_5059_cast_fp16 = slice_by_index(begin = var_5059_begin_0, end = var_5059_end_0, end_mask = var_5059_end_mask_0, x = mh_q_129_cast_fp16)[name = string("op_5059_cast_fp16")]; + tensor var_5065_begin_0 = const()[name = string("op_5065_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_5065_end_0 = const()[name = string("op_5065_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_5065_end_mask_0 = const()[name = string("op_5065_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_5065_cast_fp16 = slice_by_index(begin = var_5065_begin_0, end = var_5065_end_0, end_mask = var_5065_end_mask_0, x = mh_q_129_cast_fp16)[name = string("op_5065_cast_fp16")]; + fp16 const_337_promoted_to_fp16 = const()[name = string("const_337_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5067_cast_fp16 = mul(x = var_5065_cast_fp16, y = const_337_promoted_to_fp16)[name = string("op_5067_cast_fp16")]; + bool var_5069_interleave_0 = const()[name = string("op_5069_interleave_0"), val = bool(false)]; + tensor var_5069_cast_fp16 = concat(axis = var_4953, interleave = var_5069_interleave_0, values = (var_5067_cast_fp16, var_5059_cast_fp16))[name = string("op_5069_cast_fp16")]; + tensor var_5070_cast_fp16 = mul(x = var_5069_cast_fp16, y = sin_31_to_fp16)[name = string("op_5070_cast_fp16")]; + tensor mh_q_131_cast_fp16 = add(x = var_5054_cast_fp16, y = var_5070_cast_fp16)[name = string("mh_q_131_cast_fp16")]; + tensor var_5072_cast_fp16 = mul(x = mh_k_129_cast_fp16, y = cos_31_to_fp16)[name = string("op_5072_cast_fp16")]; + tensor var_5077_begin_0 = const()[name = string("op_5077_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5077_end_0 = const()[name = string("op_5077_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_5077_end_mask_0 = const()[name = string("op_5077_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_5077_cast_fp16 = slice_by_index(begin = var_5077_begin_0, end = var_5077_end_0, end_mask = var_5077_end_mask_0, x = mh_k_129_cast_fp16)[name = string("op_5077_cast_fp16")]; + tensor var_5083_begin_0 = const()[name = string("op_5083_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_5083_end_0 = const()[name = string("op_5083_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_5083_end_mask_0 = const()[name = string("op_5083_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_5083_cast_fp16 = slice_by_index(begin = var_5083_begin_0, end = var_5083_end_0, end_mask = var_5083_end_mask_0, x = mh_k_129_cast_fp16)[name = string("op_5083_cast_fp16")]; + fp16 const_340_promoted_to_fp16 = const()[name = string("const_340_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5085_cast_fp16 = mul(x = var_5083_cast_fp16, y = const_340_promoted_to_fp16)[name = string("op_5085_cast_fp16")]; + bool var_5087_interleave_0 = const()[name = string("op_5087_interleave_0"), val = bool(false)]; + tensor var_5087_cast_fp16 = concat(axis = var_4953, interleave = var_5087_interleave_0, values = (var_5085_cast_fp16, var_5077_cast_fp16))[name = string("op_5087_cast_fp16")]; + tensor var_5088_cast_fp16 = mul(x = var_5087_cast_fp16, y = sin_31_to_fp16)[name = string("op_5088_cast_fp16")]; + tensor mh_k_131_cast_fp16 = add(x = var_5072_cast_fp16, y = var_5088_cast_fp16)[name = string("mh_k_131_cast_fp16")]; + tensor var_5092 = const()[name = string("op_5092"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_67_cast_fp16 = reshape(shape = var_5092, x = mh_k_131_cast_fp16)[name = string("current_key_67_cast_fp16")]; + tensor var_5098_to_fp16 = const()[name = string("op_5098_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192256)))]; + tensor var_5099_cast_fp16 = mul(x = obj_147_cast_fp16, y = var_5098_to_fp16)[name = string("op_5099_cast_fp16")]; + tensor var_5096_to_fp16 = const()[name = string("op_5096_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192384)))]; + tensor var_5100_cast_fp16 = mul(x = current_key_67_cast_fp16, y = var_5096_to_fp16)[name = string("op_5100_cast_fp16")]; + tensor key_67_cast_fp16 = add(x = var_5099_cast_fp16, y = var_5100_cast_fp16)[name = string("key_67_cast_fp16")]; + tensor var_5102_to_fp16 = const()[name = string("op_5102_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192256)))]; + tensor var_5103_cast_fp16 = mul(x = obj_149_cast_fp16, y = var_5102_to_fp16)[name = string("op_5103_cast_fp16")]; + tensor var_5104_cast_fp16 = mul(x = current_value_33_cast_fp16, y = var_5096_to_fp16)[name = string("op_5104_cast_fp16")]; + tensor value_33_cast_fp16 = add(x = var_5103_cast_fp16, y = var_5104_cast_fp16)[name = string("value_33_cast_fp16")]; + fp16 var_5111_to_fp16 = const()[name = string("op_5111_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_135_cast_fp16 = mul(x = mh_q_131_cast_fp16, y = var_5111_to_fp16)[name = string("mh_q_135_cast_fp16")]; + tensor var_5113 = const()[name = string("op_5113"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_133_cast_fp16 = reshape(shape = var_5113, x = key_67_cast_fp16)[name = string("mh_k_133_cast_fp16")]; + tensor var_5115 = const()[name = string("op_5115"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_65_cast_fp16 = reshape(shape = var_5115, x = value_33_cast_fp16)[name = string("mh_v_65_cast_fp16")]; + tensor transpose_64_perm_0 = const()[name = string("transpose_64_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_32_reps_0 = const()[name = string("tile_32_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_64_cast_fp16 = transpose(perm = transpose_64_perm_0, x = mh_k_133_cast_fp16)[name = string("transpose_383")]; + tensor tile_32_cast_fp16 = tile(reps = tile_32_reps_0, x = transpose_64_cast_fp16)[name = string("tile_32_cast_fp16")]; + tensor concat_82 = const()[name = string("concat_82"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_64_cast_fp16 = reshape(shape = concat_82, x = tile_32_cast_fp16)[name = string("reshape_64_cast_fp16")]; + tensor transpose_65_perm_0 = const()[name = string("transpose_65_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_83 = const()[name = string("concat_83"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_65_cast_fp16 = transpose(perm = transpose_65_perm_0, x = reshape_64_cast_fp16)[name = string("transpose_382")]; + tensor reshape_65_cast_fp16 = reshape(shape = concat_83, x = transpose_65_cast_fp16)[name = string("reshape_65_cast_fp16")]; + tensor transpose_66_perm_0 = const()[name = string("transpose_66_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_33_reps_0 = const()[name = string("tile_33_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_66_cast_fp16 = transpose(perm = transpose_66_perm_0, x = mh_v_65_cast_fp16)[name = string("transpose_381")]; + tensor tile_33_cast_fp16 = tile(reps = tile_33_reps_0, x = transpose_66_cast_fp16)[name = string("tile_33_cast_fp16")]; + tensor concat_84 = const()[name = string("concat_84"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_66_cast_fp16 = reshape(shape = concat_84, x = tile_33_cast_fp16)[name = string("reshape_66_cast_fp16")]; + tensor transpose_67_perm_0 = const()[name = string("transpose_67_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_85 = const()[name = string("concat_85"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_67_cast_fp16 = transpose(perm = transpose_67_perm_0, x = reshape_66_cast_fp16)[name = string("transpose_380")]; + tensor reshape_67_cast_fp16 = reshape(shape = concat_85, x = transpose_67_cast_fp16)[name = string("reshape_67_cast_fp16")]; + tensor transpose_381_perm_0 = const()[name = string("transpose_381_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_97_transpose_x_1 = const()[name = string("mh_w_97_transpose_x_1"), val = bool(true)]; + bool mh_w_97_transpose_y_1 = const()[name = string("mh_w_97_transpose_y_1"), val = bool(false)]; + tensor transpose_381_cast_fp16 = transpose(perm = transpose_381_perm_0, x = reshape_65_cast_fp16)[name = string("transpose_379")]; + tensor mh_w_97_cast_fp16 = matmul(transpose_x = mh_w_97_transpose_x_1, transpose_y = mh_w_97_transpose_y_1, x = mh_q_135_cast_fp16, y = transpose_381_cast_fp16)[name = string("mh_w_97_cast_fp16")]; + tensor var_5123_to_fp16 = const()[name = string("op_5123_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192512)))]; + tensor mh_w_99_cast_fp16 = add(x = mh_w_97_cast_fp16, y = var_5123_to_fp16)[name = string("mh_w_99_cast_fp16")]; + tensor mh_w_101_cast_fp16 = softmax(axis = var_4943, x = mh_w_99_cast_fp16)[name = string("mh_w_101_cast_fp16")]; + tensor transpose_382_perm_0 = const()[name = string("transpose_382_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_33_transpose_x_1 = const()[name = string("attn_33_transpose_x_1"), val = bool(false)]; + bool attn_33_transpose_y_1 = const()[name = string("attn_33_transpose_y_1"), val = bool(true)]; + tensor transpose_382_cast_fp16 = transpose(perm = transpose_382_perm_0, x = reshape_67_cast_fp16)[name = string("transpose_378")]; + tensor attn_33_cast_fp16 = matmul(transpose_x = attn_33_transpose_x_1, transpose_y = attn_33_transpose_y_1, x = transpose_382_cast_fp16, y = mh_w_101_cast_fp16)[name = string("attn_33_cast_fp16")]; + tensor var_5129 = const()[name = string("op_5129"), val = tensor([1, 2048, 1, 1])]; + tensor input_137_cast_fp16 = reshape(shape = var_5129, x = attn_33_cast_fp16)[name = string("input_137_cast_fp16")]; + string obj_151_pad_type_0 = const()[name = string("obj_151_pad_type_0"), val = string("valid")]; + tensor obj_151_strides_0 = const()[name = string("obj_151_strides_0"), val = tensor([1, 1])]; + tensor obj_151_pad_0 = const()[name = string("obj_151_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_151_dilations_0 = const()[name = string("obj_151_dilations_0"), val = tensor([1, 1])]; + int32 obj_151_groups_0 = const()[name = string("obj_151_groups_0"), val = int32(1)]; + tensor obj_151_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_151_dilations_0, groups = obj_151_groups_0, pad = obj_151_pad_0, pad_type = obj_151_pad_type_0, strides = obj_151_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_137_cast_fp16)[name = string("obj_151_cast_fp16")]; + tensor inputs_139_cast_fp16 = add(x = inputs_133_cast_fp16, y = obj_151_cast_fp16)[name = string("inputs_139_cast_fp16")]; + tensor inputs_sq_139_cast_fp16 = mul(x = inputs_139_cast_fp16, y = inputs_139_cast_fp16)[name = string("inputs_sq_139_cast_fp16")]; + tensor variance_139_axes_0 = const()[name = string("variance_139_axes_0"), val = tensor([1])]; + bool variance_139_keep_dims_0 = const()[name = string("variance_139_keep_dims_0"), val = bool(true)]; + tensor variance_139_cast_fp16 = reduce_mean(axes = variance_139_axes_0, keep_dims = variance_139_keep_dims_0, x = inputs_sq_139_cast_fp16)[name = string("variance_139_cast_fp16")]; + fp16 var_5147_to_fp16 = const()[name = string("op_5147_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5148_cast_fp16 = add(x = variance_139_cast_fp16, y = var_5147_to_fp16)[name = string("op_5148_cast_fp16")]; + fp32 var_5149_epsilon_0 = const()[name = string("op_5149_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5149_cast_fp16 = rsqrt(epsilon = var_5149_epsilon_0, x = var_5148_cast_fp16)[name = string("op_5149_cast_fp16")]; + tensor hidden_states_171_cast_fp16 = mul(x = inputs_139_cast_fp16, y = var_5149_cast_fp16)[name = string("hidden_states_171_cast_fp16")]; + tensor input_139_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_171_cast_fp16)[name = string("input_139_cast_fp16")]; + string input_141_pad_type_0 = const()[name = string("input_141_pad_type_0"), val = string("valid")]; + tensor input_141_strides_0 = const()[name = string("input_141_strides_0"), val = tensor([1, 1])]; + tensor input_141_pad_0 = const()[name = string("input_141_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_141_dilations_0 = const()[name = string("input_141_dilations_0"), val = tensor([1, 1])]; + int32 input_141_groups_0 = const()[name = string("input_141_groups_0"), val = int32(1)]; + tensor input_141_cast_fp16 = conv(dilations = input_141_dilations_0, groups = input_141_groups_0, pad = input_141_pad_0, pad_type = input_141_pad_type_0, strides = input_141_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_139_cast_fp16)[name = string("input_141_cast_fp16")]; + tensor var_5163_cast_fp16 = silu(x = input_141_cast_fp16)[name = string("op_5163_cast_fp16")]; + string var_5169_pad_type_0 = const()[name = string("op_5169_pad_type_0"), val = string("valid")]; + tensor var_5169_strides_0 = const()[name = string("op_5169_strides_0"), val = tensor([1, 1])]; + tensor var_5169_pad_0 = const()[name = string("op_5169_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5169_dilations_0 = const()[name = string("op_5169_dilations_0"), val = tensor([1, 1])]; + int32 var_5169_groups_0 = const()[name = string("op_5169_groups_0"), val = int32(1)]; + tensor var_5169_cast_fp16 = conv(dilations = var_5169_dilations_0, groups = var_5169_groups_0, pad = var_5169_pad_0, pad_type = var_5169_pad_type_0, strides = var_5169_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_139_cast_fp16)[name = string("op_5169_cast_fp16")]; + tensor input_143_cast_fp16 = mul(x = var_5163_cast_fp16, y = var_5169_cast_fp16)[name = string("input_143_cast_fp16")]; + string hidden_states_173_pad_type_0 = const()[name = string("hidden_states_173_pad_type_0"), val = string("valid")]; + tensor hidden_states_173_strides_0 = const()[name = string("hidden_states_173_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_173_pad_0 = const()[name = string("hidden_states_173_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_173_dilations_0 = const()[name = string("hidden_states_173_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_173_groups_0 = const()[name = string("hidden_states_173_groups_0"), val = int32(1)]; + tensor hidden_states_173_cast_fp16 = conv(dilations = hidden_states_173_dilations_0, groups = hidden_states_173_groups_0, pad = hidden_states_173_pad_0, pad_type = hidden_states_173_pad_type_0, strides = hidden_states_173_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_143_cast_fp16)[name = string("hidden_states_173_cast_fp16")]; + tensor inputs_141_cast_fp16 = add(x = inputs_139_cast_fp16, y = hidden_states_173_cast_fp16)[name = string("inputs_141_cast_fp16")]; + tensor obj_155_begin_0 = const()[name = string("obj_155_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_155_end_0 = const()[name = string("obj_155_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_155_end_mask_0 = const()[name = string("obj_155_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_155_cast_fp16 = slice_by_index(begin = obj_155_begin_0, end = obj_155_end_0, end_mask = obj_155_end_mask_0, x = key_caches_7_cast_fp16)[name = string("obj_155_cast_fp16")]; + tensor obj_157_begin_0 = const()[name = string("obj_157_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_157_end_0 = const()[name = string("obj_157_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_157_end_mask_0 = const()[name = string("obj_157_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_157_cast_fp16 = slice_by_index(begin = obj_157_begin_0, end = obj_157_end_0, end_mask = obj_157_end_mask_0, x = value_caches_7_cast_fp16)[name = string("obj_157_cast_fp16")]; + int32 var_5217 = const()[name = string("op_5217"), val = int32(3)]; + int32 var_5227 = const()[name = string("op_5227"), val = int32(-2)]; + tensor inputs_sq_141_cast_fp16 = mul(x = inputs_141_cast_fp16, y = inputs_141_cast_fp16)[name = string("inputs_sq_141_cast_fp16")]; + tensor variance_141_axes_0 = const()[name = string("variance_141_axes_0"), val = tensor([1])]; + bool variance_141_keep_dims_0 = const()[name = string("variance_141_keep_dims_0"), val = bool(true)]; + tensor variance_141_cast_fp16 = reduce_mean(axes = variance_141_axes_0, keep_dims = variance_141_keep_dims_0, x = inputs_sq_141_cast_fp16)[name = string("variance_141_cast_fp16")]; + fp16 var_5241_to_fp16 = const()[name = string("op_5241_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5242_cast_fp16 = add(x = variance_141_cast_fp16, y = var_5241_to_fp16)[name = string("op_5242_cast_fp16")]; + fp32 var_5243_epsilon_0 = const()[name = string("op_5243_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5243_cast_fp16 = rsqrt(epsilon = var_5243_epsilon_0, x = var_5242_cast_fp16)[name = string("op_5243_cast_fp16")]; + tensor hidden_states_175_cast_fp16 = mul(x = inputs_141_cast_fp16, y = var_5243_cast_fp16)[name = string("hidden_states_175_cast_fp16")]; + tensor obj_153_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_175_cast_fp16)[name = string("obj_153_cast_fp16")]; + string query_103_pad_type_0 = const()[name = string("query_103_pad_type_0"), val = string("valid")]; + tensor query_103_strides_0 = const()[name = string("query_103_strides_0"), val = tensor([1, 1])]; + tensor query_103_pad_0 = const()[name = string("query_103_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_103_dilations_0 = const()[name = string("query_103_dilations_0"), val = tensor([1, 1])]; + int32 query_103_groups_0 = const()[name = string("query_103_groups_0"), val = int32(1)]; + tensor query_103_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_103_dilations_0, groups = query_103_groups_0, pad = query_103_pad_0, pad_type = query_103_pad_type_0, strides = query_103_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_153_cast_fp16)[name = string("query_103_cast_fp16")]; + string current_key_69_pad_type_0 = const()[name = string("current_key_69_pad_type_0"), val = string("valid")]; + tensor current_key_69_strides_0 = const()[name = string("current_key_69_strides_0"), val = tensor([1, 1])]; + tensor current_key_69_pad_0 = const()[name = string("current_key_69_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_69_dilations_0 = const()[name = string("current_key_69_dilations_0"), val = tensor([1, 1])]; + int32 current_key_69_groups_0 = const()[name = string("current_key_69_groups_0"), val = int32(1)]; + tensor current_key_69_cast_fp16 = conv(dilations = current_key_69_dilations_0, groups = current_key_69_groups_0, pad = current_key_69_pad_0, pad_type = current_key_69_pad_type_0, strides = current_key_69_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_153_cast_fp16)[name = string("current_key_69_cast_fp16")]; + string current_value_35_pad_type_0 = const()[name = string("current_value_35_pad_type_0"), val = string("valid")]; + tensor current_value_35_strides_0 = const()[name = string("current_value_35_strides_0"), val = tensor([1, 1])]; + tensor current_value_35_pad_0 = const()[name = string("current_value_35_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_35_dilations_0 = const()[name = string("current_value_35_dilations_0"), val = tensor([1, 1])]; + int32 current_value_35_groups_0 = const()[name = string("current_value_35_groups_0"), val = int32(1)]; + tensor current_value_35_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_35_dilations_0, groups = current_value_35_groups_0, pad = current_value_35_pad_0, pad_type = current_value_35_pad_type_0, strides = current_value_35_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_153_cast_fp16)[name = string("current_value_35_cast_fp16")]; + tensor var_5280 = const()[name = string("op_5280"), val = tensor([16, 128, 1, 1])]; + tensor inputs_143_cast_fp16 = reshape(shape = var_5280, x = query_103_cast_fp16)[name = string("inputs_143_cast_fp16")]; + tensor inputs_sq_143_cast_fp16 = mul(x = inputs_143_cast_fp16, y = inputs_143_cast_fp16)[name = string("inputs_sq_143_cast_fp16")]; + tensor variance_143_axes_0 = const()[name = string("variance_143_axes_0"), val = tensor([1])]; + bool variance_143_keep_dims_0 = const()[name = string("variance_143_keep_dims_0"), val = bool(true)]; + tensor variance_143_cast_fp16 = reduce_mean(axes = variance_143_axes_0, keep_dims = variance_143_keep_dims_0, x = inputs_sq_143_cast_fp16)[name = string("variance_143_cast_fp16")]; + fp16 var_5286_to_fp16 = const()[name = string("op_5286_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5287_cast_fp16 = add(x = variance_143_cast_fp16, y = var_5286_to_fp16)[name = string("op_5287_cast_fp16")]; + fp32 var_5288_epsilon_0 = const()[name = string("op_5288_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5288_cast_fp16 = rsqrt(epsilon = var_5288_epsilon_0, x = var_5287_cast_fp16)[name = string("op_5288_cast_fp16")]; + tensor hidden_states_177_cast_fp16 = mul(x = inputs_143_cast_fp16, y = var_5288_cast_fp16)[name = string("hidden_states_177_cast_fp16")]; + tensor query_normed_35_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_177_cast_fp16)[name = string("query_normed_35_cast_fp16")]; + tensor var_5296 = const()[name = string("op_5296"), val = tensor([8, 128, 1, 1])]; + tensor inputs_145_cast_fp16 = reshape(shape = var_5296, x = current_key_69_cast_fp16)[name = string("inputs_145_cast_fp16")]; + tensor inputs_sq_145_cast_fp16 = mul(x = inputs_145_cast_fp16, y = inputs_145_cast_fp16)[name = string("inputs_sq_145_cast_fp16")]; + tensor variance_145_axes_0 = const()[name = string("variance_145_axes_0"), val = tensor([1])]; + bool variance_145_keep_dims_0 = const()[name = string("variance_145_keep_dims_0"), val = bool(true)]; + tensor variance_145_cast_fp16 = reduce_mean(axes = variance_145_axes_0, keep_dims = variance_145_keep_dims_0, x = inputs_sq_145_cast_fp16)[name = string("variance_145_cast_fp16")]; + fp16 var_5302_to_fp16 = const()[name = string("op_5302_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5303_cast_fp16 = add(x = variance_145_cast_fp16, y = var_5302_to_fp16)[name = string("op_5303_cast_fp16")]; + fp32 var_5304_epsilon_0 = const()[name = string("op_5304_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5304_cast_fp16 = rsqrt(epsilon = var_5304_epsilon_0, x = var_5303_cast_fp16)[name = string("op_5304_cast_fp16")]; + tensor hidden_states_179_cast_fp16 = mul(x = inputs_145_cast_fp16, y = var_5304_cast_fp16)[name = string("hidden_states_179_cast_fp16")]; + tensor current_key_normed_35_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_179_cast_fp16)[name = string("current_key_normed_35_cast_fp16")]; + tensor var_5322 = const()[name = string("op_5322"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_137_cast_fp16 = reshape(shape = var_5322, x = query_normed_35_cast_fp16)[name = string("mh_q_137_cast_fp16")]; + tensor var_5324 = const()[name = string("op_5324"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_137_cast_fp16 = reshape(shape = var_5324, x = current_key_normed_35_cast_fp16)[name = string("mh_k_137_cast_fp16")]; + tensor var_5328_cast_fp16 = mul(x = mh_q_137_cast_fp16, y = cos_31_to_fp16)[name = string("op_5328_cast_fp16")]; + tensor var_5333_begin_0 = const()[name = string("op_5333_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5333_end_0 = const()[name = string("op_5333_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_5333_end_mask_0 = const()[name = string("op_5333_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_5333_cast_fp16 = slice_by_index(begin = var_5333_begin_0, end = var_5333_end_0, end_mask = var_5333_end_mask_0, x = mh_q_137_cast_fp16)[name = string("op_5333_cast_fp16")]; + tensor var_5339_begin_0 = const()[name = string("op_5339_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_5339_end_0 = const()[name = string("op_5339_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_5339_end_mask_0 = const()[name = string("op_5339_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_5339_cast_fp16 = slice_by_index(begin = var_5339_begin_0, end = var_5339_end_0, end_mask = var_5339_end_mask_0, x = mh_q_137_cast_fp16)[name = string("op_5339_cast_fp16")]; + fp16 const_357_promoted_to_fp16 = const()[name = string("const_357_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5341_cast_fp16 = mul(x = var_5339_cast_fp16, y = const_357_promoted_to_fp16)[name = string("op_5341_cast_fp16")]; + bool var_5343_interleave_0 = const()[name = string("op_5343_interleave_0"), val = bool(false)]; + tensor var_5343_cast_fp16 = concat(axis = var_5227, interleave = var_5343_interleave_0, values = (var_5341_cast_fp16, var_5333_cast_fp16))[name = string("op_5343_cast_fp16")]; + tensor var_5344_cast_fp16 = mul(x = var_5343_cast_fp16, y = sin_31_to_fp16)[name = string("op_5344_cast_fp16")]; + tensor mh_q_139_cast_fp16 = add(x = var_5328_cast_fp16, y = var_5344_cast_fp16)[name = string("mh_q_139_cast_fp16")]; + tensor var_5346_cast_fp16 = mul(x = mh_k_137_cast_fp16, y = cos_31_to_fp16)[name = string("op_5346_cast_fp16")]; + tensor var_5351_begin_0 = const()[name = string("op_5351_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5351_end_0 = const()[name = string("op_5351_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_5351_end_mask_0 = const()[name = string("op_5351_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_5351_cast_fp16 = slice_by_index(begin = var_5351_begin_0, end = var_5351_end_0, end_mask = var_5351_end_mask_0, x = mh_k_137_cast_fp16)[name = string("op_5351_cast_fp16")]; + tensor var_5357_begin_0 = const()[name = string("op_5357_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_5357_end_0 = const()[name = string("op_5357_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_5357_end_mask_0 = const()[name = string("op_5357_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_5357_cast_fp16 = slice_by_index(begin = var_5357_begin_0, end = var_5357_end_0, end_mask = var_5357_end_mask_0, x = mh_k_137_cast_fp16)[name = string("op_5357_cast_fp16")]; + fp16 const_360_promoted_to_fp16 = const()[name = string("const_360_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5359_cast_fp16 = mul(x = var_5357_cast_fp16, y = const_360_promoted_to_fp16)[name = string("op_5359_cast_fp16")]; + bool var_5361_interleave_0 = const()[name = string("op_5361_interleave_0"), val = bool(false)]; + tensor var_5361_cast_fp16 = concat(axis = var_5227, interleave = var_5361_interleave_0, values = (var_5359_cast_fp16, var_5351_cast_fp16))[name = string("op_5361_cast_fp16")]; + tensor var_5362_cast_fp16 = mul(x = var_5361_cast_fp16, y = sin_31_to_fp16)[name = string("op_5362_cast_fp16")]; + tensor mh_k_139_cast_fp16 = add(x = var_5346_cast_fp16, y = var_5362_cast_fp16)[name = string("mh_k_139_cast_fp16")]; + tensor var_5366 = const()[name = string("op_5366"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_71_cast_fp16 = reshape(shape = var_5366, x = mh_k_139_cast_fp16)[name = string("current_key_71_cast_fp16")]; + tensor var_5372_to_fp16 = const()[name = string("op_5372_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192256)))]; + tensor var_5373_cast_fp16 = mul(x = obj_155_cast_fp16, y = var_5372_to_fp16)[name = string("op_5373_cast_fp16")]; + tensor var_5370_to_fp16 = const()[name = string("op_5370_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192384)))]; + tensor var_5374_cast_fp16 = mul(x = current_key_71_cast_fp16, y = var_5370_to_fp16)[name = string("op_5374_cast_fp16")]; + tensor key_71_cast_fp16 = add(x = var_5373_cast_fp16, y = var_5374_cast_fp16)[name = string("key_71_cast_fp16")]; + tensor var_5376_to_fp16 = const()[name = string("op_5376_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192256)))]; + tensor var_5377_cast_fp16 = mul(x = obj_157_cast_fp16, y = var_5376_to_fp16)[name = string("op_5377_cast_fp16")]; + tensor var_5378_cast_fp16 = mul(x = current_value_35_cast_fp16, y = var_5370_to_fp16)[name = string("op_5378_cast_fp16")]; + tensor value_35_cast_fp16 = add(x = var_5377_cast_fp16, y = var_5378_cast_fp16)[name = string("value_35_cast_fp16")]; + fp16 var_5385_to_fp16 = const()[name = string("op_5385_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_143_cast_fp16 = mul(x = mh_q_139_cast_fp16, y = var_5385_to_fp16)[name = string("mh_q_143_cast_fp16")]; + tensor var_5387 = const()[name = string("op_5387"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_141_cast_fp16 = reshape(shape = var_5387, x = key_71_cast_fp16)[name = string("mh_k_141_cast_fp16")]; + tensor var_5389 = const()[name = string("op_5389"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_69_cast_fp16 = reshape(shape = var_5389, x = value_35_cast_fp16)[name = string("mh_v_69_cast_fp16")]; + tensor transpose_68_perm_0 = const()[name = string("transpose_68_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_34_reps_0 = const()[name = string("tile_34_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_68_cast_fp16 = transpose(perm = transpose_68_perm_0, x = mh_k_141_cast_fp16)[name = string("transpose_377")]; + tensor tile_34_cast_fp16 = tile(reps = tile_34_reps_0, x = transpose_68_cast_fp16)[name = string("tile_34_cast_fp16")]; + tensor concat_86 = const()[name = string("concat_86"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_68_cast_fp16 = reshape(shape = concat_86, x = tile_34_cast_fp16)[name = string("reshape_68_cast_fp16")]; + tensor transpose_69_perm_0 = const()[name = string("transpose_69_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_87 = const()[name = string("concat_87"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_69_cast_fp16 = transpose(perm = transpose_69_perm_0, x = reshape_68_cast_fp16)[name = string("transpose_376")]; + tensor reshape_69_cast_fp16 = reshape(shape = concat_87, x = transpose_69_cast_fp16)[name = string("reshape_69_cast_fp16")]; + tensor transpose_70_perm_0 = const()[name = string("transpose_70_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_35_reps_0 = const()[name = string("tile_35_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_70_cast_fp16 = transpose(perm = transpose_70_perm_0, x = mh_v_69_cast_fp16)[name = string("transpose_375")]; + tensor tile_35_cast_fp16 = tile(reps = tile_35_reps_0, x = transpose_70_cast_fp16)[name = string("tile_35_cast_fp16")]; + tensor concat_88 = const()[name = string("concat_88"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_70_cast_fp16 = reshape(shape = concat_88, x = tile_35_cast_fp16)[name = string("reshape_70_cast_fp16")]; + tensor transpose_71_perm_0 = const()[name = string("transpose_71_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_89 = const()[name = string("concat_89"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_71_cast_fp16 = transpose(perm = transpose_71_perm_0, x = reshape_70_cast_fp16)[name = string("transpose_374")]; + tensor reshape_71_cast_fp16 = reshape(shape = concat_89, x = transpose_71_cast_fp16)[name = string("reshape_71_cast_fp16")]; + tensor transpose_385_perm_0 = const()[name = string("transpose_385_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_103_transpose_x_1 = const()[name = string("mh_w_103_transpose_x_1"), val = bool(true)]; + bool mh_w_103_transpose_y_1 = const()[name = string("mh_w_103_transpose_y_1"), val = bool(false)]; + tensor transpose_385_cast_fp16 = transpose(perm = transpose_385_perm_0, x = reshape_69_cast_fp16)[name = string("transpose_373")]; + tensor mh_w_103_cast_fp16 = matmul(transpose_x = mh_w_103_transpose_x_1, transpose_y = mh_w_103_transpose_y_1, x = mh_q_143_cast_fp16, y = transpose_385_cast_fp16)[name = string("mh_w_103_cast_fp16")]; + tensor var_5397_to_fp16 = const()[name = string("op_5397_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192512)))]; + tensor mh_w_105_cast_fp16 = add(x = mh_w_103_cast_fp16, y = var_5397_to_fp16)[name = string("mh_w_105_cast_fp16")]; + tensor mh_w_107_cast_fp16 = softmax(axis = var_5217, x = mh_w_105_cast_fp16)[name = string("mh_w_107_cast_fp16")]; + tensor transpose_386_perm_0 = const()[name = string("transpose_386_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_35_transpose_x_1 = const()[name = string("attn_35_transpose_x_1"), val = bool(false)]; + bool attn_35_transpose_y_1 = const()[name = string("attn_35_transpose_y_1"), val = bool(true)]; + tensor transpose_386_cast_fp16 = transpose(perm = transpose_386_perm_0, x = reshape_71_cast_fp16)[name = string("transpose_372")]; + tensor attn_35_cast_fp16 = matmul(transpose_x = attn_35_transpose_x_1, transpose_y = attn_35_transpose_y_1, x = transpose_386_cast_fp16, y = mh_w_107_cast_fp16)[name = string("attn_35_cast_fp16")]; + tensor var_5403 = const()[name = string("op_5403"), val = tensor([1, 2048, 1, 1])]; + tensor input_145_cast_fp16 = reshape(shape = var_5403, x = attn_35_cast_fp16)[name = string("input_145_cast_fp16")]; + string obj_159_pad_type_0 = const()[name = string("obj_159_pad_type_0"), val = string("valid")]; + tensor obj_159_strides_0 = const()[name = string("obj_159_strides_0"), val = tensor([1, 1])]; + tensor obj_159_pad_0 = const()[name = string("obj_159_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_159_dilations_0 = const()[name = string("obj_159_dilations_0"), val = tensor([1, 1])]; + int32 obj_159_groups_0 = const()[name = string("obj_159_groups_0"), val = int32(1)]; + tensor obj_159_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_159_dilations_0, groups = obj_159_groups_0, pad = obj_159_pad_0, pad_type = obj_159_pad_type_0, strides = obj_159_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_145_cast_fp16)[name = string("obj_159_cast_fp16")]; + tensor inputs_147_cast_fp16 = add(x = inputs_141_cast_fp16, y = obj_159_cast_fp16)[name = string("inputs_147_cast_fp16")]; + tensor inputs_sq_147_cast_fp16 = mul(x = inputs_147_cast_fp16, y = inputs_147_cast_fp16)[name = string("inputs_sq_147_cast_fp16")]; + tensor variance_147_axes_0 = const()[name = string("variance_147_axes_0"), val = tensor([1])]; + bool variance_147_keep_dims_0 = const()[name = string("variance_147_keep_dims_0"), val = bool(true)]; + tensor variance_147_cast_fp16 = reduce_mean(axes = variance_147_axes_0, keep_dims = variance_147_keep_dims_0, x = inputs_sq_147_cast_fp16)[name = string("variance_147_cast_fp16")]; + fp16 var_5421_to_fp16 = const()[name = string("op_5421_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5422_cast_fp16 = add(x = variance_147_cast_fp16, y = var_5421_to_fp16)[name = string("op_5422_cast_fp16")]; + fp32 var_5423_epsilon_0 = const()[name = string("op_5423_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5423_cast_fp16 = rsqrt(epsilon = var_5423_epsilon_0, x = var_5422_cast_fp16)[name = string("op_5423_cast_fp16")]; + tensor hidden_states_181_cast_fp16 = mul(x = inputs_147_cast_fp16, y = var_5423_cast_fp16)[name = string("hidden_states_181_cast_fp16")]; + tensor input_147_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_181_cast_fp16)[name = string("input_147_cast_fp16")]; + string input_149_pad_type_0 = const()[name = string("input_149_pad_type_0"), val = string("valid")]; + tensor input_149_strides_0 = const()[name = string("input_149_strides_0"), val = tensor([1, 1])]; + tensor input_149_pad_0 = const()[name = string("input_149_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_149_dilations_0 = const()[name = string("input_149_dilations_0"), val = tensor([1, 1])]; + int32 input_149_groups_0 = const()[name = string("input_149_groups_0"), val = int32(1)]; + tensor input_149_cast_fp16 = conv(dilations = input_149_dilations_0, groups = input_149_groups_0, pad = input_149_pad_0, pad_type = input_149_pad_type_0, strides = input_149_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_147_cast_fp16)[name = string("input_149_cast_fp16")]; + tensor var_5437_cast_fp16 = silu(x = input_149_cast_fp16)[name = string("op_5437_cast_fp16")]; + string var_5443_pad_type_0 = const()[name = string("op_5443_pad_type_0"), val = string("valid")]; + tensor var_5443_strides_0 = const()[name = string("op_5443_strides_0"), val = tensor([1, 1])]; + tensor var_5443_pad_0 = const()[name = string("op_5443_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5443_dilations_0 = const()[name = string("op_5443_dilations_0"), val = tensor([1, 1])]; + int32 var_5443_groups_0 = const()[name = string("op_5443_groups_0"), val = int32(1)]; + tensor var_5443_cast_fp16 = conv(dilations = var_5443_dilations_0, groups = var_5443_groups_0, pad = var_5443_pad_0, pad_type = var_5443_pad_type_0, strides = var_5443_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_147_cast_fp16)[name = string("op_5443_cast_fp16")]; + tensor input_151_cast_fp16 = mul(x = var_5437_cast_fp16, y = var_5443_cast_fp16)[name = string("input_151_cast_fp16")]; + string hidden_states_183_pad_type_0 = const()[name = string("hidden_states_183_pad_type_0"), val = string("valid")]; + tensor hidden_states_183_strides_0 = const()[name = string("hidden_states_183_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_183_pad_0 = const()[name = string("hidden_states_183_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_183_dilations_0 = const()[name = string("hidden_states_183_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_183_groups_0 = const()[name = string("hidden_states_183_groups_0"), val = int32(1)]; + tensor hidden_states_183_cast_fp16 = conv(dilations = hidden_states_183_dilations_0, groups = hidden_states_183_groups_0, pad = hidden_states_183_pad_0, pad_type = hidden_states_183_pad_type_0, strides = hidden_states_183_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_151_cast_fp16)[name = string("hidden_states_183_cast_fp16")]; + tensor inputs_149_cast_fp16 = add(x = inputs_147_cast_fp16, y = hidden_states_183_cast_fp16)[name = string("inputs_149_cast_fp16")]; + tensor obj_163_begin_0 = const()[name = string("obj_163_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_163_end_0 = const()[name = string("obj_163_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_163_end_mask_0 = const()[name = string("obj_163_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_163_cast_fp16 = slice_by_index(begin = obj_163_begin_0, end = obj_163_end_0, end_mask = obj_163_end_mask_0, x = key_caches_7_cast_fp16)[name = string("obj_163_cast_fp16")]; + tensor obj_165_begin_0 = const()[name = string("obj_165_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_165_end_0 = const()[name = string("obj_165_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_165_end_mask_0 = const()[name = string("obj_165_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_165_cast_fp16 = slice_by_index(begin = obj_165_begin_0, end = obj_165_end_0, end_mask = obj_165_end_mask_0, x = value_caches_7_cast_fp16)[name = string("obj_165_cast_fp16")]; + int32 var_5491 = const()[name = string("op_5491"), val = int32(3)]; + int32 var_5501 = const()[name = string("op_5501"), val = int32(-2)]; + tensor inputs_sq_149_cast_fp16 = mul(x = inputs_149_cast_fp16, y = inputs_149_cast_fp16)[name = string("inputs_sq_149_cast_fp16")]; + tensor variance_149_axes_0 = const()[name = string("variance_149_axes_0"), val = tensor([1])]; + bool variance_149_keep_dims_0 = const()[name = string("variance_149_keep_dims_0"), val = bool(true)]; + tensor variance_149_cast_fp16 = reduce_mean(axes = variance_149_axes_0, keep_dims = variance_149_keep_dims_0, x = inputs_sq_149_cast_fp16)[name = string("variance_149_cast_fp16")]; + fp16 var_5515_to_fp16 = const()[name = string("op_5515_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5516_cast_fp16 = add(x = variance_149_cast_fp16, y = var_5515_to_fp16)[name = string("op_5516_cast_fp16")]; + fp32 var_5517_epsilon_0 = const()[name = string("op_5517_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5517_cast_fp16 = rsqrt(epsilon = var_5517_epsilon_0, x = var_5516_cast_fp16)[name = string("op_5517_cast_fp16")]; + tensor hidden_states_185_cast_fp16 = mul(x = inputs_149_cast_fp16, y = var_5517_cast_fp16)[name = string("hidden_states_185_cast_fp16")]; + tensor obj_161_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_185_cast_fp16)[name = string("obj_161_cast_fp16")]; + string query_109_pad_type_0 = const()[name = string("query_109_pad_type_0"), val = string("valid")]; + tensor query_109_strides_0 = const()[name = string("query_109_strides_0"), val = tensor([1, 1])]; + tensor query_109_pad_0 = const()[name = string("query_109_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_109_dilations_0 = const()[name = string("query_109_dilations_0"), val = tensor([1, 1])]; + int32 query_109_groups_0 = const()[name = string("query_109_groups_0"), val = int32(1)]; + tensor query_109_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_109_dilations_0, groups = query_109_groups_0, pad = query_109_pad_0, pad_type = query_109_pad_type_0, strides = query_109_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_161_cast_fp16)[name = string("query_109_cast_fp16")]; + string current_key_73_pad_type_0 = const()[name = string("current_key_73_pad_type_0"), val = string("valid")]; + tensor current_key_73_strides_0 = const()[name = string("current_key_73_strides_0"), val = tensor([1, 1])]; + tensor current_key_73_pad_0 = const()[name = string("current_key_73_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_73_dilations_0 = const()[name = string("current_key_73_dilations_0"), val = tensor([1, 1])]; + int32 current_key_73_groups_0 = const()[name = string("current_key_73_groups_0"), val = int32(1)]; + tensor current_key_73_cast_fp16 = conv(dilations = current_key_73_dilations_0, groups = current_key_73_groups_0, pad = current_key_73_pad_0, pad_type = current_key_73_pad_type_0, strides = current_key_73_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_161_cast_fp16)[name = string("current_key_73_cast_fp16")]; + string current_value_37_pad_type_0 = const()[name = string("current_value_37_pad_type_0"), val = string("valid")]; + tensor current_value_37_strides_0 = const()[name = string("current_value_37_strides_0"), val = tensor([1, 1])]; + tensor current_value_37_pad_0 = const()[name = string("current_value_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_37_dilations_0 = const()[name = string("current_value_37_dilations_0"), val = tensor([1, 1])]; + int32 current_value_37_groups_0 = const()[name = string("current_value_37_groups_0"), val = int32(1)]; + tensor current_value_37_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_37_dilations_0, groups = current_value_37_groups_0, pad = current_value_37_pad_0, pad_type = current_value_37_pad_type_0, strides = current_value_37_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_161_cast_fp16)[name = string("current_value_37_cast_fp16")]; + tensor var_5554 = const()[name = string("op_5554"), val = tensor([16, 128, 1, 1])]; + tensor inputs_151_cast_fp16 = reshape(shape = var_5554, x = query_109_cast_fp16)[name = string("inputs_151_cast_fp16")]; + tensor inputs_sq_151_cast_fp16 = mul(x = inputs_151_cast_fp16, y = inputs_151_cast_fp16)[name = string("inputs_sq_151_cast_fp16")]; + tensor variance_151_axes_0 = const()[name = string("variance_151_axes_0"), val = tensor([1])]; + bool variance_151_keep_dims_0 = const()[name = string("variance_151_keep_dims_0"), val = bool(true)]; + tensor variance_151_cast_fp16 = reduce_mean(axes = variance_151_axes_0, keep_dims = variance_151_keep_dims_0, x = inputs_sq_151_cast_fp16)[name = string("variance_151_cast_fp16")]; + fp16 var_5560_to_fp16 = const()[name = string("op_5560_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5561_cast_fp16 = add(x = variance_151_cast_fp16, y = var_5560_to_fp16)[name = string("op_5561_cast_fp16")]; + fp32 var_5562_epsilon_0 = const()[name = string("op_5562_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5562_cast_fp16 = rsqrt(epsilon = var_5562_epsilon_0, x = var_5561_cast_fp16)[name = string("op_5562_cast_fp16")]; + tensor hidden_states_187_cast_fp16 = mul(x = inputs_151_cast_fp16, y = var_5562_cast_fp16)[name = string("hidden_states_187_cast_fp16")]; + tensor query_normed_37_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_187_cast_fp16)[name = string("query_normed_37_cast_fp16")]; + tensor var_5570 = const()[name = string("op_5570"), val = tensor([8, 128, 1, 1])]; + tensor inputs_153_cast_fp16 = reshape(shape = var_5570, x = current_key_73_cast_fp16)[name = string("inputs_153_cast_fp16")]; + tensor inputs_sq_153_cast_fp16 = mul(x = inputs_153_cast_fp16, y = inputs_153_cast_fp16)[name = string("inputs_sq_153_cast_fp16")]; + tensor variance_153_axes_0 = const()[name = string("variance_153_axes_0"), val = tensor([1])]; + bool variance_153_keep_dims_0 = const()[name = string("variance_153_keep_dims_0"), val = bool(true)]; + tensor variance_153_cast_fp16 = reduce_mean(axes = variance_153_axes_0, keep_dims = variance_153_keep_dims_0, x = inputs_sq_153_cast_fp16)[name = string("variance_153_cast_fp16")]; + fp16 var_5576_to_fp16 = const()[name = string("op_5576_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5577_cast_fp16 = add(x = variance_153_cast_fp16, y = var_5576_to_fp16)[name = string("op_5577_cast_fp16")]; + fp32 var_5578_epsilon_0 = const()[name = string("op_5578_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5578_cast_fp16 = rsqrt(epsilon = var_5578_epsilon_0, x = var_5577_cast_fp16)[name = string("op_5578_cast_fp16")]; + tensor hidden_states_189_cast_fp16 = mul(x = inputs_153_cast_fp16, y = var_5578_cast_fp16)[name = string("hidden_states_189_cast_fp16")]; + tensor current_key_normed_37_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_189_cast_fp16)[name = string("current_key_normed_37_cast_fp16")]; + tensor var_5596 = const()[name = string("op_5596"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_145_cast_fp16 = reshape(shape = var_5596, x = query_normed_37_cast_fp16)[name = string("mh_q_145_cast_fp16")]; + tensor var_5598 = const()[name = string("op_5598"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_145_cast_fp16 = reshape(shape = var_5598, x = current_key_normed_37_cast_fp16)[name = string("mh_k_145_cast_fp16")]; + tensor var_5602_cast_fp16 = mul(x = mh_q_145_cast_fp16, y = cos_31_to_fp16)[name = string("op_5602_cast_fp16")]; + tensor var_5607_begin_0 = const()[name = string("op_5607_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5607_end_0 = const()[name = string("op_5607_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_5607_end_mask_0 = const()[name = string("op_5607_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_5607_cast_fp16 = slice_by_index(begin = var_5607_begin_0, end = var_5607_end_0, end_mask = var_5607_end_mask_0, x = mh_q_145_cast_fp16)[name = string("op_5607_cast_fp16")]; + tensor var_5613_begin_0 = const()[name = string("op_5613_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_5613_end_0 = const()[name = string("op_5613_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_5613_end_mask_0 = const()[name = string("op_5613_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_5613_cast_fp16 = slice_by_index(begin = var_5613_begin_0, end = var_5613_end_0, end_mask = var_5613_end_mask_0, x = mh_q_145_cast_fp16)[name = string("op_5613_cast_fp16")]; + fp16 const_377_promoted_to_fp16 = const()[name = string("const_377_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5615_cast_fp16 = mul(x = var_5613_cast_fp16, y = const_377_promoted_to_fp16)[name = string("op_5615_cast_fp16")]; + bool var_5617_interleave_0 = const()[name = string("op_5617_interleave_0"), val = bool(false)]; + tensor var_5617_cast_fp16 = concat(axis = var_5501, interleave = var_5617_interleave_0, values = (var_5615_cast_fp16, var_5607_cast_fp16))[name = string("op_5617_cast_fp16")]; + tensor var_5618_cast_fp16 = mul(x = var_5617_cast_fp16, y = sin_31_to_fp16)[name = string("op_5618_cast_fp16")]; + tensor mh_q_147_cast_fp16 = add(x = var_5602_cast_fp16, y = var_5618_cast_fp16)[name = string("mh_q_147_cast_fp16")]; + tensor var_5620_cast_fp16 = mul(x = mh_k_145_cast_fp16, y = cos_31_to_fp16)[name = string("op_5620_cast_fp16")]; + tensor var_5625_begin_0 = const()[name = string("op_5625_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5625_end_0 = const()[name = string("op_5625_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_5625_end_mask_0 = const()[name = string("op_5625_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_5625_cast_fp16 = slice_by_index(begin = var_5625_begin_0, end = var_5625_end_0, end_mask = var_5625_end_mask_0, x = mh_k_145_cast_fp16)[name = string("op_5625_cast_fp16")]; + tensor var_5631_begin_0 = const()[name = string("op_5631_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_5631_end_0 = const()[name = string("op_5631_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_5631_end_mask_0 = const()[name = string("op_5631_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_5631_cast_fp16 = slice_by_index(begin = var_5631_begin_0, end = var_5631_end_0, end_mask = var_5631_end_mask_0, x = mh_k_145_cast_fp16)[name = string("op_5631_cast_fp16")]; + fp16 const_380_promoted_to_fp16 = const()[name = string("const_380_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5633_cast_fp16 = mul(x = var_5631_cast_fp16, y = const_380_promoted_to_fp16)[name = string("op_5633_cast_fp16")]; + bool var_5635_interleave_0 = const()[name = string("op_5635_interleave_0"), val = bool(false)]; + tensor var_5635_cast_fp16 = concat(axis = var_5501, interleave = var_5635_interleave_0, values = (var_5633_cast_fp16, var_5625_cast_fp16))[name = string("op_5635_cast_fp16")]; + tensor var_5636_cast_fp16 = mul(x = var_5635_cast_fp16, y = sin_31_to_fp16)[name = string("op_5636_cast_fp16")]; + tensor mh_k_147_cast_fp16 = add(x = var_5620_cast_fp16, y = var_5636_cast_fp16)[name = string("mh_k_147_cast_fp16")]; + tensor var_5640 = const()[name = string("op_5640"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_75_cast_fp16 = reshape(shape = var_5640, x = mh_k_147_cast_fp16)[name = string("current_key_75_cast_fp16")]; + tensor var_5646_to_fp16 = const()[name = string("op_5646_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192256)))]; + tensor var_5647_cast_fp16 = mul(x = obj_163_cast_fp16, y = var_5646_to_fp16)[name = string("op_5647_cast_fp16")]; + tensor var_5644_to_fp16 = const()[name = string("op_5644_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192384)))]; + tensor var_5648_cast_fp16 = mul(x = current_key_75_cast_fp16, y = var_5644_to_fp16)[name = string("op_5648_cast_fp16")]; + tensor key_75_cast_fp16 = add(x = var_5647_cast_fp16, y = var_5648_cast_fp16)[name = string("key_75_cast_fp16")]; + tensor var_5650_to_fp16 = const()[name = string("op_5650_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192256)))]; + tensor var_5651_cast_fp16 = mul(x = obj_165_cast_fp16, y = var_5650_to_fp16)[name = string("op_5651_cast_fp16")]; + tensor var_5652_cast_fp16 = mul(x = current_value_37_cast_fp16, y = var_5644_to_fp16)[name = string("op_5652_cast_fp16")]; + tensor value_37_cast_fp16 = add(x = var_5651_cast_fp16, y = var_5652_cast_fp16)[name = string("value_37_cast_fp16")]; + fp16 var_5659_to_fp16 = const()[name = string("op_5659_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_151_cast_fp16 = mul(x = mh_q_147_cast_fp16, y = var_5659_to_fp16)[name = string("mh_q_151_cast_fp16")]; + tensor var_5661 = const()[name = string("op_5661"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_149_cast_fp16 = reshape(shape = var_5661, x = key_75_cast_fp16)[name = string("mh_k_149_cast_fp16")]; + tensor var_5663 = const()[name = string("op_5663"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_73_cast_fp16 = reshape(shape = var_5663, x = value_37_cast_fp16)[name = string("mh_v_73_cast_fp16")]; + tensor transpose_72_perm_0 = const()[name = string("transpose_72_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_36_reps_0 = const()[name = string("tile_36_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_72_cast_fp16 = transpose(perm = transpose_72_perm_0, x = mh_k_149_cast_fp16)[name = string("transpose_371")]; + tensor tile_36_cast_fp16 = tile(reps = tile_36_reps_0, x = transpose_72_cast_fp16)[name = string("tile_36_cast_fp16")]; + tensor concat_90 = const()[name = string("concat_90"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_72_cast_fp16 = reshape(shape = concat_90, x = tile_36_cast_fp16)[name = string("reshape_72_cast_fp16")]; + tensor transpose_73_perm_0 = const()[name = string("transpose_73_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_91 = const()[name = string("concat_91"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_73_cast_fp16 = transpose(perm = transpose_73_perm_0, x = reshape_72_cast_fp16)[name = string("transpose_370")]; + tensor reshape_73_cast_fp16 = reshape(shape = concat_91, x = transpose_73_cast_fp16)[name = string("reshape_73_cast_fp16")]; + tensor transpose_74_perm_0 = const()[name = string("transpose_74_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_37_reps_0 = const()[name = string("tile_37_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_74_cast_fp16 = transpose(perm = transpose_74_perm_0, x = mh_v_73_cast_fp16)[name = string("transpose_369")]; + tensor tile_37_cast_fp16 = tile(reps = tile_37_reps_0, x = transpose_74_cast_fp16)[name = string("tile_37_cast_fp16")]; + tensor concat_92 = const()[name = string("concat_92"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_74_cast_fp16 = reshape(shape = concat_92, x = tile_37_cast_fp16)[name = string("reshape_74_cast_fp16")]; + tensor transpose_75_perm_0 = const()[name = string("transpose_75_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_93 = const()[name = string("concat_93"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_75_cast_fp16 = transpose(perm = transpose_75_perm_0, x = reshape_74_cast_fp16)[name = string("transpose_368")]; + tensor reshape_75_cast_fp16 = reshape(shape = concat_93, x = transpose_75_cast_fp16)[name = string("reshape_75_cast_fp16")]; + tensor transpose_389_perm_0 = const()[name = string("transpose_389_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_109_transpose_x_1 = const()[name = string("mh_w_109_transpose_x_1"), val = bool(true)]; + bool mh_w_109_transpose_y_1 = const()[name = string("mh_w_109_transpose_y_1"), val = bool(false)]; + tensor transpose_389_cast_fp16 = transpose(perm = transpose_389_perm_0, x = reshape_73_cast_fp16)[name = string("transpose_367")]; + tensor mh_w_109_cast_fp16 = matmul(transpose_x = mh_w_109_transpose_x_1, transpose_y = mh_w_109_transpose_y_1, x = mh_q_151_cast_fp16, y = transpose_389_cast_fp16)[name = string("mh_w_109_cast_fp16")]; + tensor var_5671_to_fp16 = const()[name = string("op_5671_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192512)))]; + tensor mh_w_111_cast_fp16 = add(x = mh_w_109_cast_fp16, y = var_5671_to_fp16)[name = string("mh_w_111_cast_fp16")]; + tensor mh_w_113_cast_fp16 = softmax(axis = var_5491, x = mh_w_111_cast_fp16)[name = string("mh_w_113_cast_fp16")]; + tensor transpose_390_perm_0 = const()[name = string("transpose_390_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_37_transpose_x_1 = const()[name = string("attn_37_transpose_x_1"), val = bool(false)]; + bool attn_37_transpose_y_1 = const()[name = string("attn_37_transpose_y_1"), val = bool(true)]; + tensor transpose_390_cast_fp16 = transpose(perm = transpose_390_perm_0, x = reshape_75_cast_fp16)[name = string("transpose_366")]; + tensor attn_37_cast_fp16 = matmul(transpose_x = attn_37_transpose_x_1, transpose_y = attn_37_transpose_y_1, x = transpose_390_cast_fp16, y = mh_w_113_cast_fp16)[name = string("attn_37_cast_fp16")]; + tensor var_5677 = const()[name = string("op_5677"), val = tensor([1, 2048, 1, 1])]; + tensor input_153_cast_fp16 = reshape(shape = var_5677, x = attn_37_cast_fp16)[name = string("input_153_cast_fp16")]; + string obj_167_pad_type_0 = const()[name = string("obj_167_pad_type_0"), val = string("valid")]; + tensor obj_167_strides_0 = const()[name = string("obj_167_strides_0"), val = tensor([1, 1])]; + tensor obj_167_pad_0 = const()[name = string("obj_167_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_167_dilations_0 = const()[name = string("obj_167_dilations_0"), val = tensor([1, 1])]; + int32 obj_167_groups_0 = const()[name = string("obj_167_groups_0"), val = int32(1)]; + tensor obj_167_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_167_dilations_0, groups = obj_167_groups_0, pad = obj_167_pad_0, pad_type = obj_167_pad_type_0, strides = obj_167_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_153_cast_fp16)[name = string("obj_167_cast_fp16")]; + tensor inputs_155_cast_fp16 = add(x = inputs_149_cast_fp16, y = obj_167_cast_fp16)[name = string("inputs_155_cast_fp16")]; + tensor inputs_sq_155_cast_fp16 = mul(x = inputs_155_cast_fp16, y = inputs_155_cast_fp16)[name = string("inputs_sq_155_cast_fp16")]; + tensor variance_155_axes_0 = const()[name = string("variance_155_axes_0"), val = tensor([1])]; + bool variance_155_keep_dims_0 = const()[name = string("variance_155_keep_dims_0"), val = bool(true)]; + tensor variance_155_cast_fp16 = reduce_mean(axes = variance_155_axes_0, keep_dims = variance_155_keep_dims_0, x = inputs_sq_155_cast_fp16)[name = string("variance_155_cast_fp16")]; + fp16 var_5695_to_fp16 = const()[name = string("op_5695_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5696_cast_fp16 = add(x = variance_155_cast_fp16, y = var_5695_to_fp16)[name = string("op_5696_cast_fp16")]; + fp32 var_5697_epsilon_0 = const()[name = string("op_5697_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5697_cast_fp16 = rsqrt(epsilon = var_5697_epsilon_0, x = var_5696_cast_fp16)[name = string("op_5697_cast_fp16")]; + tensor hidden_states_191_cast_fp16 = mul(x = inputs_155_cast_fp16, y = var_5697_cast_fp16)[name = string("hidden_states_191_cast_fp16")]; + tensor input_155_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_191_cast_fp16)[name = string("input_155_cast_fp16")]; + string input_157_pad_type_0 = const()[name = string("input_157_pad_type_0"), val = string("valid")]; + tensor input_157_strides_0 = const()[name = string("input_157_strides_0"), val = tensor([1, 1])]; + tensor input_157_pad_0 = const()[name = string("input_157_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_157_dilations_0 = const()[name = string("input_157_dilations_0"), val = tensor([1, 1])]; + int32 input_157_groups_0 = const()[name = string("input_157_groups_0"), val = int32(1)]; + tensor input_157_cast_fp16 = conv(dilations = input_157_dilations_0, groups = input_157_groups_0, pad = input_157_pad_0, pad_type = input_157_pad_type_0, strides = input_157_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_155_cast_fp16)[name = string("input_157_cast_fp16")]; + tensor var_5711_cast_fp16 = silu(x = input_157_cast_fp16)[name = string("op_5711_cast_fp16")]; + string var_5717_pad_type_0 = const()[name = string("op_5717_pad_type_0"), val = string("valid")]; + tensor var_5717_strides_0 = const()[name = string("op_5717_strides_0"), val = tensor([1, 1])]; + tensor var_5717_pad_0 = const()[name = string("op_5717_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5717_dilations_0 = const()[name = string("op_5717_dilations_0"), val = tensor([1, 1])]; + int32 var_5717_groups_0 = const()[name = string("op_5717_groups_0"), val = int32(1)]; + tensor var_5717_cast_fp16 = conv(dilations = var_5717_dilations_0, groups = var_5717_groups_0, pad = var_5717_pad_0, pad_type = var_5717_pad_type_0, strides = var_5717_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_155_cast_fp16)[name = string("op_5717_cast_fp16")]; + tensor input_159_cast_fp16 = mul(x = var_5711_cast_fp16, y = var_5717_cast_fp16)[name = string("input_159_cast_fp16")]; + string hidden_states_193_pad_type_0 = const()[name = string("hidden_states_193_pad_type_0"), val = string("valid")]; + tensor hidden_states_193_strides_0 = const()[name = string("hidden_states_193_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_193_pad_0 = const()[name = string("hidden_states_193_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_193_dilations_0 = const()[name = string("hidden_states_193_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_193_groups_0 = const()[name = string("hidden_states_193_groups_0"), val = int32(1)]; + tensor hidden_states_193_cast_fp16 = conv(dilations = hidden_states_193_dilations_0, groups = hidden_states_193_groups_0, pad = hidden_states_193_pad_0, pad_type = hidden_states_193_pad_type_0, strides = hidden_states_193_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_159_cast_fp16)[name = string("hidden_states_193_cast_fp16")]; + tensor inputs_157_cast_fp16 = add(x = inputs_155_cast_fp16, y = hidden_states_193_cast_fp16)[name = string("inputs_157_cast_fp16")]; + tensor obj_171_begin_0 = const()[name = string("obj_171_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_171_end_0 = const()[name = string("obj_171_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_171_end_mask_0 = const()[name = string("obj_171_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_171_cast_fp16 = slice_by_index(begin = obj_171_begin_0, end = obj_171_end_0, end_mask = obj_171_end_mask_0, x = key_caches_7_cast_fp16)[name = string("obj_171_cast_fp16")]; + tensor obj_173_begin_0 = const()[name = string("obj_173_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_173_end_0 = const()[name = string("obj_173_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_173_end_mask_0 = const()[name = string("obj_173_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_173_cast_fp16 = slice_by_index(begin = obj_173_begin_0, end = obj_173_end_0, end_mask = obj_173_end_mask_0, x = value_caches_7_cast_fp16)[name = string("obj_173_cast_fp16")]; + int32 var_5765 = const()[name = string("op_5765"), val = int32(3)]; + int32 var_5775 = const()[name = string("op_5775"), val = int32(-2)]; + tensor inputs_sq_157_cast_fp16 = mul(x = inputs_157_cast_fp16, y = inputs_157_cast_fp16)[name = string("inputs_sq_157_cast_fp16")]; + tensor variance_157_axes_0 = const()[name = string("variance_157_axes_0"), val = tensor([1])]; + bool variance_157_keep_dims_0 = const()[name = string("variance_157_keep_dims_0"), val = bool(true)]; + tensor variance_157_cast_fp16 = reduce_mean(axes = variance_157_axes_0, keep_dims = variance_157_keep_dims_0, x = inputs_sq_157_cast_fp16)[name = string("variance_157_cast_fp16")]; + fp16 var_5789_to_fp16 = const()[name = string("op_5789_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5790_cast_fp16 = add(x = variance_157_cast_fp16, y = var_5789_to_fp16)[name = string("op_5790_cast_fp16")]; + fp32 var_5791_epsilon_0 = const()[name = string("op_5791_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5791_cast_fp16 = rsqrt(epsilon = var_5791_epsilon_0, x = var_5790_cast_fp16)[name = string("op_5791_cast_fp16")]; + tensor hidden_states_195_cast_fp16 = mul(x = inputs_157_cast_fp16, y = var_5791_cast_fp16)[name = string("hidden_states_195_cast_fp16")]; + tensor obj_169_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_195_cast_fp16)[name = string("obj_169_cast_fp16")]; + string query_115_pad_type_0 = const()[name = string("query_115_pad_type_0"), val = string("valid")]; + tensor query_115_strides_0 = const()[name = string("query_115_strides_0"), val = tensor([1, 1])]; + tensor query_115_pad_0 = const()[name = string("query_115_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_115_dilations_0 = const()[name = string("query_115_dilations_0"), val = tensor([1, 1])]; + int32 query_115_groups_0 = const()[name = string("query_115_groups_0"), val = int32(1)]; + tensor query_115_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_115_dilations_0, groups = query_115_groups_0, pad = query_115_pad_0, pad_type = query_115_pad_type_0, strides = query_115_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_169_cast_fp16)[name = string("query_115_cast_fp16")]; + string current_key_77_pad_type_0 = const()[name = string("current_key_77_pad_type_0"), val = string("valid")]; + tensor current_key_77_strides_0 = const()[name = string("current_key_77_strides_0"), val = tensor([1, 1])]; + tensor current_key_77_pad_0 = const()[name = string("current_key_77_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_77_dilations_0 = const()[name = string("current_key_77_dilations_0"), val = tensor([1, 1])]; + int32 current_key_77_groups_0 = const()[name = string("current_key_77_groups_0"), val = int32(1)]; + tensor current_key_77_cast_fp16 = conv(dilations = current_key_77_dilations_0, groups = current_key_77_groups_0, pad = current_key_77_pad_0, pad_type = current_key_77_pad_type_0, strides = current_key_77_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_169_cast_fp16)[name = string("current_key_77_cast_fp16")]; + string current_value_39_pad_type_0 = const()[name = string("current_value_39_pad_type_0"), val = string("valid")]; + tensor current_value_39_strides_0 = const()[name = string("current_value_39_strides_0"), val = tensor([1, 1])]; + tensor current_value_39_pad_0 = const()[name = string("current_value_39_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_39_dilations_0 = const()[name = string("current_value_39_dilations_0"), val = tensor([1, 1])]; + int32 current_value_39_groups_0 = const()[name = string("current_value_39_groups_0"), val = int32(1)]; + tensor current_value_39_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_39_dilations_0, groups = current_value_39_groups_0, pad = current_value_39_pad_0, pad_type = current_value_39_pad_type_0, strides = current_value_39_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_169_cast_fp16)[name = string("current_value_39_cast_fp16")]; + tensor var_5828 = const()[name = string("op_5828"), val = tensor([16, 128, 1, 1])]; + tensor inputs_159_cast_fp16 = reshape(shape = var_5828, x = query_115_cast_fp16)[name = string("inputs_159_cast_fp16")]; + tensor inputs_sq_159_cast_fp16 = mul(x = inputs_159_cast_fp16, y = inputs_159_cast_fp16)[name = string("inputs_sq_159_cast_fp16")]; + tensor variance_159_axes_0 = const()[name = string("variance_159_axes_0"), val = tensor([1])]; + bool variance_159_keep_dims_0 = const()[name = string("variance_159_keep_dims_0"), val = bool(true)]; + tensor variance_159_cast_fp16 = reduce_mean(axes = variance_159_axes_0, keep_dims = variance_159_keep_dims_0, x = inputs_sq_159_cast_fp16)[name = string("variance_159_cast_fp16")]; + fp16 var_5834_to_fp16 = const()[name = string("op_5834_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5835_cast_fp16 = add(x = variance_159_cast_fp16, y = var_5834_to_fp16)[name = string("op_5835_cast_fp16")]; + fp32 var_5836_epsilon_0 = const()[name = string("op_5836_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5836_cast_fp16 = rsqrt(epsilon = var_5836_epsilon_0, x = var_5835_cast_fp16)[name = string("op_5836_cast_fp16")]; + tensor hidden_states_197_cast_fp16 = mul(x = inputs_159_cast_fp16, y = var_5836_cast_fp16)[name = string("hidden_states_197_cast_fp16")]; + tensor query_normed_39_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_197_cast_fp16)[name = string("query_normed_39_cast_fp16")]; + tensor var_5844 = const()[name = string("op_5844"), val = tensor([8, 128, 1, 1])]; + tensor inputs_161_cast_fp16 = reshape(shape = var_5844, x = current_key_77_cast_fp16)[name = string("inputs_161_cast_fp16")]; + tensor inputs_sq_161_cast_fp16 = mul(x = inputs_161_cast_fp16, y = inputs_161_cast_fp16)[name = string("inputs_sq_161_cast_fp16")]; + tensor variance_161_axes_0 = const()[name = string("variance_161_axes_0"), val = tensor([1])]; + bool variance_161_keep_dims_0 = const()[name = string("variance_161_keep_dims_0"), val = bool(true)]; + tensor variance_161_cast_fp16 = reduce_mean(axes = variance_161_axes_0, keep_dims = variance_161_keep_dims_0, x = inputs_sq_161_cast_fp16)[name = string("variance_161_cast_fp16")]; + fp16 var_5850_to_fp16 = const()[name = string("op_5850_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5851_cast_fp16 = add(x = variance_161_cast_fp16, y = var_5850_to_fp16)[name = string("op_5851_cast_fp16")]; + fp32 var_5852_epsilon_0 = const()[name = string("op_5852_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5852_cast_fp16 = rsqrt(epsilon = var_5852_epsilon_0, x = var_5851_cast_fp16)[name = string("op_5852_cast_fp16")]; + tensor hidden_states_199_cast_fp16 = mul(x = inputs_161_cast_fp16, y = var_5852_cast_fp16)[name = string("hidden_states_199_cast_fp16")]; + tensor current_key_normed_39_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_199_cast_fp16)[name = string("current_key_normed_39_cast_fp16")]; + tensor var_5870 = const()[name = string("op_5870"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_153_cast_fp16 = reshape(shape = var_5870, x = query_normed_39_cast_fp16)[name = string("mh_q_153_cast_fp16")]; + tensor var_5872 = const()[name = string("op_5872"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_153_cast_fp16 = reshape(shape = var_5872, x = current_key_normed_39_cast_fp16)[name = string("mh_k_153_cast_fp16")]; + tensor var_5876_cast_fp16 = mul(x = mh_q_153_cast_fp16, y = cos_31_to_fp16)[name = string("op_5876_cast_fp16")]; + tensor var_5881_begin_0 = const()[name = string("op_5881_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5881_end_0 = const()[name = string("op_5881_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_5881_end_mask_0 = const()[name = string("op_5881_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_5881_cast_fp16 = slice_by_index(begin = var_5881_begin_0, end = var_5881_end_0, end_mask = var_5881_end_mask_0, x = mh_q_153_cast_fp16)[name = string("op_5881_cast_fp16")]; + tensor var_5887_begin_0 = const()[name = string("op_5887_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_5887_end_0 = const()[name = string("op_5887_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_5887_end_mask_0 = const()[name = string("op_5887_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_5887_cast_fp16 = slice_by_index(begin = var_5887_begin_0, end = var_5887_end_0, end_mask = var_5887_end_mask_0, x = mh_q_153_cast_fp16)[name = string("op_5887_cast_fp16")]; + fp16 const_397_promoted_to_fp16 = const()[name = string("const_397_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5889_cast_fp16 = mul(x = var_5887_cast_fp16, y = const_397_promoted_to_fp16)[name = string("op_5889_cast_fp16")]; + bool var_5891_interleave_0 = const()[name = string("op_5891_interleave_0"), val = bool(false)]; + tensor var_5891_cast_fp16 = concat(axis = var_5775, interleave = var_5891_interleave_0, values = (var_5889_cast_fp16, var_5881_cast_fp16))[name = string("op_5891_cast_fp16")]; + tensor var_5892_cast_fp16 = mul(x = var_5891_cast_fp16, y = sin_31_to_fp16)[name = string("op_5892_cast_fp16")]; + tensor mh_q_155_cast_fp16 = add(x = var_5876_cast_fp16, y = var_5892_cast_fp16)[name = string("mh_q_155_cast_fp16")]; + tensor var_5894_cast_fp16 = mul(x = mh_k_153_cast_fp16, y = cos_31_to_fp16)[name = string("op_5894_cast_fp16")]; + tensor var_5899_begin_0 = const()[name = string("op_5899_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5899_end_0 = const()[name = string("op_5899_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_5899_end_mask_0 = const()[name = string("op_5899_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_5899_cast_fp16 = slice_by_index(begin = var_5899_begin_0, end = var_5899_end_0, end_mask = var_5899_end_mask_0, x = mh_k_153_cast_fp16)[name = string("op_5899_cast_fp16")]; + tensor var_5905_begin_0 = const()[name = string("op_5905_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_5905_end_0 = const()[name = string("op_5905_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_5905_end_mask_0 = const()[name = string("op_5905_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_5905_cast_fp16 = slice_by_index(begin = var_5905_begin_0, end = var_5905_end_0, end_mask = var_5905_end_mask_0, x = mh_k_153_cast_fp16)[name = string("op_5905_cast_fp16")]; + fp16 const_400_promoted_to_fp16 = const()[name = string("const_400_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5907_cast_fp16 = mul(x = var_5905_cast_fp16, y = const_400_promoted_to_fp16)[name = string("op_5907_cast_fp16")]; + bool var_5909_interleave_0 = const()[name = string("op_5909_interleave_0"), val = bool(false)]; + tensor var_5909_cast_fp16 = concat(axis = var_5775, interleave = var_5909_interleave_0, values = (var_5907_cast_fp16, var_5899_cast_fp16))[name = string("op_5909_cast_fp16")]; + tensor var_5910_cast_fp16 = mul(x = var_5909_cast_fp16, y = sin_31_to_fp16)[name = string("op_5910_cast_fp16")]; + tensor mh_k_155_cast_fp16 = add(x = var_5894_cast_fp16, y = var_5910_cast_fp16)[name = string("mh_k_155_cast_fp16")]; + tensor var_5914 = const()[name = string("op_5914"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_79_cast_fp16 = reshape(shape = var_5914, x = mh_k_155_cast_fp16)[name = string("current_key_79_cast_fp16")]; + tensor var_5920_to_fp16 = const()[name = string("op_5920_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192256)))]; + tensor var_5921_cast_fp16 = mul(x = obj_171_cast_fp16, y = var_5920_to_fp16)[name = string("op_5921_cast_fp16")]; + tensor var_5918_to_fp16 = const()[name = string("op_5918_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192384)))]; + tensor var_5922_cast_fp16 = mul(x = current_key_79_cast_fp16, y = var_5918_to_fp16)[name = string("op_5922_cast_fp16")]; + tensor key_79_cast_fp16 = add(x = var_5921_cast_fp16, y = var_5922_cast_fp16)[name = string("key_79_cast_fp16")]; + tensor var_5924_to_fp16 = const()[name = string("op_5924_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192256)))]; + tensor var_5925_cast_fp16 = mul(x = obj_173_cast_fp16, y = var_5924_to_fp16)[name = string("op_5925_cast_fp16")]; + tensor var_5926_cast_fp16 = mul(x = current_value_39_cast_fp16, y = var_5918_to_fp16)[name = string("op_5926_cast_fp16")]; + tensor value_39_cast_fp16 = add(x = var_5925_cast_fp16, y = var_5926_cast_fp16)[name = string("value_39_cast_fp16")]; + fp16 var_5933_to_fp16 = const()[name = string("op_5933_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_159_cast_fp16 = mul(x = mh_q_155_cast_fp16, y = var_5933_to_fp16)[name = string("mh_q_159_cast_fp16")]; + tensor var_5935 = const()[name = string("op_5935"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_157_cast_fp16 = reshape(shape = var_5935, x = key_79_cast_fp16)[name = string("mh_k_157_cast_fp16")]; + tensor var_5937 = const()[name = string("op_5937"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_77_cast_fp16 = reshape(shape = var_5937, x = value_39_cast_fp16)[name = string("mh_v_77_cast_fp16")]; + tensor transpose_76_perm_0 = const()[name = string("transpose_76_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_38_reps_0 = const()[name = string("tile_38_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_76_cast_fp16 = transpose(perm = transpose_76_perm_0, x = mh_k_157_cast_fp16)[name = string("transpose_365")]; + tensor tile_38_cast_fp16 = tile(reps = tile_38_reps_0, x = transpose_76_cast_fp16)[name = string("tile_38_cast_fp16")]; + tensor concat_94 = const()[name = string("concat_94"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_76_cast_fp16 = reshape(shape = concat_94, x = tile_38_cast_fp16)[name = string("reshape_76_cast_fp16")]; + tensor transpose_77_perm_0 = const()[name = string("transpose_77_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_95 = const()[name = string("concat_95"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_77_cast_fp16 = transpose(perm = transpose_77_perm_0, x = reshape_76_cast_fp16)[name = string("transpose_364")]; + tensor reshape_77_cast_fp16 = reshape(shape = concat_95, x = transpose_77_cast_fp16)[name = string("reshape_77_cast_fp16")]; + tensor transpose_78_perm_0 = const()[name = string("transpose_78_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_39_reps_0 = const()[name = string("tile_39_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_78_cast_fp16 = transpose(perm = transpose_78_perm_0, x = mh_v_77_cast_fp16)[name = string("transpose_363")]; + tensor tile_39_cast_fp16 = tile(reps = tile_39_reps_0, x = transpose_78_cast_fp16)[name = string("tile_39_cast_fp16")]; + tensor concat_96 = const()[name = string("concat_96"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_78_cast_fp16 = reshape(shape = concat_96, x = tile_39_cast_fp16)[name = string("reshape_78_cast_fp16")]; + tensor transpose_79_perm_0 = const()[name = string("transpose_79_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_97 = const()[name = string("concat_97"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_79_cast_fp16 = transpose(perm = transpose_79_perm_0, x = reshape_78_cast_fp16)[name = string("transpose_362")]; + tensor reshape_79_cast_fp16 = reshape(shape = concat_97, x = transpose_79_cast_fp16)[name = string("reshape_79_cast_fp16")]; + tensor transpose_393_perm_0 = const()[name = string("transpose_393_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_115_transpose_x_1 = const()[name = string("mh_w_115_transpose_x_1"), val = bool(true)]; + bool mh_w_115_transpose_y_1 = const()[name = string("mh_w_115_transpose_y_1"), val = bool(false)]; + tensor transpose_393_cast_fp16 = transpose(perm = transpose_393_perm_0, x = reshape_77_cast_fp16)[name = string("transpose_361")]; + tensor mh_w_115_cast_fp16 = matmul(transpose_x = mh_w_115_transpose_x_1, transpose_y = mh_w_115_transpose_y_1, x = mh_q_159_cast_fp16, y = transpose_393_cast_fp16)[name = string("mh_w_115_cast_fp16")]; + tensor var_5945_to_fp16 = const()[name = string("op_5945_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192512)))]; + tensor mh_w_117_cast_fp16 = add(x = mh_w_115_cast_fp16, y = var_5945_to_fp16)[name = string("mh_w_117_cast_fp16")]; + tensor mh_w_119_cast_fp16 = softmax(axis = var_5765, x = mh_w_117_cast_fp16)[name = string("mh_w_119_cast_fp16")]; + tensor transpose_394_perm_0 = const()[name = string("transpose_394_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_39_transpose_x_1 = const()[name = string("attn_39_transpose_x_1"), val = bool(false)]; + bool attn_39_transpose_y_1 = const()[name = string("attn_39_transpose_y_1"), val = bool(true)]; + tensor transpose_394_cast_fp16 = transpose(perm = transpose_394_perm_0, x = reshape_79_cast_fp16)[name = string("transpose_360")]; + tensor attn_39_cast_fp16 = matmul(transpose_x = attn_39_transpose_x_1, transpose_y = attn_39_transpose_y_1, x = transpose_394_cast_fp16, y = mh_w_119_cast_fp16)[name = string("attn_39_cast_fp16")]; + tensor var_5951 = const()[name = string("op_5951"), val = tensor([1, 2048, 1, 1])]; + tensor input_161_cast_fp16 = reshape(shape = var_5951, x = attn_39_cast_fp16)[name = string("input_161_cast_fp16")]; + string obj_175_pad_type_0 = const()[name = string("obj_175_pad_type_0"), val = string("valid")]; + tensor obj_175_strides_0 = const()[name = string("obj_175_strides_0"), val = tensor([1, 1])]; + tensor obj_175_pad_0 = const()[name = string("obj_175_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_175_dilations_0 = const()[name = string("obj_175_dilations_0"), val = tensor([1, 1])]; + int32 obj_175_groups_0 = const()[name = string("obj_175_groups_0"), val = int32(1)]; + tensor obj_175_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_175_dilations_0, groups = obj_175_groups_0, pad = obj_175_pad_0, pad_type = obj_175_pad_type_0, strides = obj_175_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_161_cast_fp16)[name = string("obj_175_cast_fp16")]; + tensor inputs_163_cast_fp16 = add(x = inputs_157_cast_fp16, y = obj_175_cast_fp16)[name = string("inputs_163_cast_fp16")]; + tensor inputs_sq_163_cast_fp16 = mul(x = inputs_163_cast_fp16, y = inputs_163_cast_fp16)[name = string("inputs_sq_163_cast_fp16")]; + tensor variance_163_axes_0 = const()[name = string("variance_163_axes_0"), val = tensor([1])]; + bool variance_163_keep_dims_0 = const()[name = string("variance_163_keep_dims_0"), val = bool(true)]; + tensor variance_163_cast_fp16 = reduce_mean(axes = variance_163_axes_0, keep_dims = variance_163_keep_dims_0, x = inputs_sq_163_cast_fp16)[name = string("variance_163_cast_fp16")]; + fp16 var_5969_to_fp16 = const()[name = string("op_5969_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_5970_cast_fp16 = add(x = variance_163_cast_fp16, y = var_5969_to_fp16)[name = string("op_5970_cast_fp16")]; + fp32 var_5971_epsilon_0 = const()[name = string("op_5971_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_5971_cast_fp16 = rsqrt(epsilon = var_5971_epsilon_0, x = var_5970_cast_fp16)[name = string("op_5971_cast_fp16")]; + tensor hidden_states_201_cast_fp16 = mul(x = inputs_163_cast_fp16, y = var_5971_cast_fp16)[name = string("hidden_states_201_cast_fp16")]; + tensor input_163_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_201_cast_fp16)[name = string("input_163_cast_fp16")]; + string input_165_pad_type_0 = const()[name = string("input_165_pad_type_0"), val = string("valid")]; + tensor input_165_strides_0 = const()[name = string("input_165_strides_0"), val = tensor([1, 1])]; + tensor input_165_pad_0 = const()[name = string("input_165_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_165_dilations_0 = const()[name = string("input_165_dilations_0"), val = tensor([1, 1])]; + int32 input_165_groups_0 = const()[name = string("input_165_groups_0"), val = int32(1)]; + tensor input_165_cast_fp16 = conv(dilations = input_165_dilations_0, groups = input_165_groups_0, pad = input_165_pad_0, pad_type = input_165_pad_type_0, strides = input_165_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_163_cast_fp16)[name = string("input_165_cast_fp16")]; + tensor var_5985_cast_fp16 = silu(x = input_165_cast_fp16)[name = string("op_5985_cast_fp16")]; + string var_5991_pad_type_0 = const()[name = string("op_5991_pad_type_0"), val = string("valid")]; + tensor var_5991_strides_0 = const()[name = string("op_5991_strides_0"), val = tensor([1, 1])]; + tensor var_5991_pad_0 = const()[name = string("op_5991_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5991_dilations_0 = const()[name = string("op_5991_dilations_0"), val = tensor([1, 1])]; + int32 var_5991_groups_0 = const()[name = string("op_5991_groups_0"), val = int32(1)]; + tensor var_5991_cast_fp16 = conv(dilations = var_5991_dilations_0, groups = var_5991_groups_0, pad = var_5991_pad_0, pad_type = var_5991_pad_type_0, strides = var_5991_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_163_cast_fp16)[name = string("op_5991_cast_fp16")]; + tensor input_167_cast_fp16 = mul(x = var_5985_cast_fp16, y = var_5991_cast_fp16)[name = string("input_167_cast_fp16")]; + string hidden_states_203_pad_type_0 = const()[name = string("hidden_states_203_pad_type_0"), val = string("valid")]; + tensor hidden_states_203_strides_0 = const()[name = string("hidden_states_203_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_203_pad_0 = const()[name = string("hidden_states_203_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_203_dilations_0 = const()[name = string("hidden_states_203_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_203_groups_0 = const()[name = string("hidden_states_203_groups_0"), val = int32(1)]; + tensor hidden_states_203_cast_fp16 = conv(dilations = hidden_states_203_dilations_0, groups = hidden_states_203_groups_0, pad = hidden_states_203_pad_0, pad_type = hidden_states_203_pad_type_0, strides = hidden_states_203_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_167_cast_fp16)[name = string("hidden_states_203_cast_fp16")]; + tensor inputs_165_cast_fp16 = add(x = inputs_163_cast_fp16, y = hidden_states_203_cast_fp16)[name = string("inputs_165_cast_fp16")]; + int32 var_6019 = const()[name = string("op_6019"), val = int32(1)]; + bool key_caches_9_interleave_0 = const()[name = string("key_caches_9_interleave_0"), val = bool(false)]; + tensor key_caches_9_cast_fp16 = concat(axis = var_6019, interleave = key_caches_9_interleave_0, values = (key_63_cast_fp16, key_67_cast_fp16, key_71_cast_fp16, key_75_cast_fp16, key_79_cast_fp16))[name = string("key_caches_9_cast_fp16")]; + int32 var_6022 = const()[name = string("op_6022"), val = int32(1)]; + bool value_caches_9_interleave_0 = const()[name = string("value_caches_9_interleave_0"), val = bool(false)]; + tensor value_caches_9_cast_fp16 = concat(axis = var_6022, interleave = value_caches_9_interleave_0, values = (value_31_cast_fp16, value_33_cast_fp16, value_35_cast_fp16, value_37_cast_fp16, value_39_cast_fp16))[name = string("value_caches_9_cast_fp16")]; + tensor inputs_sq_165_cast_fp16 = mul(x = inputs_165_cast_fp16, y = inputs_165_cast_fp16)[name = string("inputs_sq_165_cast_fp16")]; + tensor variance_165_axes_0 = const()[name = string("variance_165_axes_0"), val = tensor([1])]; + bool variance_165_keep_dims_0 = const()[name = string("variance_165_keep_dims_0"), val = bool(true)]; + tensor variance_165_cast_fp16 = reduce_mean(axes = variance_165_axes_0, keep_dims = variance_165_keep_dims_0, x = inputs_sq_165_cast_fp16)[name = string("variance_165_cast_fp16")]; + fp16 var_6032_to_fp16 = const()[name = string("op_6032_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6033_cast_fp16 = add(x = variance_165_cast_fp16, y = var_6032_to_fp16)[name = string("op_6033_cast_fp16")]; + fp32 var_6034_epsilon_0 = const()[name = string("op_6034_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6034_cast_fp16 = rsqrt(epsilon = var_6034_epsilon_0, x = var_6033_cast_fp16)[name = string("op_6034_cast_fp16")]; + tensor hidden_states_205_cast_fp16 = mul(x = inputs_165_cast_fp16, y = var_6034_cast_fp16)[name = string("hidden_states_205_cast_fp16")]; + tensor input_169_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_205_cast_fp16)[name = string("input_169_cast_fp16")]; + string logits_9_pad_type_0 = const()[name = string("logits_9_pad_type_0"), val = string("valid")]; + tensor logits_9_strides_0 = const()[name = string("logits_9_strides_0"), val = tensor([1, 1])]; + tensor logits_9_pad_0 = const()[name = string("logits_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_9_dilations_0 = const()[name = string("logits_9_dilations_0"), val = tensor([1, 1])]; + int32 logits_9_groups_0 = const()[name = string("logits_9_groups_0"), val = int32(1)]; + tensor lm_heads_2_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85002176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87099392))))[name = string("lm_heads_2_weight_to_fp16_palettized")]; + tensor logits_9_cast_fp16 = conv(dilations = logits_9_dilations_0, groups = logits_9_groups_0, pad = logits_9_pad_0, pad_type = logits_9_pad_type_0, strides = logits_9_strides_0, weight = lm_heads_2_weight_to_fp16_palettized, x = input_169_cast_fp16)[name = string("logits_9_cast_fp16")]; + tensor var_6052 = const()[name = string("op_6052"), val = tensor([1, 2048])]; + tensor logits_11_cast_fp16 = reshape(shape = var_6052, x = logits_9_cast_fp16)[name = string("logits_11_cast_fp16")]; + tensor scaled_logits_5_cast_fp16 = real_div(x = logits_11_cast_fp16, y = temperature)[name = string("scaled_logits_5_cast_fp16")]; + int32 var_6062 = const()[name = string("op_6062"), val = int32(100)]; + int32 top_values_5_axis_0 = const()[name = string("top_values_5_axis_0"), val = int32(1)]; + bool top_values_5_ascending_0 = const()[name = string("top_values_5_ascending_0"), val = bool(false)]; + bool top_values_5_sort_0 = const()[name = string("top_values_5_sort_0"), val = bool(true)]; + bool top_values_5_return_indices_0 = const()[name = string("top_values_5_return_indices_0"), val = bool(true)]; + string top_values_5_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_5_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_5_cast_fp16_cast_uint16_0, tensor top_values_5_cast_fp16_cast_uint16_1 = topk(ascending = top_values_5_ascending_0, axis = top_values_5_axis_0, k = var_6062, output_indices_dtype = top_values_5_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_5_return_indices_0, sort = top_values_5_sort_0, x = scaled_logits_5_cast_fp16)[name = string("top_values_5_cast_fp16_cast_uint16")]; + tensor var_6068_cast_fp16 = mul(x = top_values_5_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_6068_cast_fp16")]; + tensor var_6072_cast_fp16 = add(x = var_6068_cast_fp16, y = var_3001_cast_fp16)[name = string("op_6072_cast_fp16")]; + tensor reduce_min_2_axes_0 = const()[name = string("reduce_min_2_axes_0"), val = tensor([1])]; + bool reduce_min_2_keep_dims_0 = const()[name = string("reduce_min_2_keep_dims_0"), val = bool(true)]; + tensor reduce_min_2_cast_fp16 = reduce_min(axes = reduce_min_2_axes_0, keep_dims = reduce_min_2_keep_dims_0, x = var_6072_cast_fp16)[name = string("reduce_min_2_cast_fp16")]; + tensor var_6075_cast_fp16 = greater_equal(x = scaled_logits_5_cast_fp16, y = reduce_min_2_cast_fp16)[name = string("op_6075_cast_fp16")]; + fp16 var_6076_value_0_to_fp16 = const()[name = string("op_6076_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_6076_cast_fp16 = fill_like(ref_tensor = scaled_logits_5_cast_fp16, value = var_6076_value_0_to_fp16)[name = string("op_6076_cast_fp16")]; + tensor masked_logits_5_cast_fp16 = select(a = scaled_logits_5_cast_fp16, b = var_6076_cast_fp16, cond = var_6075_cast_fp16)[name = string("masked_logits_5_cast_fp16")]; + tensor var_6080_begin_0 = const()[name = string("op_6080_begin_0"), val = tensor([2, 0])]; + tensor var_6080_end_0 = const()[name = string("op_6080_end_0"), val = tensor([3, 2048])]; + tensor var_6080_end_mask_0 = const()[name = string("op_6080_end_mask_0"), val = tensor([false, true])]; + tensor var_6080_squeeze_mask_0 = const()[name = string("op_6080_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_6080_cast_fp16 = slice_by_index(begin = var_6080_begin_0, end = var_6080_end_0, end_mask = var_6080_end_mask_0, squeeze_mask = var_6080_squeeze_mask_0, x = gumbel)[name = string("op_6080_cast_fp16")]; + tensor var_6083 = const()[name = string("op_6083"), val = tensor([1, 2048])]; + tensor var_6084_cast_fp16 = reshape(shape = var_6083, x = var_6080_cast_fp16)[name = string("op_6084_cast_fp16")]; + tensor noisy_logits_5_cast_fp16 = add(x = masked_logits_5_cast_fp16, y = var_6084_cast_fp16)[name = string("noisy_logits_5_cast_fp16")]; + int32 code_5_axis_0 = const()[name = string("code_5_axis_0"), val = int32(1)]; + bool code_5_keep_dims_0 = const()[name = string("code_5_keep_dims_0"), val = bool(false)]; + string code_5_output_dtype_0 = const()[name = string("code_5_output_dtype_0"), val = string("int32")]; + tensor code_5_cast_fp16 = reduce_argmax(axis = code_5_axis_0, keep_dims = code_5_keep_dims_0, output_dtype = code_5_output_dtype_0, x = noisy_logits_5_cast_fp16)[name = string("code_5_cast_fp16")]; + int32 var_6095 = const()[name = string("op_6095"), val = int32(4096)]; + tensor input_171 = add(x = code_5_cast_fp16, y = var_6095)[name = string("input_171")]; + int32 code_embed_9_axis_0 = const()[name = string("code_embed_9_axis_0"), val = int32(0)]; + int32 code_embed_9_batch_dims_0 = const()[name = string("code_embed_9_batch_dims_0"), val = int32(0)]; + bool code_embed_9_validate_indices_0 = const()[name = string("code_embed_9_validate_indices_0"), val = bool(false)]; + string input_171_to_uint16_dtype_0 = const()[name = string("input_171_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_171_to_uint16 = cast(dtype = input_171_to_uint16_dtype_0, x = input_171)[name = string("cast_12")]; + tensor code_embed_9_cast_fp16_cast_uint16 = gather(axis = code_embed_9_axis_0, batch_dims = code_embed_9_batch_dims_0, indices = input_171_to_uint16, validate_indices = code_embed_9_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_9_cast_fp16_cast_uint16")]; + tensor var_6099 = const()[name = string("op_6099"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_11_cast_fp16 = reshape(shape = var_6099, x = code_embed_9_cast_fp16_cast_uint16)[name = string("code_embed_11_cast_fp16")]; + tensor embed_sum_7_cast_fp16 = add(x = embed_sum_5_cast_fp16, y = code_embed_11_cast_fp16)[name = string("embed_sum_7_cast_fp16")]; + string inputs_167_pad_type_0 = const()[name = string("inputs_167_pad_type_0"), val = string("valid")]; + tensor inputs_167_strides_0 = const()[name = string("inputs_167_strides_0"), val = tensor([1, 1])]; + tensor inputs_167_pad_0 = const()[name = string("inputs_167_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_167_dilations_0 = const()[name = string("inputs_167_dilations_0"), val = tensor([1, 1])]; + int32 inputs_167_groups_0 = const()[name = string("inputs_167_groups_0"), val = int32(1)]; + tensor inputs_167_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_167_dilations_0, groups = inputs_167_groups_0, pad = inputs_167_pad_0, pad_type = inputs_167_pad_type_0, strides = inputs_167_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_11_cast_fp16)[name = string("inputs_167_cast_fp16")]; + tensor obj_179_begin_0 = const()[name = string("obj_179_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_179_end_0 = const()[name = string("obj_179_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_179_end_mask_0 = const()[name = string("obj_179_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_179_cast_fp16 = slice_by_index(begin = obj_179_begin_0, end = obj_179_end_0, end_mask = obj_179_end_mask_0, x = key_caches_9_cast_fp16)[name = string("obj_179_cast_fp16")]; + tensor obj_181_begin_0 = const()[name = string("obj_181_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_181_end_0 = const()[name = string("obj_181_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_181_end_mask_0 = const()[name = string("obj_181_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_181_cast_fp16 = slice_by_index(begin = obj_181_begin_0, end = obj_181_end_0, end_mask = obj_181_end_mask_0, x = value_caches_9_cast_fp16)[name = string("obj_181_cast_fp16")]; + int32 var_6204 = const()[name = string("op_6204"), val = int32(3)]; + int32 var_6214 = const()[name = string("op_6214"), val = int32(-2)]; + tensor inputs_sq_167_cast_fp16 = mul(x = inputs_167_cast_fp16, y = inputs_167_cast_fp16)[name = string("inputs_sq_167_cast_fp16")]; + tensor variance_167_axes_0 = const()[name = string("variance_167_axes_0"), val = tensor([1])]; + bool variance_167_keep_dims_0 = const()[name = string("variance_167_keep_dims_0"), val = bool(true)]; + tensor variance_167_cast_fp16 = reduce_mean(axes = variance_167_axes_0, keep_dims = variance_167_keep_dims_0, x = inputs_sq_167_cast_fp16)[name = string("variance_167_cast_fp16")]; + fp16 var_6228_to_fp16 = const()[name = string("op_6228_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6229_cast_fp16 = add(x = variance_167_cast_fp16, y = var_6228_to_fp16)[name = string("op_6229_cast_fp16")]; + fp32 var_6230_epsilon_0 = const()[name = string("op_6230_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6230_cast_fp16 = rsqrt(epsilon = var_6230_epsilon_0, x = var_6229_cast_fp16)[name = string("op_6230_cast_fp16")]; + tensor hidden_states_207_cast_fp16 = mul(x = inputs_167_cast_fp16, y = var_6230_cast_fp16)[name = string("hidden_states_207_cast_fp16")]; + tensor obj_177_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_207_cast_fp16)[name = string("obj_177_cast_fp16")]; + string query_121_pad_type_0 = const()[name = string("query_121_pad_type_0"), val = string("valid")]; + tensor query_121_strides_0 = const()[name = string("query_121_strides_0"), val = tensor([1, 1])]; + tensor query_121_pad_0 = const()[name = string("query_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_121_dilations_0 = const()[name = string("query_121_dilations_0"), val = tensor([1, 1])]; + int32 query_121_groups_0 = const()[name = string("query_121_groups_0"), val = int32(1)]; + tensor query_121_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_121_dilations_0, groups = query_121_groups_0, pad = query_121_pad_0, pad_type = query_121_pad_type_0, strides = query_121_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_177_cast_fp16)[name = string("query_121_cast_fp16")]; + string current_key_81_pad_type_0 = const()[name = string("current_key_81_pad_type_0"), val = string("valid")]; + tensor current_key_81_strides_0 = const()[name = string("current_key_81_strides_0"), val = tensor([1, 1])]; + tensor current_key_81_pad_0 = const()[name = string("current_key_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_81_dilations_0 = const()[name = string("current_key_81_dilations_0"), val = tensor([1, 1])]; + int32 current_key_81_groups_0 = const()[name = string("current_key_81_groups_0"), val = int32(1)]; + tensor current_key_81_cast_fp16 = conv(dilations = current_key_81_dilations_0, groups = current_key_81_groups_0, pad = current_key_81_pad_0, pad_type = current_key_81_pad_type_0, strides = current_key_81_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_177_cast_fp16)[name = string("current_key_81_cast_fp16")]; + string current_value_41_pad_type_0 = const()[name = string("current_value_41_pad_type_0"), val = string("valid")]; + tensor current_value_41_strides_0 = const()[name = string("current_value_41_strides_0"), val = tensor([1, 1])]; + tensor current_value_41_pad_0 = const()[name = string("current_value_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_41_dilations_0 = const()[name = string("current_value_41_dilations_0"), val = tensor([1, 1])]; + int32 current_value_41_groups_0 = const()[name = string("current_value_41_groups_0"), val = int32(1)]; + tensor current_value_41_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_41_dilations_0, groups = current_value_41_groups_0, pad = current_value_41_pad_0, pad_type = current_value_41_pad_type_0, strides = current_value_41_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_177_cast_fp16)[name = string("current_value_41_cast_fp16")]; + tensor var_6267 = const()[name = string("op_6267"), val = tensor([16, 128, 1, 1])]; + tensor inputs_169_cast_fp16 = reshape(shape = var_6267, x = query_121_cast_fp16)[name = string("inputs_169_cast_fp16")]; + tensor inputs_sq_169_cast_fp16 = mul(x = inputs_169_cast_fp16, y = inputs_169_cast_fp16)[name = string("inputs_sq_169_cast_fp16")]; + tensor variance_169_axes_0 = const()[name = string("variance_169_axes_0"), val = tensor([1])]; + bool variance_169_keep_dims_0 = const()[name = string("variance_169_keep_dims_0"), val = bool(true)]; + tensor variance_169_cast_fp16 = reduce_mean(axes = variance_169_axes_0, keep_dims = variance_169_keep_dims_0, x = inputs_sq_169_cast_fp16)[name = string("variance_169_cast_fp16")]; + fp16 var_6273_to_fp16 = const()[name = string("op_6273_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6274_cast_fp16 = add(x = variance_169_cast_fp16, y = var_6273_to_fp16)[name = string("op_6274_cast_fp16")]; + fp32 var_6275_epsilon_0 = const()[name = string("op_6275_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6275_cast_fp16 = rsqrt(epsilon = var_6275_epsilon_0, x = var_6274_cast_fp16)[name = string("op_6275_cast_fp16")]; + tensor hidden_states_209_cast_fp16 = mul(x = inputs_169_cast_fp16, y = var_6275_cast_fp16)[name = string("hidden_states_209_cast_fp16")]; + tensor query_normed_41_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_209_cast_fp16)[name = string("query_normed_41_cast_fp16")]; + tensor var_6283 = const()[name = string("op_6283"), val = tensor([8, 128, 1, 1])]; + tensor inputs_171_cast_fp16 = reshape(shape = var_6283, x = current_key_81_cast_fp16)[name = string("inputs_171_cast_fp16")]; + tensor inputs_sq_171_cast_fp16 = mul(x = inputs_171_cast_fp16, y = inputs_171_cast_fp16)[name = string("inputs_sq_171_cast_fp16")]; + tensor variance_171_axes_0 = const()[name = string("variance_171_axes_0"), val = tensor([1])]; + bool variance_171_keep_dims_0 = const()[name = string("variance_171_keep_dims_0"), val = bool(true)]; + tensor variance_171_cast_fp16 = reduce_mean(axes = variance_171_axes_0, keep_dims = variance_171_keep_dims_0, x = inputs_sq_171_cast_fp16)[name = string("variance_171_cast_fp16")]; + fp16 var_6289_to_fp16 = const()[name = string("op_6289_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6290_cast_fp16 = add(x = variance_171_cast_fp16, y = var_6289_to_fp16)[name = string("op_6290_cast_fp16")]; + fp32 var_6291_epsilon_0 = const()[name = string("op_6291_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6291_cast_fp16 = rsqrt(epsilon = var_6291_epsilon_0, x = var_6290_cast_fp16)[name = string("op_6291_cast_fp16")]; + tensor hidden_states_211_cast_fp16 = mul(x = inputs_171_cast_fp16, y = var_6291_cast_fp16)[name = string("hidden_states_211_cast_fp16")]; + tensor current_key_normed_41_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_211_cast_fp16)[name = string("current_key_normed_41_cast_fp16")]; + tensor var_6309 = const()[name = string("op_6309"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_161_cast_fp16 = reshape(shape = var_6309, x = query_normed_41_cast_fp16)[name = string("mh_q_161_cast_fp16")]; + tensor var_6311 = const()[name = string("op_6311"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_161_cast_fp16 = reshape(shape = var_6311, x = current_key_normed_41_cast_fp16)[name = string("mh_k_161_cast_fp16")]; + tensor cos_41_to_fp16 = const()[name = string("cos_41_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192640)))]; + tensor var_6315_cast_fp16 = mul(x = mh_q_161_cast_fp16, y = cos_41_to_fp16)[name = string("op_6315_cast_fp16")]; + tensor var_6320_begin_0 = const()[name = string("op_6320_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6320_end_0 = const()[name = string("op_6320_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_6320_end_mask_0 = const()[name = string("op_6320_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_6320_cast_fp16 = slice_by_index(begin = var_6320_begin_0, end = var_6320_end_0, end_mask = var_6320_end_mask_0, x = mh_q_161_cast_fp16)[name = string("op_6320_cast_fp16")]; + tensor var_6326_begin_0 = const()[name = string("op_6326_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_6326_end_0 = const()[name = string("op_6326_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_6326_end_mask_0 = const()[name = string("op_6326_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_6326_cast_fp16 = slice_by_index(begin = var_6326_begin_0, end = var_6326_end_0, end_mask = var_6326_end_mask_0, x = mh_q_161_cast_fp16)[name = string("op_6326_cast_fp16")]; + fp16 const_418_promoted_to_fp16 = const()[name = string("const_418_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6328_cast_fp16 = mul(x = var_6326_cast_fp16, y = const_418_promoted_to_fp16)[name = string("op_6328_cast_fp16")]; + bool var_6330_interleave_0 = const()[name = string("op_6330_interleave_0"), val = bool(false)]; + tensor var_6330_cast_fp16 = concat(axis = var_6214, interleave = var_6330_interleave_0, values = (var_6328_cast_fp16, var_6320_cast_fp16))[name = string("op_6330_cast_fp16")]; + tensor sin_41_to_fp16 = const()[name = string("sin_41_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175192960)))]; + tensor var_6331_cast_fp16 = mul(x = var_6330_cast_fp16, y = sin_41_to_fp16)[name = string("op_6331_cast_fp16")]; + tensor mh_q_163_cast_fp16 = add(x = var_6315_cast_fp16, y = var_6331_cast_fp16)[name = string("mh_q_163_cast_fp16")]; + tensor var_6333_cast_fp16 = mul(x = mh_k_161_cast_fp16, y = cos_41_to_fp16)[name = string("op_6333_cast_fp16")]; + tensor var_6338_begin_0 = const()[name = string("op_6338_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6338_end_0 = const()[name = string("op_6338_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_6338_end_mask_0 = const()[name = string("op_6338_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_6338_cast_fp16 = slice_by_index(begin = var_6338_begin_0, end = var_6338_end_0, end_mask = var_6338_end_mask_0, x = mh_k_161_cast_fp16)[name = string("op_6338_cast_fp16")]; + tensor var_6344_begin_0 = const()[name = string("op_6344_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_6344_end_0 = const()[name = string("op_6344_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_6344_end_mask_0 = const()[name = string("op_6344_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_6344_cast_fp16 = slice_by_index(begin = var_6344_begin_0, end = var_6344_end_0, end_mask = var_6344_end_mask_0, x = mh_k_161_cast_fp16)[name = string("op_6344_cast_fp16")]; + fp16 const_421_promoted_to_fp16 = const()[name = string("const_421_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6346_cast_fp16 = mul(x = var_6344_cast_fp16, y = const_421_promoted_to_fp16)[name = string("op_6346_cast_fp16")]; + bool var_6348_interleave_0 = const()[name = string("op_6348_interleave_0"), val = bool(false)]; + tensor var_6348_cast_fp16 = concat(axis = var_6214, interleave = var_6348_interleave_0, values = (var_6346_cast_fp16, var_6338_cast_fp16))[name = string("op_6348_cast_fp16")]; + tensor var_6349_cast_fp16 = mul(x = var_6348_cast_fp16, y = sin_41_to_fp16)[name = string("op_6349_cast_fp16")]; + tensor mh_k_163_cast_fp16 = add(x = var_6333_cast_fp16, y = var_6349_cast_fp16)[name = string("mh_k_163_cast_fp16")]; + tensor var_6353 = const()[name = string("op_6353"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_83_cast_fp16 = reshape(shape = var_6353, x = mh_k_163_cast_fp16)[name = string("current_key_83_cast_fp16")]; + tensor var_6359_to_fp16 = const()[name = string("op_6359_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193280)))]; + tensor var_6360_cast_fp16 = mul(x = obj_179_cast_fp16, y = var_6359_to_fp16)[name = string("op_6360_cast_fp16")]; + tensor var_6357_to_fp16 = const()[name = string("op_6357_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193408)))]; + tensor var_6361_cast_fp16 = mul(x = current_key_83_cast_fp16, y = var_6357_to_fp16)[name = string("op_6361_cast_fp16")]; + tensor key_83_cast_fp16 = add(x = var_6360_cast_fp16, y = var_6361_cast_fp16)[name = string("key_83_cast_fp16")]; + tensor var_6363_to_fp16 = const()[name = string("op_6363_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193280)))]; + tensor var_6364_cast_fp16 = mul(x = obj_181_cast_fp16, y = var_6363_to_fp16)[name = string("op_6364_cast_fp16")]; + tensor var_6365_cast_fp16 = mul(x = current_value_41_cast_fp16, y = var_6357_to_fp16)[name = string("op_6365_cast_fp16")]; + tensor value_41_cast_fp16 = add(x = var_6364_cast_fp16, y = var_6365_cast_fp16)[name = string("value_41_cast_fp16")]; + fp16 var_6372_to_fp16 = const()[name = string("op_6372_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_167_cast_fp16 = mul(x = mh_q_163_cast_fp16, y = var_6372_to_fp16)[name = string("mh_q_167_cast_fp16")]; + tensor var_6374 = const()[name = string("op_6374"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_165_cast_fp16 = reshape(shape = var_6374, x = key_83_cast_fp16)[name = string("mh_k_165_cast_fp16")]; + tensor var_6376 = const()[name = string("op_6376"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_81_cast_fp16 = reshape(shape = var_6376, x = value_41_cast_fp16)[name = string("mh_v_81_cast_fp16")]; + tensor transpose_80_perm_0 = const()[name = string("transpose_80_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_40_reps_0 = const()[name = string("tile_40_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_80_cast_fp16 = transpose(perm = transpose_80_perm_0, x = mh_k_165_cast_fp16)[name = string("transpose_359")]; + tensor tile_40_cast_fp16 = tile(reps = tile_40_reps_0, x = transpose_80_cast_fp16)[name = string("tile_40_cast_fp16")]; + tensor concat_103 = const()[name = string("concat_103"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_80_cast_fp16 = reshape(shape = concat_103, x = tile_40_cast_fp16)[name = string("reshape_80_cast_fp16")]; + tensor transpose_81_perm_0 = const()[name = string("transpose_81_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_104 = const()[name = string("concat_104"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_81_cast_fp16 = transpose(perm = transpose_81_perm_0, x = reshape_80_cast_fp16)[name = string("transpose_358")]; + tensor reshape_81_cast_fp16 = reshape(shape = concat_104, x = transpose_81_cast_fp16)[name = string("reshape_81_cast_fp16")]; + tensor transpose_82_perm_0 = const()[name = string("transpose_82_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_41_reps_0 = const()[name = string("tile_41_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_82_cast_fp16 = transpose(perm = transpose_82_perm_0, x = mh_v_81_cast_fp16)[name = string("transpose_357")]; + tensor tile_41_cast_fp16 = tile(reps = tile_41_reps_0, x = transpose_82_cast_fp16)[name = string("tile_41_cast_fp16")]; + tensor concat_105 = const()[name = string("concat_105"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_82_cast_fp16 = reshape(shape = concat_105, x = tile_41_cast_fp16)[name = string("reshape_82_cast_fp16")]; + tensor transpose_83_perm_0 = const()[name = string("transpose_83_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_106 = const()[name = string("concat_106"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_83_cast_fp16 = transpose(perm = transpose_83_perm_0, x = reshape_82_cast_fp16)[name = string("transpose_356")]; + tensor reshape_83_cast_fp16 = reshape(shape = concat_106, x = transpose_83_cast_fp16)[name = string("reshape_83_cast_fp16")]; + tensor transpose_397_perm_0 = const()[name = string("transpose_397_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_121_transpose_x_1 = const()[name = string("mh_w_121_transpose_x_1"), val = bool(true)]; + bool mh_w_121_transpose_y_1 = const()[name = string("mh_w_121_transpose_y_1"), val = bool(false)]; + tensor transpose_397_cast_fp16 = transpose(perm = transpose_397_perm_0, x = reshape_81_cast_fp16)[name = string("transpose_355")]; + tensor mh_w_121_cast_fp16 = matmul(transpose_x = mh_w_121_transpose_x_1, transpose_y = mh_w_121_transpose_y_1, x = mh_q_167_cast_fp16, y = transpose_397_cast_fp16)[name = string("mh_w_121_cast_fp16")]; + tensor var_6384_to_fp16 = const()[name = string("op_6384_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193536)))]; + tensor mh_w_123_cast_fp16 = add(x = mh_w_121_cast_fp16, y = var_6384_to_fp16)[name = string("mh_w_123_cast_fp16")]; + tensor mh_w_125_cast_fp16 = softmax(axis = var_6204, x = mh_w_123_cast_fp16)[name = string("mh_w_125_cast_fp16")]; + tensor transpose_398_perm_0 = const()[name = string("transpose_398_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_41_transpose_x_1 = const()[name = string("attn_41_transpose_x_1"), val = bool(false)]; + bool attn_41_transpose_y_1 = const()[name = string("attn_41_transpose_y_1"), val = bool(true)]; + tensor transpose_398_cast_fp16 = transpose(perm = transpose_398_perm_0, x = reshape_83_cast_fp16)[name = string("transpose_354")]; + tensor attn_41_cast_fp16 = matmul(transpose_x = attn_41_transpose_x_1, transpose_y = attn_41_transpose_y_1, x = transpose_398_cast_fp16, y = mh_w_125_cast_fp16)[name = string("attn_41_cast_fp16")]; + tensor var_6390 = const()[name = string("op_6390"), val = tensor([1, 2048, 1, 1])]; + tensor input_173_cast_fp16 = reshape(shape = var_6390, x = attn_41_cast_fp16)[name = string("input_173_cast_fp16")]; + string obj_187_pad_type_0 = const()[name = string("obj_187_pad_type_0"), val = string("valid")]; + tensor obj_187_strides_0 = const()[name = string("obj_187_strides_0"), val = tensor([1, 1])]; + tensor obj_187_pad_0 = const()[name = string("obj_187_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_187_dilations_0 = const()[name = string("obj_187_dilations_0"), val = tensor([1, 1])]; + int32 obj_187_groups_0 = const()[name = string("obj_187_groups_0"), val = int32(1)]; + tensor obj_187_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_187_dilations_0, groups = obj_187_groups_0, pad = obj_187_pad_0, pad_type = obj_187_pad_type_0, strides = obj_187_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_173_cast_fp16)[name = string("obj_187_cast_fp16")]; + tensor inputs_173_cast_fp16 = add(x = inputs_167_cast_fp16, y = obj_187_cast_fp16)[name = string("inputs_173_cast_fp16")]; + tensor inputs_sq_173_cast_fp16 = mul(x = inputs_173_cast_fp16, y = inputs_173_cast_fp16)[name = string("inputs_sq_173_cast_fp16")]; + tensor variance_173_axes_0 = const()[name = string("variance_173_axes_0"), val = tensor([1])]; + bool variance_173_keep_dims_0 = const()[name = string("variance_173_keep_dims_0"), val = bool(true)]; + tensor variance_173_cast_fp16 = reduce_mean(axes = variance_173_axes_0, keep_dims = variance_173_keep_dims_0, x = inputs_sq_173_cast_fp16)[name = string("variance_173_cast_fp16")]; + fp16 var_6408_to_fp16 = const()[name = string("op_6408_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6409_cast_fp16 = add(x = variance_173_cast_fp16, y = var_6408_to_fp16)[name = string("op_6409_cast_fp16")]; + fp32 var_6410_epsilon_0 = const()[name = string("op_6410_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6410_cast_fp16 = rsqrt(epsilon = var_6410_epsilon_0, x = var_6409_cast_fp16)[name = string("op_6410_cast_fp16")]; + tensor hidden_states_213_cast_fp16 = mul(x = inputs_173_cast_fp16, y = var_6410_cast_fp16)[name = string("hidden_states_213_cast_fp16")]; + tensor input_175_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_213_cast_fp16)[name = string("input_175_cast_fp16")]; + string input_177_pad_type_0 = const()[name = string("input_177_pad_type_0"), val = string("valid")]; + tensor input_177_strides_0 = const()[name = string("input_177_strides_0"), val = tensor([1, 1])]; + tensor input_177_pad_0 = const()[name = string("input_177_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_177_dilations_0 = const()[name = string("input_177_dilations_0"), val = tensor([1, 1])]; + int32 input_177_groups_0 = const()[name = string("input_177_groups_0"), val = int32(1)]; + tensor input_177_cast_fp16 = conv(dilations = input_177_dilations_0, groups = input_177_groups_0, pad = input_177_pad_0, pad_type = input_177_pad_type_0, strides = input_177_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_175_cast_fp16)[name = string("input_177_cast_fp16")]; + tensor var_6424_cast_fp16 = silu(x = input_177_cast_fp16)[name = string("op_6424_cast_fp16")]; + string var_6430_pad_type_0 = const()[name = string("op_6430_pad_type_0"), val = string("valid")]; + tensor var_6430_strides_0 = const()[name = string("op_6430_strides_0"), val = tensor([1, 1])]; + tensor var_6430_pad_0 = const()[name = string("op_6430_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6430_dilations_0 = const()[name = string("op_6430_dilations_0"), val = tensor([1, 1])]; + int32 var_6430_groups_0 = const()[name = string("op_6430_groups_0"), val = int32(1)]; + tensor var_6430_cast_fp16 = conv(dilations = var_6430_dilations_0, groups = var_6430_groups_0, pad = var_6430_pad_0, pad_type = var_6430_pad_type_0, strides = var_6430_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_175_cast_fp16)[name = string("op_6430_cast_fp16")]; + tensor input_179_cast_fp16 = mul(x = var_6424_cast_fp16, y = var_6430_cast_fp16)[name = string("input_179_cast_fp16")]; + string hidden_states_215_pad_type_0 = const()[name = string("hidden_states_215_pad_type_0"), val = string("valid")]; + tensor hidden_states_215_strides_0 = const()[name = string("hidden_states_215_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_215_pad_0 = const()[name = string("hidden_states_215_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_215_dilations_0 = const()[name = string("hidden_states_215_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_215_groups_0 = const()[name = string("hidden_states_215_groups_0"), val = int32(1)]; + tensor hidden_states_215_cast_fp16 = conv(dilations = hidden_states_215_dilations_0, groups = hidden_states_215_groups_0, pad = hidden_states_215_pad_0, pad_type = hidden_states_215_pad_type_0, strides = hidden_states_215_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_179_cast_fp16)[name = string("hidden_states_215_cast_fp16")]; + tensor inputs_175_cast_fp16 = add(x = inputs_173_cast_fp16, y = hidden_states_215_cast_fp16)[name = string("inputs_175_cast_fp16")]; + tensor obj_191_begin_0 = const()[name = string("obj_191_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_191_end_0 = const()[name = string("obj_191_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_191_end_mask_0 = const()[name = string("obj_191_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_191_cast_fp16 = slice_by_index(begin = obj_191_begin_0, end = obj_191_end_0, end_mask = obj_191_end_mask_0, x = key_caches_9_cast_fp16)[name = string("obj_191_cast_fp16")]; + tensor obj_193_begin_0 = const()[name = string("obj_193_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_193_end_0 = const()[name = string("obj_193_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_193_end_mask_0 = const()[name = string("obj_193_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_193_cast_fp16 = slice_by_index(begin = obj_193_begin_0, end = obj_193_end_0, end_mask = obj_193_end_mask_0, x = value_caches_9_cast_fp16)[name = string("obj_193_cast_fp16")]; + int32 var_6478 = const()[name = string("op_6478"), val = int32(3)]; + int32 var_6488 = const()[name = string("op_6488"), val = int32(-2)]; + tensor inputs_sq_175_cast_fp16 = mul(x = inputs_175_cast_fp16, y = inputs_175_cast_fp16)[name = string("inputs_sq_175_cast_fp16")]; + tensor variance_175_axes_0 = const()[name = string("variance_175_axes_0"), val = tensor([1])]; + bool variance_175_keep_dims_0 = const()[name = string("variance_175_keep_dims_0"), val = bool(true)]; + tensor variance_175_cast_fp16 = reduce_mean(axes = variance_175_axes_0, keep_dims = variance_175_keep_dims_0, x = inputs_sq_175_cast_fp16)[name = string("variance_175_cast_fp16")]; + fp16 var_6502_to_fp16 = const()[name = string("op_6502_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6503_cast_fp16 = add(x = variance_175_cast_fp16, y = var_6502_to_fp16)[name = string("op_6503_cast_fp16")]; + fp32 var_6504_epsilon_0 = const()[name = string("op_6504_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6504_cast_fp16 = rsqrt(epsilon = var_6504_epsilon_0, x = var_6503_cast_fp16)[name = string("op_6504_cast_fp16")]; + tensor hidden_states_217_cast_fp16 = mul(x = inputs_175_cast_fp16, y = var_6504_cast_fp16)[name = string("hidden_states_217_cast_fp16")]; + tensor obj_189_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_217_cast_fp16)[name = string("obj_189_cast_fp16")]; + string query_127_pad_type_0 = const()[name = string("query_127_pad_type_0"), val = string("valid")]; + tensor query_127_strides_0 = const()[name = string("query_127_strides_0"), val = tensor([1, 1])]; + tensor query_127_pad_0 = const()[name = string("query_127_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_127_dilations_0 = const()[name = string("query_127_dilations_0"), val = tensor([1, 1])]; + int32 query_127_groups_0 = const()[name = string("query_127_groups_0"), val = int32(1)]; + tensor query_127_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_127_dilations_0, groups = query_127_groups_0, pad = query_127_pad_0, pad_type = query_127_pad_type_0, strides = query_127_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_189_cast_fp16)[name = string("query_127_cast_fp16")]; + string current_key_85_pad_type_0 = const()[name = string("current_key_85_pad_type_0"), val = string("valid")]; + tensor current_key_85_strides_0 = const()[name = string("current_key_85_strides_0"), val = tensor([1, 1])]; + tensor current_key_85_pad_0 = const()[name = string("current_key_85_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_85_dilations_0 = const()[name = string("current_key_85_dilations_0"), val = tensor([1, 1])]; + int32 current_key_85_groups_0 = const()[name = string("current_key_85_groups_0"), val = int32(1)]; + tensor current_key_85_cast_fp16 = conv(dilations = current_key_85_dilations_0, groups = current_key_85_groups_0, pad = current_key_85_pad_0, pad_type = current_key_85_pad_type_0, strides = current_key_85_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_189_cast_fp16)[name = string("current_key_85_cast_fp16")]; + string current_value_43_pad_type_0 = const()[name = string("current_value_43_pad_type_0"), val = string("valid")]; + tensor current_value_43_strides_0 = const()[name = string("current_value_43_strides_0"), val = tensor([1, 1])]; + tensor current_value_43_pad_0 = const()[name = string("current_value_43_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_43_dilations_0 = const()[name = string("current_value_43_dilations_0"), val = tensor([1, 1])]; + int32 current_value_43_groups_0 = const()[name = string("current_value_43_groups_0"), val = int32(1)]; + tensor current_value_43_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_43_dilations_0, groups = current_value_43_groups_0, pad = current_value_43_pad_0, pad_type = current_value_43_pad_type_0, strides = current_value_43_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_189_cast_fp16)[name = string("current_value_43_cast_fp16")]; + tensor var_6541 = const()[name = string("op_6541"), val = tensor([16, 128, 1, 1])]; + tensor inputs_177_cast_fp16 = reshape(shape = var_6541, x = query_127_cast_fp16)[name = string("inputs_177_cast_fp16")]; + tensor inputs_sq_177_cast_fp16 = mul(x = inputs_177_cast_fp16, y = inputs_177_cast_fp16)[name = string("inputs_sq_177_cast_fp16")]; + tensor variance_177_axes_0 = const()[name = string("variance_177_axes_0"), val = tensor([1])]; + bool variance_177_keep_dims_0 = const()[name = string("variance_177_keep_dims_0"), val = bool(true)]; + tensor variance_177_cast_fp16 = reduce_mean(axes = variance_177_axes_0, keep_dims = variance_177_keep_dims_0, x = inputs_sq_177_cast_fp16)[name = string("variance_177_cast_fp16")]; + fp16 var_6547_to_fp16 = const()[name = string("op_6547_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6548_cast_fp16 = add(x = variance_177_cast_fp16, y = var_6547_to_fp16)[name = string("op_6548_cast_fp16")]; + fp32 var_6549_epsilon_0 = const()[name = string("op_6549_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6549_cast_fp16 = rsqrt(epsilon = var_6549_epsilon_0, x = var_6548_cast_fp16)[name = string("op_6549_cast_fp16")]; + tensor hidden_states_219_cast_fp16 = mul(x = inputs_177_cast_fp16, y = var_6549_cast_fp16)[name = string("hidden_states_219_cast_fp16")]; + tensor query_normed_43_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_219_cast_fp16)[name = string("query_normed_43_cast_fp16")]; + tensor var_6557 = const()[name = string("op_6557"), val = tensor([8, 128, 1, 1])]; + tensor inputs_179_cast_fp16 = reshape(shape = var_6557, x = current_key_85_cast_fp16)[name = string("inputs_179_cast_fp16")]; + tensor inputs_sq_179_cast_fp16 = mul(x = inputs_179_cast_fp16, y = inputs_179_cast_fp16)[name = string("inputs_sq_179_cast_fp16")]; + tensor variance_179_axes_0 = const()[name = string("variance_179_axes_0"), val = tensor([1])]; + bool variance_179_keep_dims_0 = const()[name = string("variance_179_keep_dims_0"), val = bool(true)]; + tensor variance_179_cast_fp16 = reduce_mean(axes = variance_179_axes_0, keep_dims = variance_179_keep_dims_0, x = inputs_sq_179_cast_fp16)[name = string("variance_179_cast_fp16")]; + fp16 var_6563_to_fp16 = const()[name = string("op_6563_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6564_cast_fp16 = add(x = variance_179_cast_fp16, y = var_6563_to_fp16)[name = string("op_6564_cast_fp16")]; + fp32 var_6565_epsilon_0 = const()[name = string("op_6565_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6565_cast_fp16 = rsqrt(epsilon = var_6565_epsilon_0, x = var_6564_cast_fp16)[name = string("op_6565_cast_fp16")]; + tensor hidden_states_221_cast_fp16 = mul(x = inputs_179_cast_fp16, y = var_6565_cast_fp16)[name = string("hidden_states_221_cast_fp16")]; + tensor current_key_normed_43_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_221_cast_fp16)[name = string("current_key_normed_43_cast_fp16")]; + tensor var_6583 = const()[name = string("op_6583"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_169_cast_fp16 = reshape(shape = var_6583, x = query_normed_43_cast_fp16)[name = string("mh_q_169_cast_fp16")]; + tensor var_6585 = const()[name = string("op_6585"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_169_cast_fp16 = reshape(shape = var_6585, x = current_key_normed_43_cast_fp16)[name = string("mh_k_169_cast_fp16")]; + tensor var_6589_cast_fp16 = mul(x = mh_q_169_cast_fp16, y = cos_41_to_fp16)[name = string("op_6589_cast_fp16")]; + tensor var_6594_begin_0 = const()[name = string("op_6594_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6594_end_0 = const()[name = string("op_6594_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_6594_end_mask_0 = const()[name = string("op_6594_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_6594_cast_fp16 = slice_by_index(begin = var_6594_begin_0, end = var_6594_end_0, end_mask = var_6594_end_mask_0, x = mh_q_169_cast_fp16)[name = string("op_6594_cast_fp16")]; + tensor var_6600_begin_0 = const()[name = string("op_6600_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_6600_end_0 = const()[name = string("op_6600_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_6600_end_mask_0 = const()[name = string("op_6600_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_6600_cast_fp16 = slice_by_index(begin = var_6600_begin_0, end = var_6600_end_0, end_mask = var_6600_end_mask_0, x = mh_q_169_cast_fp16)[name = string("op_6600_cast_fp16")]; + fp16 const_438_promoted_to_fp16 = const()[name = string("const_438_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6602_cast_fp16 = mul(x = var_6600_cast_fp16, y = const_438_promoted_to_fp16)[name = string("op_6602_cast_fp16")]; + bool var_6604_interleave_0 = const()[name = string("op_6604_interleave_0"), val = bool(false)]; + tensor var_6604_cast_fp16 = concat(axis = var_6488, interleave = var_6604_interleave_0, values = (var_6602_cast_fp16, var_6594_cast_fp16))[name = string("op_6604_cast_fp16")]; + tensor var_6605_cast_fp16 = mul(x = var_6604_cast_fp16, y = sin_41_to_fp16)[name = string("op_6605_cast_fp16")]; + tensor mh_q_171_cast_fp16 = add(x = var_6589_cast_fp16, y = var_6605_cast_fp16)[name = string("mh_q_171_cast_fp16")]; + tensor var_6607_cast_fp16 = mul(x = mh_k_169_cast_fp16, y = cos_41_to_fp16)[name = string("op_6607_cast_fp16")]; + tensor var_6612_begin_0 = const()[name = string("op_6612_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6612_end_0 = const()[name = string("op_6612_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_6612_end_mask_0 = const()[name = string("op_6612_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_6612_cast_fp16 = slice_by_index(begin = var_6612_begin_0, end = var_6612_end_0, end_mask = var_6612_end_mask_0, x = mh_k_169_cast_fp16)[name = string("op_6612_cast_fp16")]; + tensor var_6618_begin_0 = const()[name = string("op_6618_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_6618_end_0 = const()[name = string("op_6618_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_6618_end_mask_0 = const()[name = string("op_6618_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_6618_cast_fp16 = slice_by_index(begin = var_6618_begin_0, end = var_6618_end_0, end_mask = var_6618_end_mask_0, x = mh_k_169_cast_fp16)[name = string("op_6618_cast_fp16")]; + fp16 const_441_promoted_to_fp16 = const()[name = string("const_441_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6620_cast_fp16 = mul(x = var_6618_cast_fp16, y = const_441_promoted_to_fp16)[name = string("op_6620_cast_fp16")]; + bool var_6622_interleave_0 = const()[name = string("op_6622_interleave_0"), val = bool(false)]; + tensor var_6622_cast_fp16 = concat(axis = var_6488, interleave = var_6622_interleave_0, values = (var_6620_cast_fp16, var_6612_cast_fp16))[name = string("op_6622_cast_fp16")]; + tensor var_6623_cast_fp16 = mul(x = var_6622_cast_fp16, y = sin_41_to_fp16)[name = string("op_6623_cast_fp16")]; + tensor mh_k_171_cast_fp16 = add(x = var_6607_cast_fp16, y = var_6623_cast_fp16)[name = string("mh_k_171_cast_fp16")]; + tensor var_6627 = const()[name = string("op_6627"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_87_cast_fp16 = reshape(shape = var_6627, x = mh_k_171_cast_fp16)[name = string("current_key_87_cast_fp16")]; + tensor var_6633_to_fp16 = const()[name = string("op_6633_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193280)))]; + tensor var_6634_cast_fp16 = mul(x = obj_191_cast_fp16, y = var_6633_to_fp16)[name = string("op_6634_cast_fp16")]; + tensor var_6631_to_fp16 = const()[name = string("op_6631_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193408)))]; + tensor var_6635_cast_fp16 = mul(x = current_key_87_cast_fp16, y = var_6631_to_fp16)[name = string("op_6635_cast_fp16")]; + tensor key_87_cast_fp16 = add(x = var_6634_cast_fp16, y = var_6635_cast_fp16)[name = string("key_87_cast_fp16")]; + tensor var_6637_to_fp16 = const()[name = string("op_6637_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193280)))]; + tensor var_6638_cast_fp16 = mul(x = obj_193_cast_fp16, y = var_6637_to_fp16)[name = string("op_6638_cast_fp16")]; + tensor var_6639_cast_fp16 = mul(x = current_value_43_cast_fp16, y = var_6631_to_fp16)[name = string("op_6639_cast_fp16")]; + tensor value_43_cast_fp16 = add(x = var_6638_cast_fp16, y = var_6639_cast_fp16)[name = string("value_43_cast_fp16")]; + fp16 var_6646_to_fp16 = const()[name = string("op_6646_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_175_cast_fp16 = mul(x = mh_q_171_cast_fp16, y = var_6646_to_fp16)[name = string("mh_q_175_cast_fp16")]; + tensor var_6648 = const()[name = string("op_6648"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_173_cast_fp16 = reshape(shape = var_6648, x = key_87_cast_fp16)[name = string("mh_k_173_cast_fp16")]; + tensor var_6650 = const()[name = string("op_6650"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_85_cast_fp16 = reshape(shape = var_6650, x = value_43_cast_fp16)[name = string("mh_v_85_cast_fp16")]; + tensor transpose_84_perm_0 = const()[name = string("transpose_84_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_42_reps_0 = const()[name = string("tile_42_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_84_cast_fp16 = transpose(perm = transpose_84_perm_0, x = mh_k_173_cast_fp16)[name = string("transpose_353")]; + tensor tile_42_cast_fp16 = tile(reps = tile_42_reps_0, x = transpose_84_cast_fp16)[name = string("tile_42_cast_fp16")]; + tensor concat_107 = const()[name = string("concat_107"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_84_cast_fp16 = reshape(shape = concat_107, x = tile_42_cast_fp16)[name = string("reshape_84_cast_fp16")]; + tensor transpose_85_perm_0 = const()[name = string("transpose_85_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_108 = const()[name = string("concat_108"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_85_cast_fp16 = transpose(perm = transpose_85_perm_0, x = reshape_84_cast_fp16)[name = string("transpose_352")]; + tensor reshape_85_cast_fp16 = reshape(shape = concat_108, x = transpose_85_cast_fp16)[name = string("reshape_85_cast_fp16")]; + tensor transpose_86_perm_0 = const()[name = string("transpose_86_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_43_reps_0 = const()[name = string("tile_43_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_86_cast_fp16 = transpose(perm = transpose_86_perm_0, x = mh_v_85_cast_fp16)[name = string("transpose_351")]; + tensor tile_43_cast_fp16 = tile(reps = tile_43_reps_0, x = transpose_86_cast_fp16)[name = string("tile_43_cast_fp16")]; + tensor concat_109 = const()[name = string("concat_109"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_86_cast_fp16 = reshape(shape = concat_109, x = tile_43_cast_fp16)[name = string("reshape_86_cast_fp16")]; + tensor transpose_87_perm_0 = const()[name = string("transpose_87_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_110 = const()[name = string("concat_110"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_87_cast_fp16 = transpose(perm = transpose_87_perm_0, x = reshape_86_cast_fp16)[name = string("transpose_350")]; + tensor reshape_87_cast_fp16 = reshape(shape = concat_110, x = transpose_87_cast_fp16)[name = string("reshape_87_cast_fp16")]; + tensor transpose_401_perm_0 = const()[name = string("transpose_401_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_127_transpose_x_1 = const()[name = string("mh_w_127_transpose_x_1"), val = bool(true)]; + bool mh_w_127_transpose_y_1 = const()[name = string("mh_w_127_transpose_y_1"), val = bool(false)]; + tensor transpose_401_cast_fp16 = transpose(perm = transpose_401_perm_0, x = reshape_85_cast_fp16)[name = string("transpose_349")]; + tensor mh_w_127_cast_fp16 = matmul(transpose_x = mh_w_127_transpose_x_1, transpose_y = mh_w_127_transpose_y_1, x = mh_q_175_cast_fp16, y = transpose_401_cast_fp16)[name = string("mh_w_127_cast_fp16")]; + tensor var_6658_to_fp16 = const()[name = string("op_6658_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193536)))]; + tensor mh_w_129_cast_fp16 = add(x = mh_w_127_cast_fp16, y = var_6658_to_fp16)[name = string("mh_w_129_cast_fp16")]; + tensor mh_w_131_cast_fp16 = softmax(axis = var_6478, x = mh_w_129_cast_fp16)[name = string("mh_w_131_cast_fp16")]; + tensor transpose_402_perm_0 = const()[name = string("transpose_402_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_43_transpose_x_1 = const()[name = string("attn_43_transpose_x_1"), val = bool(false)]; + bool attn_43_transpose_y_1 = const()[name = string("attn_43_transpose_y_1"), val = bool(true)]; + tensor transpose_402_cast_fp16 = transpose(perm = transpose_402_perm_0, x = reshape_87_cast_fp16)[name = string("transpose_348")]; + tensor attn_43_cast_fp16 = matmul(transpose_x = attn_43_transpose_x_1, transpose_y = attn_43_transpose_y_1, x = transpose_402_cast_fp16, y = mh_w_131_cast_fp16)[name = string("attn_43_cast_fp16")]; + tensor var_6664 = const()[name = string("op_6664"), val = tensor([1, 2048, 1, 1])]; + tensor input_181_cast_fp16 = reshape(shape = var_6664, x = attn_43_cast_fp16)[name = string("input_181_cast_fp16")]; + string obj_195_pad_type_0 = const()[name = string("obj_195_pad_type_0"), val = string("valid")]; + tensor obj_195_strides_0 = const()[name = string("obj_195_strides_0"), val = tensor([1, 1])]; + tensor obj_195_pad_0 = const()[name = string("obj_195_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_195_dilations_0 = const()[name = string("obj_195_dilations_0"), val = tensor([1, 1])]; + int32 obj_195_groups_0 = const()[name = string("obj_195_groups_0"), val = int32(1)]; + tensor obj_195_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_195_dilations_0, groups = obj_195_groups_0, pad = obj_195_pad_0, pad_type = obj_195_pad_type_0, strides = obj_195_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_181_cast_fp16)[name = string("obj_195_cast_fp16")]; + tensor inputs_181_cast_fp16 = add(x = inputs_175_cast_fp16, y = obj_195_cast_fp16)[name = string("inputs_181_cast_fp16")]; + tensor inputs_sq_181_cast_fp16 = mul(x = inputs_181_cast_fp16, y = inputs_181_cast_fp16)[name = string("inputs_sq_181_cast_fp16")]; + tensor variance_181_axes_0 = const()[name = string("variance_181_axes_0"), val = tensor([1])]; + bool variance_181_keep_dims_0 = const()[name = string("variance_181_keep_dims_0"), val = bool(true)]; + tensor variance_181_cast_fp16 = reduce_mean(axes = variance_181_axes_0, keep_dims = variance_181_keep_dims_0, x = inputs_sq_181_cast_fp16)[name = string("variance_181_cast_fp16")]; + fp16 var_6682_to_fp16 = const()[name = string("op_6682_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6683_cast_fp16 = add(x = variance_181_cast_fp16, y = var_6682_to_fp16)[name = string("op_6683_cast_fp16")]; + fp32 var_6684_epsilon_0 = const()[name = string("op_6684_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6684_cast_fp16 = rsqrt(epsilon = var_6684_epsilon_0, x = var_6683_cast_fp16)[name = string("op_6684_cast_fp16")]; + tensor hidden_states_223_cast_fp16 = mul(x = inputs_181_cast_fp16, y = var_6684_cast_fp16)[name = string("hidden_states_223_cast_fp16")]; + tensor input_183_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_223_cast_fp16)[name = string("input_183_cast_fp16")]; + string input_185_pad_type_0 = const()[name = string("input_185_pad_type_0"), val = string("valid")]; + tensor input_185_strides_0 = const()[name = string("input_185_strides_0"), val = tensor([1, 1])]; + tensor input_185_pad_0 = const()[name = string("input_185_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_185_dilations_0 = const()[name = string("input_185_dilations_0"), val = tensor([1, 1])]; + int32 input_185_groups_0 = const()[name = string("input_185_groups_0"), val = int32(1)]; + tensor input_185_cast_fp16 = conv(dilations = input_185_dilations_0, groups = input_185_groups_0, pad = input_185_pad_0, pad_type = input_185_pad_type_0, strides = input_185_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_183_cast_fp16)[name = string("input_185_cast_fp16")]; + tensor var_6698_cast_fp16 = silu(x = input_185_cast_fp16)[name = string("op_6698_cast_fp16")]; + string var_6704_pad_type_0 = const()[name = string("op_6704_pad_type_0"), val = string("valid")]; + tensor var_6704_strides_0 = const()[name = string("op_6704_strides_0"), val = tensor([1, 1])]; + tensor var_6704_pad_0 = const()[name = string("op_6704_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6704_dilations_0 = const()[name = string("op_6704_dilations_0"), val = tensor([1, 1])]; + int32 var_6704_groups_0 = const()[name = string("op_6704_groups_0"), val = int32(1)]; + tensor var_6704_cast_fp16 = conv(dilations = var_6704_dilations_0, groups = var_6704_groups_0, pad = var_6704_pad_0, pad_type = var_6704_pad_type_0, strides = var_6704_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_183_cast_fp16)[name = string("op_6704_cast_fp16")]; + tensor input_187_cast_fp16 = mul(x = var_6698_cast_fp16, y = var_6704_cast_fp16)[name = string("input_187_cast_fp16")]; + string hidden_states_225_pad_type_0 = const()[name = string("hidden_states_225_pad_type_0"), val = string("valid")]; + tensor hidden_states_225_strides_0 = const()[name = string("hidden_states_225_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_225_pad_0 = const()[name = string("hidden_states_225_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_225_dilations_0 = const()[name = string("hidden_states_225_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_225_groups_0 = const()[name = string("hidden_states_225_groups_0"), val = int32(1)]; + tensor hidden_states_225_cast_fp16 = conv(dilations = hidden_states_225_dilations_0, groups = hidden_states_225_groups_0, pad = hidden_states_225_pad_0, pad_type = hidden_states_225_pad_type_0, strides = hidden_states_225_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_187_cast_fp16)[name = string("hidden_states_225_cast_fp16")]; + tensor inputs_183_cast_fp16 = add(x = inputs_181_cast_fp16, y = hidden_states_225_cast_fp16)[name = string("inputs_183_cast_fp16")]; + tensor obj_199_begin_0 = const()[name = string("obj_199_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_199_end_0 = const()[name = string("obj_199_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_199_end_mask_0 = const()[name = string("obj_199_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_199_cast_fp16 = slice_by_index(begin = obj_199_begin_0, end = obj_199_end_0, end_mask = obj_199_end_mask_0, x = key_caches_9_cast_fp16)[name = string("obj_199_cast_fp16")]; + tensor obj_201_begin_0 = const()[name = string("obj_201_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_201_end_0 = const()[name = string("obj_201_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_201_end_mask_0 = const()[name = string("obj_201_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_201_cast_fp16 = slice_by_index(begin = obj_201_begin_0, end = obj_201_end_0, end_mask = obj_201_end_mask_0, x = value_caches_9_cast_fp16)[name = string("obj_201_cast_fp16")]; + int32 var_6752 = const()[name = string("op_6752"), val = int32(3)]; + int32 var_6762 = const()[name = string("op_6762"), val = int32(-2)]; + tensor inputs_sq_183_cast_fp16 = mul(x = inputs_183_cast_fp16, y = inputs_183_cast_fp16)[name = string("inputs_sq_183_cast_fp16")]; + tensor variance_183_axes_0 = const()[name = string("variance_183_axes_0"), val = tensor([1])]; + bool variance_183_keep_dims_0 = const()[name = string("variance_183_keep_dims_0"), val = bool(true)]; + tensor variance_183_cast_fp16 = reduce_mean(axes = variance_183_axes_0, keep_dims = variance_183_keep_dims_0, x = inputs_sq_183_cast_fp16)[name = string("variance_183_cast_fp16")]; + fp16 var_6776_to_fp16 = const()[name = string("op_6776_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6777_cast_fp16 = add(x = variance_183_cast_fp16, y = var_6776_to_fp16)[name = string("op_6777_cast_fp16")]; + fp32 var_6778_epsilon_0 = const()[name = string("op_6778_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6778_cast_fp16 = rsqrt(epsilon = var_6778_epsilon_0, x = var_6777_cast_fp16)[name = string("op_6778_cast_fp16")]; + tensor hidden_states_227_cast_fp16 = mul(x = inputs_183_cast_fp16, y = var_6778_cast_fp16)[name = string("hidden_states_227_cast_fp16")]; + tensor obj_197_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_227_cast_fp16)[name = string("obj_197_cast_fp16")]; + string query_133_pad_type_0 = const()[name = string("query_133_pad_type_0"), val = string("valid")]; + tensor query_133_strides_0 = const()[name = string("query_133_strides_0"), val = tensor([1, 1])]; + tensor query_133_pad_0 = const()[name = string("query_133_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_133_dilations_0 = const()[name = string("query_133_dilations_0"), val = tensor([1, 1])]; + int32 query_133_groups_0 = const()[name = string("query_133_groups_0"), val = int32(1)]; + tensor query_133_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_133_dilations_0, groups = query_133_groups_0, pad = query_133_pad_0, pad_type = query_133_pad_type_0, strides = query_133_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_197_cast_fp16)[name = string("query_133_cast_fp16")]; + string current_key_89_pad_type_0 = const()[name = string("current_key_89_pad_type_0"), val = string("valid")]; + tensor current_key_89_strides_0 = const()[name = string("current_key_89_strides_0"), val = tensor([1, 1])]; + tensor current_key_89_pad_0 = const()[name = string("current_key_89_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_89_dilations_0 = const()[name = string("current_key_89_dilations_0"), val = tensor([1, 1])]; + int32 current_key_89_groups_0 = const()[name = string("current_key_89_groups_0"), val = int32(1)]; + tensor current_key_89_cast_fp16 = conv(dilations = current_key_89_dilations_0, groups = current_key_89_groups_0, pad = current_key_89_pad_0, pad_type = current_key_89_pad_type_0, strides = current_key_89_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_197_cast_fp16)[name = string("current_key_89_cast_fp16")]; + string current_value_45_pad_type_0 = const()[name = string("current_value_45_pad_type_0"), val = string("valid")]; + tensor current_value_45_strides_0 = const()[name = string("current_value_45_strides_0"), val = tensor([1, 1])]; + tensor current_value_45_pad_0 = const()[name = string("current_value_45_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_45_dilations_0 = const()[name = string("current_value_45_dilations_0"), val = tensor([1, 1])]; + int32 current_value_45_groups_0 = const()[name = string("current_value_45_groups_0"), val = int32(1)]; + tensor current_value_45_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_45_dilations_0, groups = current_value_45_groups_0, pad = current_value_45_pad_0, pad_type = current_value_45_pad_type_0, strides = current_value_45_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_197_cast_fp16)[name = string("current_value_45_cast_fp16")]; + tensor var_6815 = const()[name = string("op_6815"), val = tensor([16, 128, 1, 1])]; + tensor inputs_185_cast_fp16 = reshape(shape = var_6815, x = query_133_cast_fp16)[name = string("inputs_185_cast_fp16")]; + tensor inputs_sq_185_cast_fp16 = mul(x = inputs_185_cast_fp16, y = inputs_185_cast_fp16)[name = string("inputs_sq_185_cast_fp16")]; + tensor variance_185_axes_0 = const()[name = string("variance_185_axes_0"), val = tensor([1])]; + bool variance_185_keep_dims_0 = const()[name = string("variance_185_keep_dims_0"), val = bool(true)]; + tensor variance_185_cast_fp16 = reduce_mean(axes = variance_185_axes_0, keep_dims = variance_185_keep_dims_0, x = inputs_sq_185_cast_fp16)[name = string("variance_185_cast_fp16")]; + fp16 var_6821_to_fp16 = const()[name = string("op_6821_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6822_cast_fp16 = add(x = variance_185_cast_fp16, y = var_6821_to_fp16)[name = string("op_6822_cast_fp16")]; + fp32 var_6823_epsilon_0 = const()[name = string("op_6823_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6823_cast_fp16 = rsqrt(epsilon = var_6823_epsilon_0, x = var_6822_cast_fp16)[name = string("op_6823_cast_fp16")]; + tensor hidden_states_229_cast_fp16 = mul(x = inputs_185_cast_fp16, y = var_6823_cast_fp16)[name = string("hidden_states_229_cast_fp16")]; + tensor query_normed_45_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_229_cast_fp16)[name = string("query_normed_45_cast_fp16")]; + tensor var_6831 = const()[name = string("op_6831"), val = tensor([8, 128, 1, 1])]; + tensor inputs_187_cast_fp16 = reshape(shape = var_6831, x = current_key_89_cast_fp16)[name = string("inputs_187_cast_fp16")]; + tensor inputs_sq_187_cast_fp16 = mul(x = inputs_187_cast_fp16, y = inputs_187_cast_fp16)[name = string("inputs_sq_187_cast_fp16")]; + tensor variance_187_axes_0 = const()[name = string("variance_187_axes_0"), val = tensor([1])]; + bool variance_187_keep_dims_0 = const()[name = string("variance_187_keep_dims_0"), val = bool(true)]; + tensor variance_187_cast_fp16 = reduce_mean(axes = variance_187_axes_0, keep_dims = variance_187_keep_dims_0, x = inputs_sq_187_cast_fp16)[name = string("variance_187_cast_fp16")]; + fp16 var_6837_to_fp16 = const()[name = string("op_6837_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6838_cast_fp16 = add(x = variance_187_cast_fp16, y = var_6837_to_fp16)[name = string("op_6838_cast_fp16")]; + fp32 var_6839_epsilon_0 = const()[name = string("op_6839_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6839_cast_fp16 = rsqrt(epsilon = var_6839_epsilon_0, x = var_6838_cast_fp16)[name = string("op_6839_cast_fp16")]; + tensor hidden_states_231_cast_fp16 = mul(x = inputs_187_cast_fp16, y = var_6839_cast_fp16)[name = string("hidden_states_231_cast_fp16")]; + tensor current_key_normed_45_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_231_cast_fp16)[name = string("current_key_normed_45_cast_fp16")]; + tensor var_6857 = const()[name = string("op_6857"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_177_cast_fp16 = reshape(shape = var_6857, x = query_normed_45_cast_fp16)[name = string("mh_q_177_cast_fp16")]; + tensor var_6859 = const()[name = string("op_6859"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_177_cast_fp16 = reshape(shape = var_6859, x = current_key_normed_45_cast_fp16)[name = string("mh_k_177_cast_fp16")]; + tensor var_6863_cast_fp16 = mul(x = mh_q_177_cast_fp16, y = cos_41_to_fp16)[name = string("op_6863_cast_fp16")]; + tensor var_6868_begin_0 = const()[name = string("op_6868_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6868_end_0 = const()[name = string("op_6868_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_6868_end_mask_0 = const()[name = string("op_6868_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_6868_cast_fp16 = slice_by_index(begin = var_6868_begin_0, end = var_6868_end_0, end_mask = var_6868_end_mask_0, x = mh_q_177_cast_fp16)[name = string("op_6868_cast_fp16")]; + tensor var_6874_begin_0 = const()[name = string("op_6874_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_6874_end_0 = const()[name = string("op_6874_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_6874_end_mask_0 = const()[name = string("op_6874_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_6874_cast_fp16 = slice_by_index(begin = var_6874_begin_0, end = var_6874_end_0, end_mask = var_6874_end_mask_0, x = mh_q_177_cast_fp16)[name = string("op_6874_cast_fp16")]; + fp16 const_458_promoted_to_fp16 = const()[name = string("const_458_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6876_cast_fp16 = mul(x = var_6874_cast_fp16, y = const_458_promoted_to_fp16)[name = string("op_6876_cast_fp16")]; + bool var_6878_interleave_0 = const()[name = string("op_6878_interleave_0"), val = bool(false)]; + tensor var_6878_cast_fp16 = concat(axis = var_6762, interleave = var_6878_interleave_0, values = (var_6876_cast_fp16, var_6868_cast_fp16))[name = string("op_6878_cast_fp16")]; + tensor var_6879_cast_fp16 = mul(x = var_6878_cast_fp16, y = sin_41_to_fp16)[name = string("op_6879_cast_fp16")]; + tensor mh_q_179_cast_fp16 = add(x = var_6863_cast_fp16, y = var_6879_cast_fp16)[name = string("mh_q_179_cast_fp16")]; + tensor var_6881_cast_fp16 = mul(x = mh_k_177_cast_fp16, y = cos_41_to_fp16)[name = string("op_6881_cast_fp16")]; + tensor var_6886_begin_0 = const()[name = string("op_6886_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6886_end_0 = const()[name = string("op_6886_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_6886_end_mask_0 = const()[name = string("op_6886_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_6886_cast_fp16 = slice_by_index(begin = var_6886_begin_0, end = var_6886_end_0, end_mask = var_6886_end_mask_0, x = mh_k_177_cast_fp16)[name = string("op_6886_cast_fp16")]; + tensor var_6892_begin_0 = const()[name = string("op_6892_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_6892_end_0 = const()[name = string("op_6892_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_6892_end_mask_0 = const()[name = string("op_6892_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_6892_cast_fp16 = slice_by_index(begin = var_6892_begin_0, end = var_6892_end_0, end_mask = var_6892_end_mask_0, x = mh_k_177_cast_fp16)[name = string("op_6892_cast_fp16")]; + fp16 const_461_promoted_to_fp16 = const()[name = string("const_461_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6894_cast_fp16 = mul(x = var_6892_cast_fp16, y = const_461_promoted_to_fp16)[name = string("op_6894_cast_fp16")]; + bool var_6896_interleave_0 = const()[name = string("op_6896_interleave_0"), val = bool(false)]; + tensor var_6896_cast_fp16 = concat(axis = var_6762, interleave = var_6896_interleave_0, values = (var_6894_cast_fp16, var_6886_cast_fp16))[name = string("op_6896_cast_fp16")]; + tensor var_6897_cast_fp16 = mul(x = var_6896_cast_fp16, y = sin_41_to_fp16)[name = string("op_6897_cast_fp16")]; + tensor mh_k_179_cast_fp16 = add(x = var_6881_cast_fp16, y = var_6897_cast_fp16)[name = string("mh_k_179_cast_fp16")]; + tensor var_6901 = const()[name = string("op_6901"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_91_cast_fp16 = reshape(shape = var_6901, x = mh_k_179_cast_fp16)[name = string("current_key_91_cast_fp16")]; + tensor var_6907_to_fp16 = const()[name = string("op_6907_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193280)))]; + tensor var_6908_cast_fp16 = mul(x = obj_199_cast_fp16, y = var_6907_to_fp16)[name = string("op_6908_cast_fp16")]; + tensor var_6905_to_fp16 = const()[name = string("op_6905_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193408)))]; + tensor var_6909_cast_fp16 = mul(x = current_key_91_cast_fp16, y = var_6905_to_fp16)[name = string("op_6909_cast_fp16")]; + tensor key_91_cast_fp16 = add(x = var_6908_cast_fp16, y = var_6909_cast_fp16)[name = string("key_91_cast_fp16")]; + tensor var_6911_to_fp16 = const()[name = string("op_6911_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193280)))]; + tensor var_6912_cast_fp16 = mul(x = obj_201_cast_fp16, y = var_6911_to_fp16)[name = string("op_6912_cast_fp16")]; + tensor var_6913_cast_fp16 = mul(x = current_value_45_cast_fp16, y = var_6905_to_fp16)[name = string("op_6913_cast_fp16")]; + tensor value_45_cast_fp16 = add(x = var_6912_cast_fp16, y = var_6913_cast_fp16)[name = string("value_45_cast_fp16")]; + fp16 var_6920_to_fp16 = const()[name = string("op_6920_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_183_cast_fp16 = mul(x = mh_q_179_cast_fp16, y = var_6920_to_fp16)[name = string("mh_q_183_cast_fp16")]; + tensor var_6922 = const()[name = string("op_6922"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_181_cast_fp16 = reshape(shape = var_6922, x = key_91_cast_fp16)[name = string("mh_k_181_cast_fp16")]; + tensor var_6924 = const()[name = string("op_6924"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_89_cast_fp16 = reshape(shape = var_6924, x = value_45_cast_fp16)[name = string("mh_v_89_cast_fp16")]; + tensor transpose_88_perm_0 = const()[name = string("transpose_88_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_44_reps_0 = const()[name = string("tile_44_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_88_cast_fp16 = transpose(perm = transpose_88_perm_0, x = mh_k_181_cast_fp16)[name = string("transpose_347")]; + tensor tile_44_cast_fp16 = tile(reps = tile_44_reps_0, x = transpose_88_cast_fp16)[name = string("tile_44_cast_fp16")]; + tensor concat_111 = const()[name = string("concat_111"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_88_cast_fp16 = reshape(shape = concat_111, x = tile_44_cast_fp16)[name = string("reshape_88_cast_fp16")]; + tensor transpose_89_perm_0 = const()[name = string("transpose_89_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_112 = const()[name = string("concat_112"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_89_cast_fp16 = transpose(perm = transpose_89_perm_0, x = reshape_88_cast_fp16)[name = string("transpose_346")]; + tensor reshape_89_cast_fp16 = reshape(shape = concat_112, x = transpose_89_cast_fp16)[name = string("reshape_89_cast_fp16")]; + tensor transpose_90_perm_0 = const()[name = string("transpose_90_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_45_reps_0 = const()[name = string("tile_45_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_90_cast_fp16 = transpose(perm = transpose_90_perm_0, x = mh_v_89_cast_fp16)[name = string("transpose_345")]; + tensor tile_45_cast_fp16 = tile(reps = tile_45_reps_0, x = transpose_90_cast_fp16)[name = string("tile_45_cast_fp16")]; + tensor concat_113 = const()[name = string("concat_113"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_90_cast_fp16 = reshape(shape = concat_113, x = tile_45_cast_fp16)[name = string("reshape_90_cast_fp16")]; + tensor transpose_91_perm_0 = const()[name = string("transpose_91_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_114 = const()[name = string("concat_114"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_91_cast_fp16 = transpose(perm = transpose_91_perm_0, x = reshape_90_cast_fp16)[name = string("transpose_344")]; + tensor reshape_91_cast_fp16 = reshape(shape = concat_114, x = transpose_91_cast_fp16)[name = string("reshape_91_cast_fp16")]; + tensor transpose_405_perm_0 = const()[name = string("transpose_405_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_133_transpose_x_1 = const()[name = string("mh_w_133_transpose_x_1"), val = bool(true)]; + bool mh_w_133_transpose_y_1 = const()[name = string("mh_w_133_transpose_y_1"), val = bool(false)]; + tensor transpose_405_cast_fp16 = transpose(perm = transpose_405_perm_0, x = reshape_89_cast_fp16)[name = string("transpose_343")]; + tensor mh_w_133_cast_fp16 = matmul(transpose_x = mh_w_133_transpose_x_1, transpose_y = mh_w_133_transpose_y_1, x = mh_q_183_cast_fp16, y = transpose_405_cast_fp16)[name = string("mh_w_133_cast_fp16")]; + tensor var_6932_to_fp16 = const()[name = string("op_6932_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193536)))]; + tensor mh_w_135_cast_fp16 = add(x = mh_w_133_cast_fp16, y = var_6932_to_fp16)[name = string("mh_w_135_cast_fp16")]; + tensor mh_w_137_cast_fp16 = softmax(axis = var_6752, x = mh_w_135_cast_fp16)[name = string("mh_w_137_cast_fp16")]; + tensor transpose_406_perm_0 = const()[name = string("transpose_406_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_45_transpose_x_1 = const()[name = string("attn_45_transpose_x_1"), val = bool(false)]; + bool attn_45_transpose_y_1 = const()[name = string("attn_45_transpose_y_1"), val = bool(true)]; + tensor transpose_406_cast_fp16 = transpose(perm = transpose_406_perm_0, x = reshape_91_cast_fp16)[name = string("transpose_342")]; + tensor attn_45_cast_fp16 = matmul(transpose_x = attn_45_transpose_x_1, transpose_y = attn_45_transpose_y_1, x = transpose_406_cast_fp16, y = mh_w_137_cast_fp16)[name = string("attn_45_cast_fp16")]; + tensor var_6938 = const()[name = string("op_6938"), val = tensor([1, 2048, 1, 1])]; + tensor input_189_cast_fp16 = reshape(shape = var_6938, x = attn_45_cast_fp16)[name = string("input_189_cast_fp16")]; + string obj_203_pad_type_0 = const()[name = string("obj_203_pad_type_0"), val = string("valid")]; + tensor obj_203_strides_0 = const()[name = string("obj_203_strides_0"), val = tensor([1, 1])]; + tensor obj_203_pad_0 = const()[name = string("obj_203_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_203_dilations_0 = const()[name = string("obj_203_dilations_0"), val = tensor([1, 1])]; + int32 obj_203_groups_0 = const()[name = string("obj_203_groups_0"), val = int32(1)]; + tensor obj_203_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_203_dilations_0, groups = obj_203_groups_0, pad = obj_203_pad_0, pad_type = obj_203_pad_type_0, strides = obj_203_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_189_cast_fp16)[name = string("obj_203_cast_fp16")]; + tensor inputs_189_cast_fp16 = add(x = inputs_183_cast_fp16, y = obj_203_cast_fp16)[name = string("inputs_189_cast_fp16")]; + tensor inputs_sq_189_cast_fp16 = mul(x = inputs_189_cast_fp16, y = inputs_189_cast_fp16)[name = string("inputs_sq_189_cast_fp16")]; + tensor variance_189_axes_0 = const()[name = string("variance_189_axes_0"), val = tensor([1])]; + bool variance_189_keep_dims_0 = const()[name = string("variance_189_keep_dims_0"), val = bool(true)]; + tensor variance_189_cast_fp16 = reduce_mean(axes = variance_189_axes_0, keep_dims = variance_189_keep_dims_0, x = inputs_sq_189_cast_fp16)[name = string("variance_189_cast_fp16")]; + fp16 var_6956_to_fp16 = const()[name = string("op_6956_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_6957_cast_fp16 = add(x = variance_189_cast_fp16, y = var_6956_to_fp16)[name = string("op_6957_cast_fp16")]; + fp32 var_6958_epsilon_0 = const()[name = string("op_6958_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_6958_cast_fp16 = rsqrt(epsilon = var_6958_epsilon_0, x = var_6957_cast_fp16)[name = string("op_6958_cast_fp16")]; + tensor hidden_states_233_cast_fp16 = mul(x = inputs_189_cast_fp16, y = var_6958_cast_fp16)[name = string("hidden_states_233_cast_fp16")]; + tensor input_191_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_233_cast_fp16)[name = string("input_191_cast_fp16")]; + string input_193_pad_type_0 = const()[name = string("input_193_pad_type_0"), val = string("valid")]; + tensor input_193_strides_0 = const()[name = string("input_193_strides_0"), val = tensor([1, 1])]; + tensor input_193_pad_0 = const()[name = string("input_193_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_193_dilations_0 = const()[name = string("input_193_dilations_0"), val = tensor([1, 1])]; + int32 input_193_groups_0 = const()[name = string("input_193_groups_0"), val = int32(1)]; + tensor input_193_cast_fp16 = conv(dilations = input_193_dilations_0, groups = input_193_groups_0, pad = input_193_pad_0, pad_type = input_193_pad_type_0, strides = input_193_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_191_cast_fp16)[name = string("input_193_cast_fp16")]; + tensor var_6972_cast_fp16 = silu(x = input_193_cast_fp16)[name = string("op_6972_cast_fp16")]; + string var_6978_pad_type_0 = const()[name = string("op_6978_pad_type_0"), val = string("valid")]; + tensor var_6978_strides_0 = const()[name = string("op_6978_strides_0"), val = tensor([1, 1])]; + tensor var_6978_pad_0 = const()[name = string("op_6978_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6978_dilations_0 = const()[name = string("op_6978_dilations_0"), val = tensor([1, 1])]; + int32 var_6978_groups_0 = const()[name = string("op_6978_groups_0"), val = int32(1)]; + tensor var_6978_cast_fp16 = conv(dilations = var_6978_dilations_0, groups = var_6978_groups_0, pad = var_6978_pad_0, pad_type = var_6978_pad_type_0, strides = var_6978_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_191_cast_fp16)[name = string("op_6978_cast_fp16")]; + tensor input_195_cast_fp16 = mul(x = var_6972_cast_fp16, y = var_6978_cast_fp16)[name = string("input_195_cast_fp16")]; + string hidden_states_235_pad_type_0 = const()[name = string("hidden_states_235_pad_type_0"), val = string("valid")]; + tensor hidden_states_235_strides_0 = const()[name = string("hidden_states_235_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_235_pad_0 = const()[name = string("hidden_states_235_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_235_dilations_0 = const()[name = string("hidden_states_235_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_235_groups_0 = const()[name = string("hidden_states_235_groups_0"), val = int32(1)]; + tensor hidden_states_235_cast_fp16 = conv(dilations = hidden_states_235_dilations_0, groups = hidden_states_235_groups_0, pad = hidden_states_235_pad_0, pad_type = hidden_states_235_pad_type_0, strides = hidden_states_235_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_195_cast_fp16)[name = string("hidden_states_235_cast_fp16")]; + tensor inputs_191_cast_fp16 = add(x = inputs_189_cast_fp16, y = hidden_states_235_cast_fp16)[name = string("inputs_191_cast_fp16")]; + tensor obj_207_begin_0 = const()[name = string("obj_207_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_207_end_0 = const()[name = string("obj_207_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_207_end_mask_0 = const()[name = string("obj_207_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_207_cast_fp16 = slice_by_index(begin = obj_207_begin_0, end = obj_207_end_0, end_mask = obj_207_end_mask_0, x = key_caches_9_cast_fp16)[name = string("obj_207_cast_fp16")]; + tensor obj_209_begin_0 = const()[name = string("obj_209_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_209_end_0 = const()[name = string("obj_209_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_209_end_mask_0 = const()[name = string("obj_209_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_209_cast_fp16 = slice_by_index(begin = obj_209_begin_0, end = obj_209_end_0, end_mask = obj_209_end_mask_0, x = value_caches_9_cast_fp16)[name = string("obj_209_cast_fp16")]; + int32 var_7026 = const()[name = string("op_7026"), val = int32(3)]; + int32 var_7036 = const()[name = string("op_7036"), val = int32(-2)]; + tensor inputs_sq_191_cast_fp16 = mul(x = inputs_191_cast_fp16, y = inputs_191_cast_fp16)[name = string("inputs_sq_191_cast_fp16")]; + tensor variance_191_axes_0 = const()[name = string("variance_191_axes_0"), val = tensor([1])]; + bool variance_191_keep_dims_0 = const()[name = string("variance_191_keep_dims_0"), val = bool(true)]; + tensor variance_191_cast_fp16 = reduce_mean(axes = variance_191_axes_0, keep_dims = variance_191_keep_dims_0, x = inputs_sq_191_cast_fp16)[name = string("variance_191_cast_fp16")]; + fp16 var_7050_to_fp16 = const()[name = string("op_7050_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7051_cast_fp16 = add(x = variance_191_cast_fp16, y = var_7050_to_fp16)[name = string("op_7051_cast_fp16")]; + fp32 var_7052_epsilon_0 = const()[name = string("op_7052_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7052_cast_fp16 = rsqrt(epsilon = var_7052_epsilon_0, x = var_7051_cast_fp16)[name = string("op_7052_cast_fp16")]; + tensor hidden_states_237_cast_fp16 = mul(x = inputs_191_cast_fp16, y = var_7052_cast_fp16)[name = string("hidden_states_237_cast_fp16")]; + tensor obj_205_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_237_cast_fp16)[name = string("obj_205_cast_fp16")]; + string query_139_pad_type_0 = const()[name = string("query_139_pad_type_0"), val = string("valid")]; + tensor query_139_strides_0 = const()[name = string("query_139_strides_0"), val = tensor([1, 1])]; + tensor query_139_pad_0 = const()[name = string("query_139_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_139_dilations_0 = const()[name = string("query_139_dilations_0"), val = tensor([1, 1])]; + int32 query_139_groups_0 = const()[name = string("query_139_groups_0"), val = int32(1)]; + tensor query_139_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_139_dilations_0, groups = query_139_groups_0, pad = query_139_pad_0, pad_type = query_139_pad_type_0, strides = query_139_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_205_cast_fp16)[name = string("query_139_cast_fp16")]; + string current_key_93_pad_type_0 = const()[name = string("current_key_93_pad_type_0"), val = string("valid")]; + tensor current_key_93_strides_0 = const()[name = string("current_key_93_strides_0"), val = tensor([1, 1])]; + tensor current_key_93_pad_0 = const()[name = string("current_key_93_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_93_dilations_0 = const()[name = string("current_key_93_dilations_0"), val = tensor([1, 1])]; + int32 current_key_93_groups_0 = const()[name = string("current_key_93_groups_0"), val = int32(1)]; + tensor current_key_93_cast_fp16 = conv(dilations = current_key_93_dilations_0, groups = current_key_93_groups_0, pad = current_key_93_pad_0, pad_type = current_key_93_pad_type_0, strides = current_key_93_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_205_cast_fp16)[name = string("current_key_93_cast_fp16")]; + string current_value_47_pad_type_0 = const()[name = string("current_value_47_pad_type_0"), val = string("valid")]; + tensor current_value_47_strides_0 = const()[name = string("current_value_47_strides_0"), val = tensor([1, 1])]; + tensor current_value_47_pad_0 = const()[name = string("current_value_47_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_47_dilations_0 = const()[name = string("current_value_47_dilations_0"), val = tensor([1, 1])]; + int32 current_value_47_groups_0 = const()[name = string("current_value_47_groups_0"), val = int32(1)]; + tensor current_value_47_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_47_dilations_0, groups = current_value_47_groups_0, pad = current_value_47_pad_0, pad_type = current_value_47_pad_type_0, strides = current_value_47_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_205_cast_fp16)[name = string("current_value_47_cast_fp16")]; + tensor var_7089 = const()[name = string("op_7089"), val = tensor([16, 128, 1, 1])]; + tensor inputs_193_cast_fp16 = reshape(shape = var_7089, x = query_139_cast_fp16)[name = string("inputs_193_cast_fp16")]; + tensor inputs_sq_193_cast_fp16 = mul(x = inputs_193_cast_fp16, y = inputs_193_cast_fp16)[name = string("inputs_sq_193_cast_fp16")]; + tensor variance_193_axes_0 = const()[name = string("variance_193_axes_0"), val = tensor([1])]; + bool variance_193_keep_dims_0 = const()[name = string("variance_193_keep_dims_0"), val = bool(true)]; + tensor variance_193_cast_fp16 = reduce_mean(axes = variance_193_axes_0, keep_dims = variance_193_keep_dims_0, x = inputs_sq_193_cast_fp16)[name = string("variance_193_cast_fp16")]; + fp16 var_7095_to_fp16 = const()[name = string("op_7095_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7096_cast_fp16 = add(x = variance_193_cast_fp16, y = var_7095_to_fp16)[name = string("op_7096_cast_fp16")]; + fp32 var_7097_epsilon_0 = const()[name = string("op_7097_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7097_cast_fp16 = rsqrt(epsilon = var_7097_epsilon_0, x = var_7096_cast_fp16)[name = string("op_7097_cast_fp16")]; + tensor hidden_states_239_cast_fp16 = mul(x = inputs_193_cast_fp16, y = var_7097_cast_fp16)[name = string("hidden_states_239_cast_fp16")]; + tensor query_normed_47_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_239_cast_fp16)[name = string("query_normed_47_cast_fp16")]; + tensor var_7105 = const()[name = string("op_7105"), val = tensor([8, 128, 1, 1])]; + tensor inputs_195_cast_fp16 = reshape(shape = var_7105, x = current_key_93_cast_fp16)[name = string("inputs_195_cast_fp16")]; + tensor inputs_sq_195_cast_fp16 = mul(x = inputs_195_cast_fp16, y = inputs_195_cast_fp16)[name = string("inputs_sq_195_cast_fp16")]; + tensor variance_195_axes_0 = const()[name = string("variance_195_axes_0"), val = tensor([1])]; + bool variance_195_keep_dims_0 = const()[name = string("variance_195_keep_dims_0"), val = bool(true)]; + tensor variance_195_cast_fp16 = reduce_mean(axes = variance_195_axes_0, keep_dims = variance_195_keep_dims_0, x = inputs_sq_195_cast_fp16)[name = string("variance_195_cast_fp16")]; + fp16 var_7111_to_fp16 = const()[name = string("op_7111_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7112_cast_fp16 = add(x = variance_195_cast_fp16, y = var_7111_to_fp16)[name = string("op_7112_cast_fp16")]; + fp32 var_7113_epsilon_0 = const()[name = string("op_7113_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7113_cast_fp16 = rsqrt(epsilon = var_7113_epsilon_0, x = var_7112_cast_fp16)[name = string("op_7113_cast_fp16")]; + tensor hidden_states_241_cast_fp16 = mul(x = inputs_195_cast_fp16, y = var_7113_cast_fp16)[name = string("hidden_states_241_cast_fp16")]; + tensor current_key_normed_47_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_241_cast_fp16)[name = string("current_key_normed_47_cast_fp16")]; + tensor var_7131 = const()[name = string("op_7131"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_185_cast_fp16 = reshape(shape = var_7131, x = query_normed_47_cast_fp16)[name = string("mh_q_185_cast_fp16")]; + tensor var_7133 = const()[name = string("op_7133"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_185_cast_fp16 = reshape(shape = var_7133, x = current_key_normed_47_cast_fp16)[name = string("mh_k_185_cast_fp16")]; + tensor var_7137_cast_fp16 = mul(x = mh_q_185_cast_fp16, y = cos_41_to_fp16)[name = string("op_7137_cast_fp16")]; + tensor var_7142_begin_0 = const()[name = string("op_7142_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7142_end_0 = const()[name = string("op_7142_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_7142_end_mask_0 = const()[name = string("op_7142_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_7142_cast_fp16 = slice_by_index(begin = var_7142_begin_0, end = var_7142_end_0, end_mask = var_7142_end_mask_0, x = mh_q_185_cast_fp16)[name = string("op_7142_cast_fp16")]; + tensor var_7148_begin_0 = const()[name = string("op_7148_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_7148_end_0 = const()[name = string("op_7148_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_7148_end_mask_0 = const()[name = string("op_7148_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_7148_cast_fp16 = slice_by_index(begin = var_7148_begin_0, end = var_7148_end_0, end_mask = var_7148_end_mask_0, x = mh_q_185_cast_fp16)[name = string("op_7148_cast_fp16")]; + fp16 const_478_promoted_to_fp16 = const()[name = string("const_478_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7150_cast_fp16 = mul(x = var_7148_cast_fp16, y = const_478_promoted_to_fp16)[name = string("op_7150_cast_fp16")]; + bool var_7152_interleave_0 = const()[name = string("op_7152_interleave_0"), val = bool(false)]; + tensor var_7152_cast_fp16 = concat(axis = var_7036, interleave = var_7152_interleave_0, values = (var_7150_cast_fp16, var_7142_cast_fp16))[name = string("op_7152_cast_fp16")]; + tensor var_7153_cast_fp16 = mul(x = var_7152_cast_fp16, y = sin_41_to_fp16)[name = string("op_7153_cast_fp16")]; + tensor mh_q_187_cast_fp16 = add(x = var_7137_cast_fp16, y = var_7153_cast_fp16)[name = string("mh_q_187_cast_fp16")]; + tensor var_7155_cast_fp16 = mul(x = mh_k_185_cast_fp16, y = cos_41_to_fp16)[name = string("op_7155_cast_fp16")]; + tensor var_7160_begin_0 = const()[name = string("op_7160_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7160_end_0 = const()[name = string("op_7160_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_7160_end_mask_0 = const()[name = string("op_7160_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_7160_cast_fp16 = slice_by_index(begin = var_7160_begin_0, end = var_7160_end_0, end_mask = var_7160_end_mask_0, x = mh_k_185_cast_fp16)[name = string("op_7160_cast_fp16")]; + tensor var_7166_begin_0 = const()[name = string("op_7166_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_7166_end_0 = const()[name = string("op_7166_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_7166_end_mask_0 = const()[name = string("op_7166_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_7166_cast_fp16 = slice_by_index(begin = var_7166_begin_0, end = var_7166_end_0, end_mask = var_7166_end_mask_0, x = mh_k_185_cast_fp16)[name = string("op_7166_cast_fp16")]; + fp16 const_481_promoted_to_fp16 = const()[name = string("const_481_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7168_cast_fp16 = mul(x = var_7166_cast_fp16, y = const_481_promoted_to_fp16)[name = string("op_7168_cast_fp16")]; + bool var_7170_interleave_0 = const()[name = string("op_7170_interleave_0"), val = bool(false)]; + tensor var_7170_cast_fp16 = concat(axis = var_7036, interleave = var_7170_interleave_0, values = (var_7168_cast_fp16, var_7160_cast_fp16))[name = string("op_7170_cast_fp16")]; + tensor var_7171_cast_fp16 = mul(x = var_7170_cast_fp16, y = sin_41_to_fp16)[name = string("op_7171_cast_fp16")]; + tensor mh_k_187_cast_fp16 = add(x = var_7155_cast_fp16, y = var_7171_cast_fp16)[name = string("mh_k_187_cast_fp16")]; + tensor var_7175 = const()[name = string("op_7175"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_95_cast_fp16 = reshape(shape = var_7175, x = mh_k_187_cast_fp16)[name = string("current_key_95_cast_fp16")]; + tensor var_7181_to_fp16 = const()[name = string("op_7181_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193280)))]; + tensor var_7182_cast_fp16 = mul(x = obj_207_cast_fp16, y = var_7181_to_fp16)[name = string("op_7182_cast_fp16")]; + tensor var_7179_to_fp16 = const()[name = string("op_7179_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193408)))]; + tensor var_7183_cast_fp16 = mul(x = current_key_95_cast_fp16, y = var_7179_to_fp16)[name = string("op_7183_cast_fp16")]; + tensor key_95_cast_fp16 = add(x = var_7182_cast_fp16, y = var_7183_cast_fp16)[name = string("key_95_cast_fp16")]; + tensor var_7185_to_fp16 = const()[name = string("op_7185_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193280)))]; + tensor var_7186_cast_fp16 = mul(x = obj_209_cast_fp16, y = var_7185_to_fp16)[name = string("op_7186_cast_fp16")]; + tensor var_7187_cast_fp16 = mul(x = current_value_47_cast_fp16, y = var_7179_to_fp16)[name = string("op_7187_cast_fp16")]; + tensor value_47_cast_fp16 = add(x = var_7186_cast_fp16, y = var_7187_cast_fp16)[name = string("value_47_cast_fp16")]; + fp16 var_7194_to_fp16 = const()[name = string("op_7194_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_191_cast_fp16 = mul(x = mh_q_187_cast_fp16, y = var_7194_to_fp16)[name = string("mh_q_191_cast_fp16")]; + tensor var_7196 = const()[name = string("op_7196"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_189_cast_fp16 = reshape(shape = var_7196, x = key_95_cast_fp16)[name = string("mh_k_189_cast_fp16")]; + tensor var_7198 = const()[name = string("op_7198"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_93_cast_fp16 = reshape(shape = var_7198, x = value_47_cast_fp16)[name = string("mh_v_93_cast_fp16")]; + tensor transpose_92_perm_0 = const()[name = string("transpose_92_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_46_reps_0 = const()[name = string("tile_46_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_92_cast_fp16 = transpose(perm = transpose_92_perm_0, x = mh_k_189_cast_fp16)[name = string("transpose_341")]; + tensor tile_46_cast_fp16 = tile(reps = tile_46_reps_0, x = transpose_92_cast_fp16)[name = string("tile_46_cast_fp16")]; + tensor concat_115 = const()[name = string("concat_115"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_92_cast_fp16 = reshape(shape = concat_115, x = tile_46_cast_fp16)[name = string("reshape_92_cast_fp16")]; + tensor transpose_93_perm_0 = const()[name = string("transpose_93_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_116 = const()[name = string("concat_116"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_93_cast_fp16 = transpose(perm = transpose_93_perm_0, x = reshape_92_cast_fp16)[name = string("transpose_340")]; + tensor reshape_93_cast_fp16 = reshape(shape = concat_116, x = transpose_93_cast_fp16)[name = string("reshape_93_cast_fp16")]; + tensor transpose_94_perm_0 = const()[name = string("transpose_94_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_47_reps_0 = const()[name = string("tile_47_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_94_cast_fp16 = transpose(perm = transpose_94_perm_0, x = mh_v_93_cast_fp16)[name = string("transpose_339")]; + tensor tile_47_cast_fp16 = tile(reps = tile_47_reps_0, x = transpose_94_cast_fp16)[name = string("tile_47_cast_fp16")]; + tensor concat_117 = const()[name = string("concat_117"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_94_cast_fp16 = reshape(shape = concat_117, x = tile_47_cast_fp16)[name = string("reshape_94_cast_fp16")]; + tensor transpose_95_perm_0 = const()[name = string("transpose_95_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_118 = const()[name = string("concat_118"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_95_cast_fp16 = transpose(perm = transpose_95_perm_0, x = reshape_94_cast_fp16)[name = string("transpose_338")]; + tensor reshape_95_cast_fp16 = reshape(shape = concat_118, x = transpose_95_cast_fp16)[name = string("reshape_95_cast_fp16")]; + tensor transpose_409_perm_0 = const()[name = string("transpose_409_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_139_transpose_x_1 = const()[name = string("mh_w_139_transpose_x_1"), val = bool(true)]; + bool mh_w_139_transpose_y_1 = const()[name = string("mh_w_139_transpose_y_1"), val = bool(false)]; + tensor transpose_409_cast_fp16 = transpose(perm = transpose_409_perm_0, x = reshape_93_cast_fp16)[name = string("transpose_337")]; + tensor mh_w_139_cast_fp16 = matmul(transpose_x = mh_w_139_transpose_x_1, transpose_y = mh_w_139_transpose_y_1, x = mh_q_191_cast_fp16, y = transpose_409_cast_fp16)[name = string("mh_w_139_cast_fp16")]; + tensor var_7206_to_fp16 = const()[name = string("op_7206_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193536)))]; + tensor mh_w_141_cast_fp16 = add(x = mh_w_139_cast_fp16, y = var_7206_to_fp16)[name = string("mh_w_141_cast_fp16")]; + tensor mh_w_143_cast_fp16 = softmax(axis = var_7026, x = mh_w_141_cast_fp16)[name = string("mh_w_143_cast_fp16")]; + tensor transpose_410_perm_0 = const()[name = string("transpose_410_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_47_transpose_x_1 = const()[name = string("attn_47_transpose_x_1"), val = bool(false)]; + bool attn_47_transpose_y_1 = const()[name = string("attn_47_transpose_y_1"), val = bool(true)]; + tensor transpose_410_cast_fp16 = transpose(perm = transpose_410_perm_0, x = reshape_95_cast_fp16)[name = string("transpose_336")]; + tensor attn_47_cast_fp16 = matmul(transpose_x = attn_47_transpose_x_1, transpose_y = attn_47_transpose_y_1, x = transpose_410_cast_fp16, y = mh_w_143_cast_fp16)[name = string("attn_47_cast_fp16")]; + tensor var_7212 = const()[name = string("op_7212"), val = tensor([1, 2048, 1, 1])]; + tensor input_197_cast_fp16 = reshape(shape = var_7212, x = attn_47_cast_fp16)[name = string("input_197_cast_fp16")]; + string obj_211_pad_type_0 = const()[name = string("obj_211_pad_type_0"), val = string("valid")]; + tensor obj_211_strides_0 = const()[name = string("obj_211_strides_0"), val = tensor([1, 1])]; + tensor obj_211_pad_0 = const()[name = string("obj_211_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_211_dilations_0 = const()[name = string("obj_211_dilations_0"), val = tensor([1, 1])]; + int32 obj_211_groups_0 = const()[name = string("obj_211_groups_0"), val = int32(1)]; + tensor obj_211_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_211_dilations_0, groups = obj_211_groups_0, pad = obj_211_pad_0, pad_type = obj_211_pad_type_0, strides = obj_211_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_197_cast_fp16)[name = string("obj_211_cast_fp16")]; + tensor inputs_197_cast_fp16 = add(x = inputs_191_cast_fp16, y = obj_211_cast_fp16)[name = string("inputs_197_cast_fp16")]; + tensor inputs_sq_197_cast_fp16 = mul(x = inputs_197_cast_fp16, y = inputs_197_cast_fp16)[name = string("inputs_sq_197_cast_fp16")]; + tensor variance_197_axes_0 = const()[name = string("variance_197_axes_0"), val = tensor([1])]; + bool variance_197_keep_dims_0 = const()[name = string("variance_197_keep_dims_0"), val = bool(true)]; + tensor variance_197_cast_fp16 = reduce_mean(axes = variance_197_axes_0, keep_dims = variance_197_keep_dims_0, x = inputs_sq_197_cast_fp16)[name = string("variance_197_cast_fp16")]; + fp16 var_7230_to_fp16 = const()[name = string("op_7230_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7231_cast_fp16 = add(x = variance_197_cast_fp16, y = var_7230_to_fp16)[name = string("op_7231_cast_fp16")]; + fp32 var_7232_epsilon_0 = const()[name = string("op_7232_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7232_cast_fp16 = rsqrt(epsilon = var_7232_epsilon_0, x = var_7231_cast_fp16)[name = string("op_7232_cast_fp16")]; + tensor hidden_states_243_cast_fp16 = mul(x = inputs_197_cast_fp16, y = var_7232_cast_fp16)[name = string("hidden_states_243_cast_fp16")]; + tensor input_199_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_243_cast_fp16)[name = string("input_199_cast_fp16")]; + string input_201_pad_type_0 = const()[name = string("input_201_pad_type_0"), val = string("valid")]; + tensor input_201_strides_0 = const()[name = string("input_201_strides_0"), val = tensor([1, 1])]; + tensor input_201_pad_0 = const()[name = string("input_201_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_201_dilations_0 = const()[name = string("input_201_dilations_0"), val = tensor([1, 1])]; + int32 input_201_groups_0 = const()[name = string("input_201_groups_0"), val = int32(1)]; + tensor input_201_cast_fp16 = conv(dilations = input_201_dilations_0, groups = input_201_groups_0, pad = input_201_pad_0, pad_type = input_201_pad_type_0, strides = input_201_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_199_cast_fp16)[name = string("input_201_cast_fp16")]; + tensor var_7246_cast_fp16 = silu(x = input_201_cast_fp16)[name = string("op_7246_cast_fp16")]; + string var_7252_pad_type_0 = const()[name = string("op_7252_pad_type_0"), val = string("valid")]; + tensor var_7252_strides_0 = const()[name = string("op_7252_strides_0"), val = tensor([1, 1])]; + tensor var_7252_pad_0 = const()[name = string("op_7252_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7252_dilations_0 = const()[name = string("op_7252_dilations_0"), val = tensor([1, 1])]; + int32 var_7252_groups_0 = const()[name = string("op_7252_groups_0"), val = int32(1)]; + tensor var_7252_cast_fp16 = conv(dilations = var_7252_dilations_0, groups = var_7252_groups_0, pad = var_7252_pad_0, pad_type = var_7252_pad_type_0, strides = var_7252_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_199_cast_fp16)[name = string("op_7252_cast_fp16")]; + tensor input_203_cast_fp16 = mul(x = var_7246_cast_fp16, y = var_7252_cast_fp16)[name = string("input_203_cast_fp16")]; + string hidden_states_245_pad_type_0 = const()[name = string("hidden_states_245_pad_type_0"), val = string("valid")]; + tensor hidden_states_245_strides_0 = const()[name = string("hidden_states_245_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_245_pad_0 = const()[name = string("hidden_states_245_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_245_dilations_0 = const()[name = string("hidden_states_245_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_245_groups_0 = const()[name = string("hidden_states_245_groups_0"), val = int32(1)]; + tensor hidden_states_245_cast_fp16 = conv(dilations = hidden_states_245_dilations_0, groups = hidden_states_245_groups_0, pad = hidden_states_245_pad_0, pad_type = hidden_states_245_pad_type_0, strides = hidden_states_245_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_203_cast_fp16)[name = string("hidden_states_245_cast_fp16")]; + tensor inputs_199_cast_fp16 = add(x = inputs_197_cast_fp16, y = hidden_states_245_cast_fp16)[name = string("inputs_199_cast_fp16")]; + tensor obj_215_begin_0 = const()[name = string("obj_215_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_215_end_0 = const()[name = string("obj_215_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_215_end_mask_0 = const()[name = string("obj_215_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_215_cast_fp16 = slice_by_index(begin = obj_215_begin_0, end = obj_215_end_0, end_mask = obj_215_end_mask_0, x = key_caches_9_cast_fp16)[name = string("obj_215_cast_fp16")]; + tensor obj_217_begin_0 = const()[name = string("obj_217_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_217_end_0 = const()[name = string("obj_217_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_217_end_mask_0 = const()[name = string("obj_217_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_217_cast_fp16 = slice_by_index(begin = obj_217_begin_0, end = obj_217_end_0, end_mask = obj_217_end_mask_0, x = value_caches_9_cast_fp16)[name = string("obj_217_cast_fp16")]; + int32 var_7300 = const()[name = string("op_7300"), val = int32(3)]; + int32 var_7310 = const()[name = string("op_7310"), val = int32(-2)]; + tensor inputs_sq_199_cast_fp16 = mul(x = inputs_199_cast_fp16, y = inputs_199_cast_fp16)[name = string("inputs_sq_199_cast_fp16")]; + tensor variance_199_axes_0 = const()[name = string("variance_199_axes_0"), val = tensor([1])]; + bool variance_199_keep_dims_0 = const()[name = string("variance_199_keep_dims_0"), val = bool(true)]; + tensor variance_199_cast_fp16 = reduce_mean(axes = variance_199_axes_0, keep_dims = variance_199_keep_dims_0, x = inputs_sq_199_cast_fp16)[name = string("variance_199_cast_fp16")]; + fp16 var_7324_to_fp16 = const()[name = string("op_7324_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7325_cast_fp16 = add(x = variance_199_cast_fp16, y = var_7324_to_fp16)[name = string("op_7325_cast_fp16")]; + fp32 var_7326_epsilon_0 = const()[name = string("op_7326_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7326_cast_fp16 = rsqrt(epsilon = var_7326_epsilon_0, x = var_7325_cast_fp16)[name = string("op_7326_cast_fp16")]; + tensor hidden_states_247_cast_fp16 = mul(x = inputs_199_cast_fp16, y = var_7326_cast_fp16)[name = string("hidden_states_247_cast_fp16")]; + tensor obj_213_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_247_cast_fp16)[name = string("obj_213_cast_fp16")]; + string query_145_pad_type_0 = const()[name = string("query_145_pad_type_0"), val = string("valid")]; + tensor query_145_strides_0 = const()[name = string("query_145_strides_0"), val = tensor([1, 1])]; + tensor query_145_pad_0 = const()[name = string("query_145_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_145_dilations_0 = const()[name = string("query_145_dilations_0"), val = tensor([1, 1])]; + int32 query_145_groups_0 = const()[name = string("query_145_groups_0"), val = int32(1)]; + tensor query_145_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_145_dilations_0, groups = query_145_groups_0, pad = query_145_pad_0, pad_type = query_145_pad_type_0, strides = query_145_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_213_cast_fp16)[name = string("query_145_cast_fp16")]; + string current_key_97_pad_type_0 = const()[name = string("current_key_97_pad_type_0"), val = string("valid")]; + tensor current_key_97_strides_0 = const()[name = string("current_key_97_strides_0"), val = tensor([1, 1])]; + tensor current_key_97_pad_0 = const()[name = string("current_key_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_97_dilations_0 = const()[name = string("current_key_97_dilations_0"), val = tensor([1, 1])]; + int32 current_key_97_groups_0 = const()[name = string("current_key_97_groups_0"), val = int32(1)]; + tensor current_key_97_cast_fp16 = conv(dilations = current_key_97_dilations_0, groups = current_key_97_groups_0, pad = current_key_97_pad_0, pad_type = current_key_97_pad_type_0, strides = current_key_97_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_213_cast_fp16)[name = string("current_key_97_cast_fp16")]; + string current_value_49_pad_type_0 = const()[name = string("current_value_49_pad_type_0"), val = string("valid")]; + tensor current_value_49_strides_0 = const()[name = string("current_value_49_strides_0"), val = tensor([1, 1])]; + tensor current_value_49_pad_0 = const()[name = string("current_value_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_49_dilations_0 = const()[name = string("current_value_49_dilations_0"), val = tensor([1, 1])]; + int32 current_value_49_groups_0 = const()[name = string("current_value_49_groups_0"), val = int32(1)]; + tensor current_value_49_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_49_dilations_0, groups = current_value_49_groups_0, pad = current_value_49_pad_0, pad_type = current_value_49_pad_type_0, strides = current_value_49_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_213_cast_fp16)[name = string("current_value_49_cast_fp16")]; + tensor var_7363 = const()[name = string("op_7363"), val = tensor([16, 128, 1, 1])]; + tensor inputs_201_cast_fp16 = reshape(shape = var_7363, x = query_145_cast_fp16)[name = string("inputs_201_cast_fp16")]; + tensor inputs_sq_201_cast_fp16 = mul(x = inputs_201_cast_fp16, y = inputs_201_cast_fp16)[name = string("inputs_sq_201_cast_fp16")]; + tensor variance_201_axes_0 = const()[name = string("variance_201_axes_0"), val = tensor([1])]; + bool variance_201_keep_dims_0 = const()[name = string("variance_201_keep_dims_0"), val = bool(true)]; + tensor variance_201_cast_fp16 = reduce_mean(axes = variance_201_axes_0, keep_dims = variance_201_keep_dims_0, x = inputs_sq_201_cast_fp16)[name = string("variance_201_cast_fp16")]; + fp16 var_7369_to_fp16 = const()[name = string("op_7369_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7370_cast_fp16 = add(x = variance_201_cast_fp16, y = var_7369_to_fp16)[name = string("op_7370_cast_fp16")]; + fp32 var_7371_epsilon_0 = const()[name = string("op_7371_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7371_cast_fp16 = rsqrt(epsilon = var_7371_epsilon_0, x = var_7370_cast_fp16)[name = string("op_7371_cast_fp16")]; + tensor hidden_states_249_cast_fp16 = mul(x = inputs_201_cast_fp16, y = var_7371_cast_fp16)[name = string("hidden_states_249_cast_fp16")]; + tensor query_normed_49_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_249_cast_fp16)[name = string("query_normed_49_cast_fp16")]; + tensor var_7379 = const()[name = string("op_7379"), val = tensor([8, 128, 1, 1])]; + tensor inputs_203_cast_fp16 = reshape(shape = var_7379, x = current_key_97_cast_fp16)[name = string("inputs_203_cast_fp16")]; + tensor inputs_sq_203_cast_fp16 = mul(x = inputs_203_cast_fp16, y = inputs_203_cast_fp16)[name = string("inputs_sq_203_cast_fp16")]; + tensor variance_203_axes_0 = const()[name = string("variance_203_axes_0"), val = tensor([1])]; + bool variance_203_keep_dims_0 = const()[name = string("variance_203_keep_dims_0"), val = bool(true)]; + tensor variance_203_cast_fp16 = reduce_mean(axes = variance_203_axes_0, keep_dims = variance_203_keep_dims_0, x = inputs_sq_203_cast_fp16)[name = string("variance_203_cast_fp16")]; + fp16 var_7385_to_fp16 = const()[name = string("op_7385_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7386_cast_fp16 = add(x = variance_203_cast_fp16, y = var_7385_to_fp16)[name = string("op_7386_cast_fp16")]; + fp32 var_7387_epsilon_0 = const()[name = string("op_7387_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7387_cast_fp16 = rsqrt(epsilon = var_7387_epsilon_0, x = var_7386_cast_fp16)[name = string("op_7387_cast_fp16")]; + tensor hidden_states_251_cast_fp16 = mul(x = inputs_203_cast_fp16, y = var_7387_cast_fp16)[name = string("hidden_states_251_cast_fp16")]; + tensor current_key_normed_49_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_251_cast_fp16)[name = string("current_key_normed_49_cast_fp16")]; + tensor var_7405 = const()[name = string("op_7405"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_193_cast_fp16 = reshape(shape = var_7405, x = query_normed_49_cast_fp16)[name = string("mh_q_193_cast_fp16")]; + tensor var_7407 = const()[name = string("op_7407"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_193_cast_fp16 = reshape(shape = var_7407, x = current_key_normed_49_cast_fp16)[name = string("mh_k_193_cast_fp16")]; + tensor var_7411_cast_fp16 = mul(x = mh_q_193_cast_fp16, y = cos_41_to_fp16)[name = string("op_7411_cast_fp16")]; + tensor var_7416_begin_0 = const()[name = string("op_7416_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7416_end_0 = const()[name = string("op_7416_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_7416_end_mask_0 = const()[name = string("op_7416_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_7416_cast_fp16 = slice_by_index(begin = var_7416_begin_0, end = var_7416_end_0, end_mask = var_7416_end_mask_0, x = mh_q_193_cast_fp16)[name = string("op_7416_cast_fp16")]; + tensor var_7422_begin_0 = const()[name = string("op_7422_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_7422_end_0 = const()[name = string("op_7422_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_7422_end_mask_0 = const()[name = string("op_7422_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_7422_cast_fp16 = slice_by_index(begin = var_7422_begin_0, end = var_7422_end_0, end_mask = var_7422_end_mask_0, x = mh_q_193_cast_fp16)[name = string("op_7422_cast_fp16")]; + fp16 const_498_promoted_to_fp16 = const()[name = string("const_498_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7424_cast_fp16 = mul(x = var_7422_cast_fp16, y = const_498_promoted_to_fp16)[name = string("op_7424_cast_fp16")]; + bool var_7426_interleave_0 = const()[name = string("op_7426_interleave_0"), val = bool(false)]; + tensor var_7426_cast_fp16 = concat(axis = var_7310, interleave = var_7426_interleave_0, values = (var_7424_cast_fp16, var_7416_cast_fp16))[name = string("op_7426_cast_fp16")]; + tensor var_7427_cast_fp16 = mul(x = var_7426_cast_fp16, y = sin_41_to_fp16)[name = string("op_7427_cast_fp16")]; + tensor mh_q_195_cast_fp16 = add(x = var_7411_cast_fp16, y = var_7427_cast_fp16)[name = string("mh_q_195_cast_fp16")]; + tensor var_7429_cast_fp16 = mul(x = mh_k_193_cast_fp16, y = cos_41_to_fp16)[name = string("op_7429_cast_fp16")]; + tensor var_7434_begin_0 = const()[name = string("op_7434_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7434_end_0 = const()[name = string("op_7434_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_7434_end_mask_0 = const()[name = string("op_7434_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_7434_cast_fp16 = slice_by_index(begin = var_7434_begin_0, end = var_7434_end_0, end_mask = var_7434_end_mask_0, x = mh_k_193_cast_fp16)[name = string("op_7434_cast_fp16")]; + tensor var_7440_begin_0 = const()[name = string("op_7440_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_7440_end_0 = const()[name = string("op_7440_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_7440_end_mask_0 = const()[name = string("op_7440_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_7440_cast_fp16 = slice_by_index(begin = var_7440_begin_0, end = var_7440_end_0, end_mask = var_7440_end_mask_0, x = mh_k_193_cast_fp16)[name = string("op_7440_cast_fp16")]; + fp16 const_501_promoted_to_fp16 = const()[name = string("const_501_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7442_cast_fp16 = mul(x = var_7440_cast_fp16, y = const_501_promoted_to_fp16)[name = string("op_7442_cast_fp16")]; + bool var_7444_interleave_0 = const()[name = string("op_7444_interleave_0"), val = bool(false)]; + tensor var_7444_cast_fp16 = concat(axis = var_7310, interleave = var_7444_interleave_0, values = (var_7442_cast_fp16, var_7434_cast_fp16))[name = string("op_7444_cast_fp16")]; + tensor var_7445_cast_fp16 = mul(x = var_7444_cast_fp16, y = sin_41_to_fp16)[name = string("op_7445_cast_fp16")]; + tensor mh_k_195_cast_fp16 = add(x = var_7429_cast_fp16, y = var_7445_cast_fp16)[name = string("mh_k_195_cast_fp16")]; + tensor var_7449 = const()[name = string("op_7449"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_99_cast_fp16 = reshape(shape = var_7449, x = mh_k_195_cast_fp16)[name = string("current_key_99_cast_fp16")]; + tensor var_7455_to_fp16 = const()[name = string("op_7455_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193280)))]; + tensor var_7456_cast_fp16 = mul(x = obj_215_cast_fp16, y = var_7455_to_fp16)[name = string("op_7456_cast_fp16")]; + tensor var_7453_to_fp16 = const()[name = string("op_7453_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193408)))]; + tensor var_7457_cast_fp16 = mul(x = current_key_99_cast_fp16, y = var_7453_to_fp16)[name = string("op_7457_cast_fp16")]; + tensor key_99_cast_fp16 = add(x = var_7456_cast_fp16, y = var_7457_cast_fp16)[name = string("key_99_cast_fp16")]; + tensor var_7459_to_fp16 = const()[name = string("op_7459_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193280)))]; + tensor var_7460_cast_fp16 = mul(x = obj_217_cast_fp16, y = var_7459_to_fp16)[name = string("op_7460_cast_fp16")]; + tensor var_7461_cast_fp16 = mul(x = current_value_49_cast_fp16, y = var_7453_to_fp16)[name = string("op_7461_cast_fp16")]; + tensor value_49_cast_fp16 = add(x = var_7460_cast_fp16, y = var_7461_cast_fp16)[name = string("value_49_cast_fp16")]; + fp16 var_7468_to_fp16 = const()[name = string("op_7468_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_199_cast_fp16 = mul(x = mh_q_195_cast_fp16, y = var_7468_to_fp16)[name = string("mh_q_199_cast_fp16")]; + tensor var_7470 = const()[name = string("op_7470"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_197_cast_fp16 = reshape(shape = var_7470, x = key_99_cast_fp16)[name = string("mh_k_197_cast_fp16")]; + tensor var_7472 = const()[name = string("op_7472"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_97_cast_fp16 = reshape(shape = var_7472, x = value_49_cast_fp16)[name = string("mh_v_97_cast_fp16")]; + tensor transpose_96_perm_0 = const()[name = string("transpose_96_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_48_reps_0 = const()[name = string("tile_48_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_96_cast_fp16 = transpose(perm = transpose_96_perm_0, x = mh_k_197_cast_fp16)[name = string("transpose_335")]; + tensor tile_48_cast_fp16 = tile(reps = tile_48_reps_0, x = transpose_96_cast_fp16)[name = string("tile_48_cast_fp16")]; + tensor concat_119 = const()[name = string("concat_119"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_96_cast_fp16 = reshape(shape = concat_119, x = tile_48_cast_fp16)[name = string("reshape_96_cast_fp16")]; + tensor transpose_97_perm_0 = const()[name = string("transpose_97_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_120 = const()[name = string("concat_120"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_97_cast_fp16 = transpose(perm = transpose_97_perm_0, x = reshape_96_cast_fp16)[name = string("transpose_334")]; + tensor reshape_97_cast_fp16 = reshape(shape = concat_120, x = transpose_97_cast_fp16)[name = string("reshape_97_cast_fp16")]; + tensor transpose_98_perm_0 = const()[name = string("transpose_98_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_49_reps_0 = const()[name = string("tile_49_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_98_cast_fp16 = transpose(perm = transpose_98_perm_0, x = mh_v_97_cast_fp16)[name = string("transpose_333")]; + tensor tile_49_cast_fp16 = tile(reps = tile_49_reps_0, x = transpose_98_cast_fp16)[name = string("tile_49_cast_fp16")]; + tensor concat_121 = const()[name = string("concat_121"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_98_cast_fp16 = reshape(shape = concat_121, x = tile_49_cast_fp16)[name = string("reshape_98_cast_fp16")]; + tensor transpose_99_perm_0 = const()[name = string("transpose_99_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_122 = const()[name = string("concat_122"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_99_cast_fp16 = transpose(perm = transpose_99_perm_0, x = reshape_98_cast_fp16)[name = string("transpose_332")]; + tensor reshape_99_cast_fp16 = reshape(shape = concat_122, x = transpose_99_cast_fp16)[name = string("reshape_99_cast_fp16")]; + tensor transpose_413_perm_0 = const()[name = string("transpose_413_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_145_transpose_x_1 = const()[name = string("mh_w_145_transpose_x_1"), val = bool(true)]; + bool mh_w_145_transpose_y_1 = const()[name = string("mh_w_145_transpose_y_1"), val = bool(false)]; + tensor transpose_413_cast_fp16 = transpose(perm = transpose_413_perm_0, x = reshape_97_cast_fp16)[name = string("transpose_331")]; + tensor mh_w_145_cast_fp16 = matmul(transpose_x = mh_w_145_transpose_x_1, transpose_y = mh_w_145_transpose_y_1, x = mh_q_199_cast_fp16, y = transpose_413_cast_fp16)[name = string("mh_w_145_cast_fp16")]; + tensor var_7480_to_fp16 = const()[name = string("op_7480_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193536)))]; + tensor mh_w_147_cast_fp16 = add(x = mh_w_145_cast_fp16, y = var_7480_to_fp16)[name = string("mh_w_147_cast_fp16")]; + tensor mh_w_149_cast_fp16 = softmax(axis = var_7300, x = mh_w_147_cast_fp16)[name = string("mh_w_149_cast_fp16")]; + tensor transpose_414_perm_0 = const()[name = string("transpose_414_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_49_transpose_x_1 = const()[name = string("attn_49_transpose_x_1"), val = bool(false)]; + bool attn_49_transpose_y_1 = const()[name = string("attn_49_transpose_y_1"), val = bool(true)]; + tensor transpose_414_cast_fp16 = transpose(perm = transpose_414_perm_0, x = reshape_99_cast_fp16)[name = string("transpose_330")]; + tensor attn_49_cast_fp16 = matmul(transpose_x = attn_49_transpose_x_1, transpose_y = attn_49_transpose_y_1, x = transpose_414_cast_fp16, y = mh_w_149_cast_fp16)[name = string("attn_49_cast_fp16")]; + tensor var_7486 = const()[name = string("op_7486"), val = tensor([1, 2048, 1, 1])]; + tensor input_205_cast_fp16 = reshape(shape = var_7486, x = attn_49_cast_fp16)[name = string("input_205_cast_fp16")]; + string obj_219_pad_type_0 = const()[name = string("obj_219_pad_type_0"), val = string("valid")]; + tensor obj_219_strides_0 = const()[name = string("obj_219_strides_0"), val = tensor([1, 1])]; + tensor obj_219_pad_0 = const()[name = string("obj_219_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_219_dilations_0 = const()[name = string("obj_219_dilations_0"), val = tensor([1, 1])]; + int32 obj_219_groups_0 = const()[name = string("obj_219_groups_0"), val = int32(1)]; + tensor obj_219_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_219_dilations_0, groups = obj_219_groups_0, pad = obj_219_pad_0, pad_type = obj_219_pad_type_0, strides = obj_219_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_205_cast_fp16)[name = string("obj_219_cast_fp16")]; + tensor inputs_205_cast_fp16 = add(x = inputs_199_cast_fp16, y = obj_219_cast_fp16)[name = string("inputs_205_cast_fp16")]; + tensor inputs_sq_205_cast_fp16 = mul(x = inputs_205_cast_fp16, y = inputs_205_cast_fp16)[name = string("inputs_sq_205_cast_fp16")]; + tensor variance_205_axes_0 = const()[name = string("variance_205_axes_0"), val = tensor([1])]; + bool variance_205_keep_dims_0 = const()[name = string("variance_205_keep_dims_0"), val = bool(true)]; + tensor variance_205_cast_fp16 = reduce_mean(axes = variance_205_axes_0, keep_dims = variance_205_keep_dims_0, x = inputs_sq_205_cast_fp16)[name = string("variance_205_cast_fp16")]; + fp16 var_7504_to_fp16 = const()[name = string("op_7504_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7505_cast_fp16 = add(x = variance_205_cast_fp16, y = var_7504_to_fp16)[name = string("op_7505_cast_fp16")]; + fp32 var_7506_epsilon_0 = const()[name = string("op_7506_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7506_cast_fp16 = rsqrt(epsilon = var_7506_epsilon_0, x = var_7505_cast_fp16)[name = string("op_7506_cast_fp16")]; + tensor hidden_states_253_cast_fp16 = mul(x = inputs_205_cast_fp16, y = var_7506_cast_fp16)[name = string("hidden_states_253_cast_fp16")]; + tensor input_207_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_253_cast_fp16)[name = string("input_207_cast_fp16")]; + string input_209_pad_type_0 = const()[name = string("input_209_pad_type_0"), val = string("valid")]; + tensor input_209_strides_0 = const()[name = string("input_209_strides_0"), val = tensor([1, 1])]; + tensor input_209_pad_0 = const()[name = string("input_209_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_209_dilations_0 = const()[name = string("input_209_dilations_0"), val = tensor([1, 1])]; + int32 input_209_groups_0 = const()[name = string("input_209_groups_0"), val = int32(1)]; + tensor input_209_cast_fp16 = conv(dilations = input_209_dilations_0, groups = input_209_groups_0, pad = input_209_pad_0, pad_type = input_209_pad_type_0, strides = input_209_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_207_cast_fp16)[name = string("input_209_cast_fp16")]; + tensor var_7520_cast_fp16 = silu(x = input_209_cast_fp16)[name = string("op_7520_cast_fp16")]; + string var_7526_pad_type_0 = const()[name = string("op_7526_pad_type_0"), val = string("valid")]; + tensor var_7526_strides_0 = const()[name = string("op_7526_strides_0"), val = tensor([1, 1])]; + tensor var_7526_pad_0 = const()[name = string("op_7526_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7526_dilations_0 = const()[name = string("op_7526_dilations_0"), val = tensor([1, 1])]; + int32 var_7526_groups_0 = const()[name = string("op_7526_groups_0"), val = int32(1)]; + tensor var_7526_cast_fp16 = conv(dilations = var_7526_dilations_0, groups = var_7526_groups_0, pad = var_7526_pad_0, pad_type = var_7526_pad_type_0, strides = var_7526_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_207_cast_fp16)[name = string("op_7526_cast_fp16")]; + tensor input_211_cast_fp16 = mul(x = var_7520_cast_fp16, y = var_7526_cast_fp16)[name = string("input_211_cast_fp16")]; + string hidden_states_255_pad_type_0 = const()[name = string("hidden_states_255_pad_type_0"), val = string("valid")]; + tensor hidden_states_255_strides_0 = const()[name = string("hidden_states_255_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_255_pad_0 = const()[name = string("hidden_states_255_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_255_dilations_0 = const()[name = string("hidden_states_255_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_255_groups_0 = const()[name = string("hidden_states_255_groups_0"), val = int32(1)]; + tensor hidden_states_255_cast_fp16 = conv(dilations = hidden_states_255_dilations_0, groups = hidden_states_255_groups_0, pad = hidden_states_255_pad_0, pad_type = hidden_states_255_pad_type_0, strides = hidden_states_255_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_211_cast_fp16)[name = string("hidden_states_255_cast_fp16")]; + tensor inputs_207_cast_fp16 = add(x = inputs_205_cast_fp16, y = hidden_states_255_cast_fp16)[name = string("inputs_207_cast_fp16")]; + int32 var_7554 = const()[name = string("op_7554"), val = int32(1)]; + bool key_caches_11_interleave_0 = const()[name = string("key_caches_11_interleave_0"), val = bool(false)]; + tensor key_caches_11_cast_fp16 = concat(axis = var_7554, interleave = key_caches_11_interleave_0, values = (key_83_cast_fp16, key_87_cast_fp16, key_91_cast_fp16, key_95_cast_fp16, key_99_cast_fp16))[name = string("key_caches_11_cast_fp16")]; + int32 var_7557 = const()[name = string("op_7557"), val = int32(1)]; + bool value_caches_11_interleave_0 = const()[name = string("value_caches_11_interleave_0"), val = bool(false)]; + tensor value_caches_11_cast_fp16 = concat(axis = var_7557, interleave = value_caches_11_interleave_0, values = (value_41_cast_fp16, value_43_cast_fp16, value_45_cast_fp16, value_47_cast_fp16, value_49_cast_fp16))[name = string("value_caches_11_cast_fp16")]; + tensor inputs_sq_207_cast_fp16 = mul(x = inputs_207_cast_fp16, y = inputs_207_cast_fp16)[name = string("inputs_sq_207_cast_fp16")]; + tensor variance_207_axes_0 = const()[name = string("variance_207_axes_0"), val = tensor([1])]; + bool variance_207_keep_dims_0 = const()[name = string("variance_207_keep_dims_0"), val = bool(true)]; + tensor variance_207_cast_fp16 = reduce_mean(axes = variance_207_axes_0, keep_dims = variance_207_keep_dims_0, x = inputs_sq_207_cast_fp16)[name = string("variance_207_cast_fp16")]; + fp16 var_7567_to_fp16 = const()[name = string("op_7567_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7568_cast_fp16 = add(x = variance_207_cast_fp16, y = var_7567_to_fp16)[name = string("op_7568_cast_fp16")]; + fp32 var_7569_epsilon_0 = const()[name = string("op_7569_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7569_cast_fp16 = rsqrt(epsilon = var_7569_epsilon_0, x = var_7568_cast_fp16)[name = string("op_7569_cast_fp16")]; + tensor hidden_states_257_cast_fp16 = mul(x = inputs_207_cast_fp16, y = var_7569_cast_fp16)[name = string("hidden_states_257_cast_fp16")]; + tensor input_213_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_257_cast_fp16)[name = string("input_213_cast_fp16")]; + string logits_13_pad_type_0 = const()[name = string("logits_13_pad_type_0"), val = string("valid")]; + tensor logits_13_strides_0 = const()[name = string("logits_13_strides_0"), val = tensor([1, 1])]; + tensor logits_13_pad_0 = const()[name = string("logits_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_13_dilations_0 = const()[name = string("logits_13_dilations_0"), val = tensor([1, 1])]; + int32 logits_13_groups_0 = const()[name = string("logits_13_groups_0"), val = int32(1)]; + tensor lm_heads_3_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87099968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89197184))))[name = string("lm_heads_3_weight_to_fp16_palettized")]; + tensor logits_13_cast_fp16 = conv(dilations = logits_13_dilations_0, groups = logits_13_groups_0, pad = logits_13_pad_0, pad_type = logits_13_pad_type_0, strides = logits_13_strides_0, weight = lm_heads_3_weight_to_fp16_palettized, x = input_213_cast_fp16)[name = string("logits_13_cast_fp16")]; + tensor var_7587 = const()[name = string("op_7587"), val = tensor([1, 2048])]; + tensor logits_15_cast_fp16 = reshape(shape = var_7587, x = logits_13_cast_fp16)[name = string("logits_15_cast_fp16")]; + tensor scaled_logits_7_cast_fp16 = real_div(x = logits_15_cast_fp16, y = temperature)[name = string("scaled_logits_7_cast_fp16")]; + int32 var_7597 = const()[name = string("op_7597"), val = int32(100)]; + int32 top_values_7_axis_0 = const()[name = string("top_values_7_axis_0"), val = int32(1)]; + bool top_values_7_ascending_0 = const()[name = string("top_values_7_ascending_0"), val = bool(false)]; + bool top_values_7_sort_0 = const()[name = string("top_values_7_sort_0"), val = bool(true)]; + bool top_values_7_return_indices_0 = const()[name = string("top_values_7_return_indices_0"), val = bool(true)]; + string top_values_7_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_7_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_7_cast_fp16_cast_uint16_0, tensor top_values_7_cast_fp16_cast_uint16_1 = topk(ascending = top_values_7_ascending_0, axis = top_values_7_axis_0, k = var_7597, output_indices_dtype = top_values_7_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_7_return_indices_0, sort = top_values_7_sort_0, x = scaled_logits_7_cast_fp16)[name = string("top_values_7_cast_fp16_cast_uint16")]; + tensor var_7603_cast_fp16 = mul(x = top_values_7_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_7603_cast_fp16")]; + tensor var_7607_cast_fp16 = add(x = var_7603_cast_fp16, y = var_3001_cast_fp16)[name = string("op_7607_cast_fp16")]; + tensor reduce_min_3_axes_0 = const()[name = string("reduce_min_3_axes_0"), val = tensor([1])]; + bool reduce_min_3_keep_dims_0 = const()[name = string("reduce_min_3_keep_dims_0"), val = bool(true)]; + tensor reduce_min_3_cast_fp16 = reduce_min(axes = reduce_min_3_axes_0, keep_dims = reduce_min_3_keep_dims_0, x = var_7607_cast_fp16)[name = string("reduce_min_3_cast_fp16")]; + tensor var_7610_cast_fp16 = greater_equal(x = scaled_logits_7_cast_fp16, y = reduce_min_3_cast_fp16)[name = string("op_7610_cast_fp16")]; + fp16 var_7611_value_0_to_fp16 = const()[name = string("op_7611_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_7611_cast_fp16 = fill_like(ref_tensor = scaled_logits_7_cast_fp16, value = var_7611_value_0_to_fp16)[name = string("op_7611_cast_fp16")]; + tensor masked_logits_7_cast_fp16 = select(a = scaled_logits_7_cast_fp16, b = var_7611_cast_fp16, cond = var_7610_cast_fp16)[name = string("masked_logits_7_cast_fp16")]; + tensor var_7615_begin_0 = const()[name = string("op_7615_begin_0"), val = tensor([3, 0])]; + tensor var_7615_end_0 = const()[name = string("op_7615_end_0"), val = tensor([4, 2048])]; + tensor var_7615_end_mask_0 = const()[name = string("op_7615_end_mask_0"), val = tensor([false, true])]; + tensor var_7615_squeeze_mask_0 = const()[name = string("op_7615_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_7615_cast_fp16 = slice_by_index(begin = var_7615_begin_0, end = var_7615_end_0, end_mask = var_7615_end_mask_0, squeeze_mask = var_7615_squeeze_mask_0, x = gumbel)[name = string("op_7615_cast_fp16")]; + tensor var_7618 = const()[name = string("op_7618"), val = tensor([1, 2048])]; + tensor var_7619_cast_fp16 = reshape(shape = var_7618, x = var_7615_cast_fp16)[name = string("op_7619_cast_fp16")]; + tensor noisy_logits_7_cast_fp16 = add(x = masked_logits_7_cast_fp16, y = var_7619_cast_fp16)[name = string("noisy_logits_7_cast_fp16")]; + int32 code_7_axis_0 = const()[name = string("code_7_axis_0"), val = int32(1)]; + bool code_7_keep_dims_0 = const()[name = string("code_7_keep_dims_0"), val = bool(false)]; + string code_7_output_dtype_0 = const()[name = string("code_7_output_dtype_0"), val = string("int32")]; + tensor code_7_cast_fp16 = reduce_argmax(axis = code_7_axis_0, keep_dims = code_7_keep_dims_0, output_dtype = code_7_output_dtype_0, x = noisy_logits_7_cast_fp16)[name = string("code_7_cast_fp16")]; + int32 var_7630 = const()[name = string("op_7630"), val = int32(6144)]; + tensor input_215 = add(x = code_7_cast_fp16, y = var_7630)[name = string("input_215")]; + int32 code_embed_13_axis_0 = const()[name = string("code_embed_13_axis_0"), val = int32(0)]; + int32 code_embed_13_batch_dims_0 = const()[name = string("code_embed_13_batch_dims_0"), val = int32(0)]; + bool code_embed_13_validate_indices_0 = const()[name = string("code_embed_13_validate_indices_0"), val = bool(false)]; + string input_215_to_uint16_dtype_0 = const()[name = string("input_215_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_215_to_uint16 = cast(dtype = input_215_to_uint16_dtype_0, x = input_215)[name = string("cast_11")]; + tensor code_embed_13_cast_fp16_cast_uint16 = gather(axis = code_embed_13_axis_0, batch_dims = code_embed_13_batch_dims_0, indices = input_215_to_uint16, validate_indices = code_embed_13_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_13_cast_fp16_cast_uint16")]; + tensor var_7634 = const()[name = string("op_7634"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_15_cast_fp16 = reshape(shape = var_7634, x = code_embed_13_cast_fp16_cast_uint16)[name = string("code_embed_15_cast_fp16")]; + tensor embed_sum_9_cast_fp16 = add(x = embed_sum_7_cast_fp16, y = code_embed_15_cast_fp16)[name = string("embed_sum_9_cast_fp16")]; + string inputs_209_pad_type_0 = const()[name = string("inputs_209_pad_type_0"), val = string("valid")]; + tensor inputs_209_strides_0 = const()[name = string("inputs_209_strides_0"), val = tensor([1, 1])]; + tensor inputs_209_pad_0 = const()[name = string("inputs_209_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_209_dilations_0 = const()[name = string("inputs_209_dilations_0"), val = tensor([1, 1])]; + int32 inputs_209_groups_0 = const()[name = string("inputs_209_groups_0"), val = int32(1)]; + tensor inputs_209_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_209_dilations_0, groups = inputs_209_groups_0, pad = inputs_209_pad_0, pad_type = inputs_209_pad_type_0, strides = inputs_209_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_15_cast_fp16)[name = string("inputs_209_cast_fp16")]; + tensor obj_223_begin_0 = const()[name = string("obj_223_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_223_end_0 = const()[name = string("obj_223_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_223_end_mask_0 = const()[name = string("obj_223_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_223_cast_fp16 = slice_by_index(begin = obj_223_begin_0, end = obj_223_end_0, end_mask = obj_223_end_mask_0, x = key_caches_11_cast_fp16)[name = string("obj_223_cast_fp16")]; + tensor obj_225_begin_0 = const()[name = string("obj_225_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_225_end_0 = const()[name = string("obj_225_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_225_end_mask_0 = const()[name = string("obj_225_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_225_cast_fp16 = slice_by_index(begin = obj_225_begin_0, end = obj_225_end_0, end_mask = obj_225_end_mask_0, x = value_caches_11_cast_fp16)[name = string("obj_225_cast_fp16")]; + int32 var_7739 = const()[name = string("op_7739"), val = int32(3)]; + int32 var_7749 = const()[name = string("op_7749"), val = int32(-2)]; + tensor inputs_sq_209_cast_fp16 = mul(x = inputs_209_cast_fp16, y = inputs_209_cast_fp16)[name = string("inputs_sq_209_cast_fp16")]; + tensor variance_209_axes_0 = const()[name = string("variance_209_axes_0"), val = tensor([1])]; + bool variance_209_keep_dims_0 = const()[name = string("variance_209_keep_dims_0"), val = bool(true)]; + tensor variance_209_cast_fp16 = reduce_mean(axes = variance_209_axes_0, keep_dims = variance_209_keep_dims_0, x = inputs_sq_209_cast_fp16)[name = string("variance_209_cast_fp16")]; + fp16 var_7763_to_fp16 = const()[name = string("op_7763_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7764_cast_fp16 = add(x = variance_209_cast_fp16, y = var_7763_to_fp16)[name = string("op_7764_cast_fp16")]; + fp32 var_7765_epsilon_0 = const()[name = string("op_7765_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7765_cast_fp16 = rsqrt(epsilon = var_7765_epsilon_0, x = var_7764_cast_fp16)[name = string("op_7765_cast_fp16")]; + tensor hidden_states_259_cast_fp16 = mul(x = inputs_209_cast_fp16, y = var_7765_cast_fp16)[name = string("hidden_states_259_cast_fp16")]; + tensor obj_221_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_259_cast_fp16)[name = string("obj_221_cast_fp16")]; + string query_151_pad_type_0 = const()[name = string("query_151_pad_type_0"), val = string("valid")]; + tensor query_151_strides_0 = const()[name = string("query_151_strides_0"), val = tensor([1, 1])]; + tensor query_151_pad_0 = const()[name = string("query_151_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_151_dilations_0 = const()[name = string("query_151_dilations_0"), val = tensor([1, 1])]; + int32 query_151_groups_0 = const()[name = string("query_151_groups_0"), val = int32(1)]; + tensor query_151_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_151_dilations_0, groups = query_151_groups_0, pad = query_151_pad_0, pad_type = query_151_pad_type_0, strides = query_151_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_221_cast_fp16)[name = string("query_151_cast_fp16")]; + string current_key_101_pad_type_0 = const()[name = string("current_key_101_pad_type_0"), val = string("valid")]; + tensor current_key_101_strides_0 = const()[name = string("current_key_101_strides_0"), val = tensor([1, 1])]; + tensor current_key_101_pad_0 = const()[name = string("current_key_101_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_101_dilations_0 = const()[name = string("current_key_101_dilations_0"), val = tensor([1, 1])]; + int32 current_key_101_groups_0 = const()[name = string("current_key_101_groups_0"), val = int32(1)]; + tensor current_key_101_cast_fp16 = conv(dilations = current_key_101_dilations_0, groups = current_key_101_groups_0, pad = current_key_101_pad_0, pad_type = current_key_101_pad_type_0, strides = current_key_101_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_221_cast_fp16)[name = string("current_key_101_cast_fp16")]; + string current_value_51_pad_type_0 = const()[name = string("current_value_51_pad_type_0"), val = string("valid")]; + tensor current_value_51_strides_0 = const()[name = string("current_value_51_strides_0"), val = tensor([1, 1])]; + tensor current_value_51_pad_0 = const()[name = string("current_value_51_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_51_dilations_0 = const()[name = string("current_value_51_dilations_0"), val = tensor([1, 1])]; + int32 current_value_51_groups_0 = const()[name = string("current_value_51_groups_0"), val = int32(1)]; + tensor current_value_51_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_51_dilations_0, groups = current_value_51_groups_0, pad = current_value_51_pad_0, pad_type = current_value_51_pad_type_0, strides = current_value_51_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_221_cast_fp16)[name = string("current_value_51_cast_fp16")]; + tensor var_7802 = const()[name = string("op_7802"), val = tensor([16, 128, 1, 1])]; + tensor inputs_211_cast_fp16 = reshape(shape = var_7802, x = query_151_cast_fp16)[name = string("inputs_211_cast_fp16")]; + tensor inputs_sq_211_cast_fp16 = mul(x = inputs_211_cast_fp16, y = inputs_211_cast_fp16)[name = string("inputs_sq_211_cast_fp16")]; + tensor variance_211_axes_0 = const()[name = string("variance_211_axes_0"), val = tensor([1])]; + bool variance_211_keep_dims_0 = const()[name = string("variance_211_keep_dims_0"), val = bool(true)]; + tensor variance_211_cast_fp16 = reduce_mean(axes = variance_211_axes_0, keep_dims = variance_211_keep_dims_0, x = inputs_sq_211_cast_fp16)[name = string("variance_211_cast_fp16")]; + fp16 var_7808_to_fp16 = const()[name = string("op_7808_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7809_cast_fp16 = add(x = variance_211_cast_fp16, y = var_7808_to_fp16)[name = string("op_7809_cast_fp16")]; + fp32 var_7810_epsilon_0 = const()[name = string("op_7810_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7810_cast_fp16 = rsqrt(epsilon = var_7810_epsilon_0, x = var_7809_cast_fp16)[name = string("op_7810_cast_fp16")]; + tensor hidden_states_261_cast_fp16 = mul(x = inputs_211_cast_fp16, y = var_7810_cast_fp16)[name = string("hidden_states_261_cast_fp16")]; + tensor query_normed_51_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_261_cast_fp16)[name = string("query_normed_51_cast_fp16")]; + tensor var_7818 = const()[name = string("op_7818"), val = tensor([8, 128, 1, 1])]; + tensor inputs_213_cast_fp16 = reshape(shape = var_7818, x = current_key_101_cast_fp16)[name = string("inputs_213_cast_fp16")]; + tensor inputs_sq_213_cast_fp16 = mul(x = inputs_213_cast_fp16, y = inputs_213_cast_fp16)[name = string("inputs_sq_213_cast_fp16")]; + tensor variance_213_axes_0 = const()[name = string("variance_213_axes_0"), val = tensor([1])]; + bool variance_213_keep_dims_0 = const()[name = string("variance_213_keep_dims_0"), val = bool(true)]; + tensor variance_213_cast_fp16 = reduce_mean(axes = variance_213_axes_0, keep_dims = variance_213_keep_dims_0, x = inputs_sq_213_cast_fp16)[name = string("variance_213_cast_fp16")]; + fp16 var_7824_to_fp16 = const()[name = string("op_7824_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7825_cast_fp16 = add(x = variance_213_cast_fp16, y = var_7824_to_fp16)[name = string("op_7825_cast_fp16")]; + fp32 var_7826_epsilon_0 = const()[name = string("op_7826_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7826_cast_fp16 = rsqrt(epsilon = var_7826_epsilon_0, x = var_7825_cast_fp16)[name = string("op_7826_cast_fp16")]; + tensor hidden_states_263_cast_fp16 = mul(x = inputs_213_cast_fp16, y = var_7826_cast_fp16)[name = string("hidden_states_263_cast_fp16")]; + tensor current_key_normed_51_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_263_cast_fp16)[name = string("current_key_normed_51_cast_fp16")]; + tensor var_7844 = const()[name = string("op_7844"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_201_cast_fp16 = reshape(shape = var_7844, x = query_normed_51_cast_fp16)[name = string("mh_q_201_cast_fp16")]; + tensor var_7846 = const()[name = string("op_7846"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_201_cast_fp16 = reshape(shape = var_7846, x = current_key_normed_51_cast_fp16)[name = string("mh_k_201_cast_fp16")]; + tensor cos_51_to_fp16 = const()[name = string("cos_51_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193664)))]; + tensor var_7850_cast_fp16 = mul(x = mh_q_201_cast_fp16, y = cos_51_to_fp16)[name = string("op_7850_cast_fp16")]; + tensor var_7855_begin_0 = const()[name = string("op_7855_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7855_end_0 = const()[name = string("op_7855_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_7855_end_mask_0 = const()[name = string("op_7855_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_7855_cast_fp16 = slice_by_index(begin = var_7855_begin_0, end = var_7855_end_0, end_mask = var_7855_end_mask_0, x = mh_q_201_cast_fp16)[name = string("op_7855_cast_fp16")]; + tensor var_7861_begin_0 = const()[name = string("op_7861_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_7861_end_0 = const()[name = string("op_7861_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_7861_end_mask_0 = const()[name = string("op_7861_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_7861_cast_fp16 = slice_by_index(begin = var_7861_begin_0, end = var_7861_end_0, end_mask = var_7861_end_mask_0, x = mh_q_201_cast_fp16)[name = string("op_7861_cast_fp16")]; + fp16 const_519_promoted_to_fp16 = const()[name = string("const_519_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7863_cast_fp16 = mul(x = var_7861_cast_fp16, y = const_519_promoted_to_fp16)[name = string("op_7863_cast_fp16")]; + bool var_7865_interleave_0 = const()[name = string("op_7865_interleave_0"), val = bool(false)]; + tensor var_7865_cast_fp16 = concat(axis = var_7749, interleave = var_7865_interleave_0, values = (var_7863_cast_fp16, var_7855_cast_fp16))[name = string("op_7865_cast_fp16")]; + tensor sin_51_to_fp16 = const()[name = string("sin_51_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175193984)))]; + tensor var_7866_cast_fp16 = mul(x = var_7865_cast_fp16, y = sin_51_to_fp16)[name = string("op_7866_cast_fp16")]; + tensor mh_q_203_cast_fp16 = add(x = var_7850_cast_fp16, y = var_7866_cast_fp16)[name = string("mh_q_203_cast_fp16")]; + tensor var_7868_cast_fp16 = mul(x = mh_k_201_cast_fp16, y = cos_51_to_fp16)[name = string("op_7868_cast_fp16")]; + tensor var_7873_begin_0 = const()[name = string("op_7873_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7873_end_0 = const()[name = string("op_7873_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_7873_end_mask_0 = const()[name = string("op_7873_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_7873_cast_fp16 = slice_by_index(begin = var_7873_begin_0, end = var_7873_end_0, end_mask = var_7873_end_mask_0, x = mh_k_201_cast_fp16)[name = string("op_7873_cast_fp16")]; + tensor var_7879_begin_0 = const()[name = string("op_7879_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_7879_end_0 = const()[name = string("op_7879_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_7879_end_mask_0 = const()[name = string("op_7879_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_7879_cast_fp16 = slice_by_index(begin = var_7879_begin_0, end = var_7879_end_0, end_mask = var_7879_end_mask_0, x = mh_k_201_cast_fp16)[name = string("op_7879_cast_fp16")]; + fp16 const_522_promoted_to_fp16 = const()[name = string("const_522_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7881_cast_fp16 = mul(x = var_7879_cast_fp16, y = const_522_promoted_to_fp16)[name = string("op_7881_cast_fp16")]; + bool var_7883_interleave_0 = const()[name = string("op_7883_interleave_0"), val = bool(false)]; + tensor var_7883_cast_fp16 = concat(axis = var_7749, interleave = var_7883_interleave_0, values = (var_7881_cast_fp16, var_7873_cast_fp16))[name = string("op_7883_cast_fp16")]; + tensor var_7884_cast_fp16 = mul(x = var_7883_cast_fp16, y = sin_51_to_fp16)[name = string("op_7884_cast_fp16")]; + tensor mh_k_203_cast_fp16 = add(x = var_7868_cast_fp16, y = var_7884_cast_fp16)[name = string("mh_k_203_cast_fp16")]; + tensor var_7888 = const()[name = string("op_7888"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_103_cast_fp16 = reshape(shape = var_7888, x = mh_k_203_cast_fp16)[name = string("current_key_103_cast_fp16")]; + tensor var_7894_to_fp16 = const()[name = string("op_7894_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194304)))]; + tensor var_7895_cast_fp16 = mul(x = obj_223_cast_fp16, y = var_7894_to_fp16)[name = string("op_7895_cast_fp16")]; + tensor var_7892_to_fp16 = const()[name = string("op_7892_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194432)))]; + tensor var_7896_cast_fp16 = mul(x = current_key_103_cast_fp16, y = var_7892_to_fp16)[name = string("op_7896_cast_fp16")]; + tensor key_103_cast_fp16 = add(x = var_7895_cast_fp16, y = var_7896_cast_fp16)[name = string("key_103_cast_fp16")]; + tensor var_7898_to_fp16 = const()[name = string("op_7898_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194304)))]; + tensor var_7899_cast_fp16 = mul(x = obj_225_cast_fp16, y = var_7898_to_fp16)[name = string("op_7899_cast_fp16")]; + tensor var_7900_cast_fp16 = mul(x = current_value_51_cast_fp16, y = var_7892_to_fp16)[name = string("op_7900_cast_fp16")]; + tensor value_51_cast_fp16 = add(x = var_7899_cast_fp16, y = var_7900_cast_fp16)[name = string("value_51_cast_fp16")]; + fp16 var_7907_to_fp16 = const()[name = string("op_7907_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_207_cast_fp16 = mul(x = mh_q_203_cast_fp16, y = var_7907_to_fp16)[name = string("mh_q_207_cast_fp16")]; + tensor var_7909 = const()[name = string("op_7909"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_205_cast_fp16 = reshape(shape = var_7909, x = key_103_cast_fp16)[name = string("mh_k_205_cast_fp16")]; + tensor var_7911 = const()[name = string("op_7911"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_101_cast_fp16 = reshape(shape = var_7911, x = value_51_cast_fp16)[name = string("mh_v_101_cast_fp16")]; + tensor transpose_100_perm_0 = const()[name = string("transpose_100_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_50_reps_0 = const()[name = string("tile_50_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_100_cast_fp16 = transpose(perm = transpose_100_perm_0, x = mh_k_205_cast_fp16)[name = string("transpose_329")]; + tensor tile_50_cast_fp16 = tile(reps = tile_50_reps_0, x = transpose_100_cast_fp16)[name = string("tile_50_cast_fp16")]; + tensor concat_128 = const()[name = string("concat_128"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_100_cast_fp16 = reshape(shape = concat_128, x = tile_50_cast_fp16)[name = string("reshape_100_cast_fp16")]; + tensor transpose_101_perm_0 = const()[name = string("transpose_101_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_129 = const()[name = string("concat_129"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_101_cast_fp16 = transpose(perm = transpose_101_perm_0, x = reshape_100_cast_fp16)[name = string("transpose_328")]; + tensor reshape_101_cast_fp16 = reshape(shape = concat_129, x = transpose_101_cast_fp16)[name = string("reshape_101_cast_fp16")]; + tensor transpose_102_perm_0 = const()[name = string("transpose_102_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_51_reps_0 = const()[name = string("tile_51_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_102_cast_fp16 = transpose(perm = transpose_102_perm_0, x = mh_v_101_cast_fp16)[name = string("transpose_327")]; + tensor tile_51_cast_fp16 = tile(reps = tile_51_reps_0, x = transpose_102_cast_fp16)[name = string("tile_51_cast_fp16")]; + tensor concat_130 = const()[name = string("concat_130"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_102_cast_fp16 = reshape(shape = concat_130, x = tile_51_cast_fp16)[name = string("reshape_102_cast_fp16")]; + tensor transpose_103_perm_0 = const()[name = string("transpose_103_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_131 = const()[name = string("concat_131"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_103_cast_fp16 = transpose(perm = transpose_103_perm_0, x = reshape_102_cast_fp16)[name = string("transpose_326")]; + tensor reshape_103_cast_fp16 = reshape(shape = concat_131, x = transpose_103_cast_fp16)[name = string("reshape_103_cast_fp16")]; + tensor transpose_417_perm_0 = const()[name = string("transpose_417_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_151_transpose_x_1 = const()[name = string("mh_w_151_transpose_x_1"), val = bool(true)]; + bool mh_w_151_transpose_y_1 = const()[name = string("mh_w_151_transpose_y_1"), val = bool(false)]; + tensor transpose_417_cast_fp16 = transpose(perm = transpose_417_perm_0, x = reshape_101_cast_fp16)[name = string("transpose_325")]; + tensor mh_w_151_cast_fp16 = matmul(transpose_x = mh_w_151_transpose_x_1, transpose_y = mh_w_151_transpose_y_1, x = mh_q_207_cast_fp16, y = transpose_417_cast_fp16)[name = string("mh_w_151_cast_fp16")]; + tensor var_7919_to_fp16 = const()[name = string("op_7919_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194560)))]; + tensor mh_w_153_cast_fp16 = add(x = mh_w_151_cast_fp16, y = var_7919_to_fp16)[name = string("mh_w_153_cast_fp16")]; + tensor mh_w_155_cast_fp16 = softmax(axis = var_7739, x = mh_w_153_cast_fp16)[name = string("mh_w_155_cast_fp16")]; + tensor transpose_418_perm_0 = const()[name = string("transpose_418_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_51_transpose_x_1 = const()[name = string("attn_51_transpose_x_1"), val = bool(false)]; + bool attn_51_transpose_y_1 = const()[name = string("attn_51_transpose_y_1"), val = bool(true)]; + tensor transpose_418_cast_fp16 = transpose(perm = transpose_418_perm_0, x = reshape_103_cast_fp16)[name = string("transpose_324")]; + tensor attn_51_cast_fp16 = matmul(transpose_x = attn_51_transpose_x_1, transpose_y = attn_51_transpose_y_1, x = transpose_418_cast_fp16, y = mh_w_155_cast_fp16)[name = string("attn_51_cast_fp16")]; + tensor var_7925 = const()[name = string("op_7925"), val = tensor([1, 2048, 1, 1])]; + tensor input_217_cast_fp16 = reshape(shape = var_7925, x = attn_51_cast_fp16)[name = string("input_217_cast_fp16")]; + string obj_231_pad_type_0 = const()[name = string("obj_231_pad_type_0"), val = string("valid")]; + tensor obj_231_strides_0 = const()[name = string("obj_231_strides_0"), val = tensor([1, 1])]; + tensor obj_231_pad_0 = const()[name = string("obj_231_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_231_dilations_0 = const()[name = string("obj_231_dilations_0"), val = tensor([1, 1])]; + int32 obj_231_groups_0 = const()[name = string("obj_231_groups_0"), val = int32(1)]; + tensor obj_231_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_231_dilations_0, groups = obj_231_groups_0, pad = obj_231_pad_0, pad_type = obj_231_pad_type_0, strides = obj_231_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_217_cast_fp16)[name = string("obj_231_cast_fp16")]; + tensor inputs_215_cast_fp16 = add(x = inputs_209_cast_fp16, y = obj_231_cast_fp16)[name = string("inputs_215_cast_fp16")]; + tensor inputs_sq_215_cast_fp16 = mul(x = inputs_215_cast_fp16, y = inputs_215_cast_fp16)[name = string("inputs_sq_215_cast_fp16")]; + tensor variance_215_axes_0 = const()[name = string("variance_215_axes_0"), val = tensor([1])]; + bool variance_215_keep_dims_0 = const()[name = string("variance_215_keep_dims_0"), val = bool(true)]; + tensor variance_215_cast_fp16 = reduce_mean(axes = variance_215_axes_0, keep_dims = variance_215_keep_dims_0, x = inputs_sq_215_cast_fp16)[name = string("variance_215_cast_fp16")]; + fp16 var_7943_to_fp16 = const()[name = string("op_7943_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_7944_cast_fp16 = add(x = variance_215_cast_fp16, y = var_7943_to_fp16)[name = string("op_7944_cast_fp16")]; + fp32 var_7945_epsilon_0 = const()[name = string("op_7945_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_7945_cast_fp16 = rsqrt(epsilon = var_7945_epsilon_0, x = var_7944_cast_fp16)[name = string("op_7945_cast_fp16")]; + tensor hidden_states_265_cast_fp16 = mul(x = inputs_215_cast_fp16, y = var_7945_cast_fp16)[name = string("hidden_states_265_cast_fp16")]; + tensor input_219_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_265_cast_fp16)[name = string("input_219_cast_fp16")]; + string input_221_pad_type_0 = const()[name = string("input_221_pad_type_0"), val = string("valid")]; + tensor input_221_strides_0 = const()[name = string("input_221_strides_0"), val = tensor([1, 1])]; + tensor input_221_pad_0 = const()[name = string("input_221_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_221_dilations_0 = const()[name = string("input_221_dilations_0"), val = tensor([1, 1])]; + int32 input_221_groups_0 = const()[name = string("input_221_groups_0"), val = int32(1)]; + tensor input_221_cast_fp16 = conv(dilations = input_221_dilations_0, groups = input_221_groups_0, pad = input_221_pad_0, pad_type = input_221_pad_type_0, strides = input_221_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_219_cast_fp16)[name = string("input_221_cast_fp16")]; + tensor var_7959_cast_fp16 = silu(x = input_221_cast_fp16)[name = string("op_7959_cast_fp16")]; + string var_7965_pad_type_0 = const()[name = string("op_7965_pad_type_0"), val = string("valid")]; + tensor var_7965_strides_0 = const()[name = string("op_7965_strides_0"), val = tensor([1, 1])]; + tensor var_7965_pad_0 = const()[name = string("op_7965_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7965_dilations_0 = const()[name = string("op_7965_dilations_0"), val = tensor([1, 1])]; + int32 var_7965_groups_0 = const()[name = string("op_7965_groups_0"), val = int32(1)]; + tensor var_7965_cast_fp16 = conv(dilations = var_7965_dilations_0, groups = var_7965_groups_0, pad = var_7965_pad_0, pad_type = var_7965_pad_type_0, strides = var_7965_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_219_cast_fp16)[name = string("op_7965_cast_fp16")]; + tensor input_223_cast_fp16 = mul(x = var_7959_cast_fp16, y = var_7965_cast_fp16)[name = string("input_223_cast_fp16")]; + string hidden_states_267_pad_type_0 = const()[name = string("hidden_states_267_pad_type_0"), val = string("valid")]; + tensor hidden_states_267_strides_0 = const()[name = string("hidden_states_267_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_267_pad_0 = const()[name = string("hidden_states_267_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_267_dilations_0 = const()[name = string("hidden_states_267_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_267_groups_0 = const()[name = string("hidden_states_267_groups_0"), val = int32(1)]; + tensor hidden_states_267_cast_fp16 = conv(dilations = hidden_states_267_dilations_0, groups = hidden_states_267_groups_0, pad = hidden_states_267_pad_0, pad_type = hidden_states_267_pad_type_0, strides = hidden_states_267_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_223_cast_fp16)[name = string("hidden_states_267_cast_fp16")]; + tensor inputs_217_cast_fp16 = add(x = inputs_215_cast_fp16, y = hidden_states_267_cast_fp16)[name = string("inputs_217_cast_fp16")]; + tensor obj_235_begin_0 = const()[name = string("obj_235_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_235_end_0 = const()[name = string("obj_235_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_235_end_mask_0 = const()[name = string("obj_235_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_235_cast_fp16 = slice_by_index(begin = obj_235_begin_0, end = obj_235_end_0, end_mask = obj_235_end_mask_0, x = key_caches_11_cast_fp16)[name = string("obj_235_cast_fp16")]; + tensor obj_237_begin_0 = const()[name = string("obj_237_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_237_end_0 = const()[name = string("obj_237_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_237_end_mask_0 = const()[name = string("obj_237_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_237_cast_fp16 = slice_by_index(begin = obj_237_begin_0, end = obj_237_end_0, end_mask = obj_237_end_mask_0, x = value_caches_11_cast_fp16)[name = string("obj_237_cast_fp16")]; + int32 var_8013 = const()[name = string("op_8013"), val = int32(3)]; + int32 var_8023 = const()[name = string("op_8023"), val = int32(-2)]; + tensor inputs_sq_217_cast_fp16 = mul(x = inputs_217_cast_fp16, y = inputs_217_cast_fp16)[name = string("inputs_sq_217_cast_fp16")]; + tensor variance_217_axes_0 = const()[name = string("variance_217_axes_0"), val = tensor([1])]; + bool variance_217_keep_dims_0 = const()[name = string("variance_217_keep_dims_0"), val = bool(true)]; + tensor variance_217_cast_fp16 = reduce_mean(axes = variance_217_axes_0, keep_dims = variance_217_keep_dims_0, x = inputs_sq_217_cast_fp16)[name = string("variance_217_cast_fp16")]; + fp16 var_8037_to_fp16 = const()[name = string("op_8037_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8038_cast_fp16 = add(x = variance_217_cast_fp16, y = var_8037_to_fp16)[name = string("op_8038_cast_fp16")]; + fp32 var_8039_epsilon_0 = const()[name = string("op_8039_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8039_cast_fp16 = rsqrt(epsilon = var_8039_epsilon_0, x = var_8038_cast_fp16)[name = string("op_8039_cast_fp16")]; + tensor hidden_states_269_cast_fp16 = mul(x = inputs_217_cast_fp16, y = var_8039_cast_fp16)[name = string("hidden_states_269_cast_fp16")]; + tensor obj_233_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_269_cast_fp16)[name = string("obj_233_cast_fp16")]; + string query_157_pad_type_0 = const()[name = string("query_157_pad_type_0"), val = string("valid")]; + tensor query_157_strides_0 = const()[name = string("query_157_strides_0"), val = tensor([1, 1])]; + tensor query_157_pad_0 = const()[name = string("query_157_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_157_dilations_0 = const()[name = string("query_157_dilations_0"), val = tensor([1, 1])]; + int32 query_157_groups_0 = const()[name = string("query_157_groups_0"), val = int32(1)]; + tensor query_157_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_157_dilations_0, groups = query_157_groups_0, pad = query_157_pad_0, pad_type = query_157_pad_type_0, strides = query_157_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_233_cast_fp16)[name = string("query_157_cast_fp16")]; + string current_key_105_pad_type_0 = const()[name = string("current_key_105_pad_type_0"), val = string("valid")]; + tensor current_key_105_strides_0 = const()[name = string("current_key_105_strides_0"), val = tensor([1, 1])]; + tensor current_key_105_pad_0 = const()[name = string("current_key_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_105_dilations_0 = const()[name = string("current_key_105_dilations_0"), val = tensor([1, 1])]; + int32 current_key_105_groups_0 = const()[name = string("current_key_105_groups_0"), val = int32(1)]; + tensor current_key_105_cast_fp16 = conv(dilations = current_key_105_dilations_0, groups = current_key_105_groups_0, pad = current_key_105_pad_0, pad_type = current_key_105_pad_type_0, strides = current_key_105_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_233_cast_fp16)[name = string("current_key_105_cast_fp16")]; + string current_value_53_pad_type_0 = const()[name = string("current_value_53_pad_type_0"), val = string("valid")]; + tensor current_value_53_strides_0 = const()[name = string("current_value_53_strides_0"), val = tensor([1, 1])]; + tensor current_value_53_pad_0 = const()[name = string("current_value_53_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_53_dilations_0 = const()[name = string("current_value_53_dilations_0"), val = tensor([1, 1])]; + int32 current_value_53_groups_0 = const()[name = string("current_value_53_groups_0"), val = int32(1)]; + tensor current_value_53_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_53_dilations_0, groups = current_value_53_groups_0, pad = current_value_53_pad_0, pad_type = current_value_53_pad_type_0, strides = current_value_53_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_233_cast_fp16)[name = string("current_value_53_cast_fp16")]; + tensor var_8076 = const()[name = string("op_8076"), val = tensor([16, 128, 1, 1])]; + tensor inputs_219_cast_fp16 = reshape(shape = var_8076, x = query_157_cast_fp16)[name = string("inputs_219_cast_fp16")]; + tensor inputs_sq_219_cast_fp16 = mul(x = inputs_219_cast_fp16, y = inputs_219_cast_fp16)[name = string("inputs_sq_219_cast_fp16")]; + tensor variance_219_axes_0 = const()[name = string("variance_219_axes_0"), val = tensor([1])]; + bool variance_219_keep_dims_0 = const()[name = string("variance_219_keep_dims_0"), val = bool(true)]; + tensor variance_219_cast_fp16 = reduce_mean(axes = variance_219_axes_0, keep_dims = variance_219_keep_dims_0, x = inputs_sq_219_cast_fp16)[name = string("variance_219_cast_fp16")]; + fp16 var_8082_to_fp16 = const()[name = string("op_8082_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8083_cast_fp16 = add(x = variance_219_cast_fp16, y = var_8082_to_fp16)[name = string("op_8083_cast_fp16")]; + fp32 var_8084_epsilon_0 = const()[name = string("op_8084_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8084_cast_fp16 = rsqrt(epsilon = var_8084_epsilon_0, x = var_8083_cast_fp16)[name = string("op_8084_cast_fp16")]; + tensor hidden_states_271_cast_fp16 = mul(x = inputs_219_cast_fp16, y = var_8084_cast_fp16)[name = string("hidden_states_271_cast_fp16")]; + tensor query_normed_53_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_271_cast_fp16)[name = string("query_normed_53_cast_fp16")]; + tensor var_8092 = const()[name = string("op_8092"), val = tensor([8, 128, 1, 1])]; + tensor inputs_221_cast_fp16 = reshape(shape = var_8092, x = current_key_105_cast_fp16)[name = string("inputs_221_cast_fp16")]; + tensor inputs_sq_221_cast_fp16 = mul(x = inputs_221_cast_fp16, y = inputs_221_cast_fp16)[name = string("inputs_sq_221_cast_fp16")]; + tensor variance_221_axes_0 = const()[name = string("variance_221_axes_0"), val = tensor([1])]; + bool variance_221_keep_dims_0 = const()[name = string("variance_221_keep_dims_0"), val = bool(true)]; + tensor variance_221_cast_fp16 = reduce_mean(axes = variance_221_axes_0, keep_dims = variance_221_keep_dims_0, x = inputs_sq_221_cast_fp16)[name = string("variance_221_cast_fp16")]; + fp16 var_8098_to_fp16 = const()[name = string("op_8098_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8099_cast_fp16 = add(x = variance_221_cast_fp16, y = var_8098_to_fp16)[name = string("op_8099_cast_fp16")]; + fp32 var_8100_epsilon_0 = const()[name = string("op_8100_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8100_cast_fp16 = rsqrt(epsilon = var_8100_epsilon_0, x = var_8099_cast_fp16)[name = string("op_8100_cast_fp16")]; + tensor hidden_states_273_cast_fp16 = mul(x = inputs_221_cast_fp16, y = var_8100_cast_fp16)[name = string("hidden_states_273_cast_fp16")]; + tensor current_key_normed_53_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_273_cast_fp16)[name = string("current_key_normed_53_cast_fp16")]; + tensor var_8118 = const()[name = string("op_8118"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_209_cast_fp16 = reshape(shape = var_8118, x = query_normed_53_cast_fp16)[name = string("mh_q_209_cast_fp16")]; + tensor var_8120 = const()[name = string("op_8120"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_209_cast_fp16 = reshape(shape = var_8120, x = current_key_normed_53_cast_fp16)[name = string("mh_k_209_cast_fp16")]; + tensor var_8124_cast_fp16 = mul(x = mh_q_209_cast_fp16, y = cos_51_to_fp16)[name = string("op_8124_cast_fp16")]; + tensor var_8129_begin_0 = const()[name = string("op_8129_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8129_end_0 = const()[name = string("op_8129_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_8129_end_mask_0 = const()[name = string("op_8129_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_8129_cast_fp16 = slice_by_index(begin = var_8129_begin_0, end = var_8129_end_0, end_mask = var_8129_end_mask_0, x = mh_q_209_cast_fp16)[name = string("op_8129_cast_fp16")]; + tensor var_8135_begin_0 = const()[name = string("op_8135_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_8135_end_0 = const()[name = string("op_8135_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_8135_end_mask_0 = const()[name = string("op_8135_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_8135_cast_fp16 = slice_by_index(begin = var_8135_begin_0, end = var_8135_end_0, end_mask = var_8135_end_mask_0, x = mh_q_209_cast_fp16)[name = string("op_8135_cast_fp16")]; + fp16 const_539_promoted_to_fp16 = const()[name = string("const_539_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8137_cast_fp16 = mul(x = var_8135_cast_fp16, y = const_539_promoted_to_fp16)[name = string("op_8137_cast_fp16")]; + bool var_8139_interleave_0 = const()[name = string("op_8139_interleave_0"), val = bool(false)]; + tensor var_8139_cast_fp16 = concat(axis = var_8023, interleave = var_8139_interleave_0, values = (var_8137_cast_fp16, var_8129_cast_fp16))[name = string("op_8139_cast_fp16")]; + tensor var_8140_cast_fp16 = mul(x = var_8139_cast_fp16, y = sin_51_to_fp16)[name = string("op_8140_cast_fp16")]; + tensor mh_q_211_cast_fp16 = add(x = var_8124_cast_fp16, y = var_8140_cast_fp16)[name = string("mh_q_211_cast_fp16")]; + tensor var_8142_cast_fp16 = mul(x = mh_k_209_cast_fp16, y = cos_51_to_fp16)[name = string("op_8142_cast_fp16")]; + tensor var_8147_begin_0 = const()[name = string("op_8147_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8147_end_0 = const()[name = string("op_8147_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_8147_end_mask_0 = const()[name = string("op_8147_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_8147_cast_fp16 = slice_by_index(begin = var_8147_begin_0, end = var_8147_end_0, end_mask = var_8147_end_mask_0, x = mh_k_209_cast_fp16)[name = string("op_8147_cast_fp16")]; + tensor var_8153_begin_0 = const()[name = string("op_8153_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_8153_end_0 = const()[name = string("op_8153_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_8153_end_mask_0 = const()[name = string("op_8153_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_8153_cast_fp16 = slice_by_index(begin = var_8153_begin_0, end = var_8153_end_0, end_mask = var_8153_end_mask_0, x = mh_k_209_cast_fp16)[name = string("op_8153_cast_fp16")]; + fp16 const_542_promoted_to_fp16 = const()[name = string("const_542_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8155_cast_fp16 = mul(x = var_8153_cast_fp16, y = const_542_promoted_to_fp16)[name = string("op_8155_cast_fp16")]; + bool var_8157_interleave_0 = const()[name = string("op_8157_interleave_0"), val = bool(false)]; + tensor var_8157_cast_fp16 = concat(axis = var_8023, interleave = var_8157_interleave_0, values = (var_8155_cast_fp16, var_8147_cast_fp16))[name = string("op_8157_cast_fp16")]; + tensor var_8158_cast_fp16 = mul(x = var_8157_cast_fp16, y = sin_51_to_fp16)[name = string("op_8158_cast_fp16")]; + tensor mh_k_211_cast_fp16 = add(x = var_8142_cast_fp16, y = var_8158_cast_fp16)[name = string("mh_k_211_cast_fp16")]; + tensor var_8162 = const()[name = string("op_8162"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_107_cast_fp16 = reshape(shape = var_8162, x = mh_k_211_cast_fp16)[name = string("current_key_107_cast_fp16")]; + tensor var_8168_to_fp16 = const()[name = string("op_8168_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194304)))]; + tensor var_8169_cast_fp16 = mul(x = obj_235_cast_fp16, y = var_8168_to_fp16)[name = string("op_8169_cast_fp16")]; + tensor var_8166_to_fp16 = const()[name = string("op_8166_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194432)))]; + tensor var_8170_cast_fp16 = mul(x = current_key_107_cast_fp16, y = var_8166_to_fp16)[name = string("op_8170_cast_fp16")]; + tensor key_107_cast_fp16 = add(x = var_8169_cast_fp16, y = var_8170_cast_fp16)[name = string("key_107_cast_fp16")]; + tensor var_8172_to_fp16 = const()[name = string("op_8172_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194304)))]; + tensor var_8173_cast_fp16 = mul(x = obj_237_cast_fp16, y = var_8172_to_fp16)[name = string("op_8173_cast_fp16")]; + tensor var_8174_cast_fp16 = mul(x = current_value_53_cast_fp16, y = var_8166_to_fp16)[name = string("op_8174_cast_fp16")]; + tensor value_53_cast_fp16 = add(x = var_8173_cast_fp16, y = var_8174_cast_fp16)[name = string("value_53_cast_fp16")]; + fp16 var_8181_to_fp16 = const()[name = string("op_8181_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_215_cast_fp16 = mul(x = mh_q_211_cast_fp16, y = var_8181_to_fp16)[name = string("mh_q_215_cast_fp16")]; + tensor var_8183 = const()[name = string("op_8183"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_213_cast_fp16 = reshape(shape = var_8183, x = key_107_cast_fp16)[name = string("mh_k_213_cast_fp16")]; + tensor var_8185 = const()[name = string("op_8185"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_105_cast_fp16 = reshape(shape = var_8185, x = value_53_cast_fp16)[name = string("mh_v_105_cast_fp16")]; + tensor transpose_104_perm_0 = const()[name = string("transpose_104_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_52_reps_0 = const()[name = string("tile_52_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_104_cast_fp16 = transpose(perm = transpose_104_perm_0, x = mh_k_213_cast_fp16)[name = string("transpose_323")]; + tensor tile_52_cast_fp16 = tile(reps = tile_52_reps_0, x = transpose_104_cast_fp16)[name = string("tile_52_cast_fp16")]; + tensor concat_132 = const()[name = string("concat_132"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_104_cast_fp16 = reshape(shape = concat_132, x = tile_52_cast_fp16)[name = string("reshape_104_cast_fp16")]; + tensor transpose_105_perm_0 = const()[name = string("transpose_105_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_133 = const()[name = string("concat_133"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_105_cast_fp16 = transpose(perm = transpose_105_perm_0, x = reshape_104_cast_fp16)[name = string("transpose_322")]; + tensor reshape_105_cast_fp16 = reshape(shape = concat_133, x = transpose_105_cast_fp16)[name = string("reshape_105_cast_fp16")]; + tensor transpose_106_perm_0 = const()[name = string("transpose_106_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_53_reps_0 = const()[name = string("tile_53_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_106_cast_fp16 = transpose(perm = transpose_106_perm_0, x = mh_v_105_cast_fp16)[name = string("transpose_321")]; + tensor tile_53_cast_fp16 = tile(reps = tile_53_reps_0, x = transpose_106_cast_fp16)[name = string("tile_53_cast_fp16")]; + tensor concat_134 = const()[name = string("concat_134"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_106_cast_fp16 = reshape(shape = concat_134, x = tile_53_cast_fp16)[name = string("reshape_106_cast_fp16")]; + tensor transpose_107_perm_0 = const()[name = string("transpose_107_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_135 = const()[name = string("concat_135"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_107_cast_fp16 = transpose(perm = transpose_107_perm_0, x = reshape_106_cast_fp16)[name = string("transpose_320")]; + tensor reshape_107_cast_fp16 = reshape(shape = concat_135, x = transpose_107_cast_fp16)[name = string("reshape_107_cast_fp16")]; + tensor transpose_421_perm_0 = const()[name = string("transpose_421_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_157_transpose_x_1 = const()[name = string("mh_w_157_transpose_x_1"), val = bool(true)]; + bool mh_w_157_transpose_y_1 = const()[name = string("mh_w_157_transpose_y_1"), val = bool(false)]; + tensor transpose_421_cast_fp16 = transpose(perm = transpose_421_perm_0, x = reshape_105_cast_fp16)[name = string("transpose_319")]; + tensor mh_w_157_cast_fp16 = matmul(transpose_x = mh_w_157_transpose_x_1, transpose_y = mh_w_157_transpose_y_1, x = mh_q_215_cast_fp16, y = transpose_421_cast_fp16)[name = string("mh_w_157_cast_fp16")]; + tensor var_8193_to_fp16 = const()[name = string("op_8193_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194560)))]; + tensor mh_w_159_cast_fp16 = add(x = mh_w_157_cast_fp16, y = var_8193_to_fp16)[name = string("mh_w_159_cast_fp16")]; + tensor mh_w_161_cast_fp16 = softmax(axis = var_8013, x = mh_w_159_cast_fp16)[name = string("mh_w_161_cast_fp16")]; + tensor transpose_422_perm_0 = const()[name = string("transpose_422_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_53_transpose_x_1 = const()[name = string("attn_53_transpose_x_1"), val = bool(false)]; + bool attn_53_transpose_y_1 = const()[name = string("attn_53_transpose_y_1"), val = bool(true)]; + tensor transpose_422_cast_fp16 = transpose(perm = transpose_422_perm_0, x = reshape_107_cast_fp16)[name = string("transpose_318")]; + tensor attn_53_cast_fp16 = matmul(transpose_x = attn_53_transpose_x_1, transpose_y = attn_53_transpose_y_1, x = transpose_422_cast_fp16, y = mh_w_161_cast_fp16)[name = string("attn_53_cast_fp16")]; + tensor var_8199 = const()[name = string("op_8199"), val = tensor([1, 2048, 1, 1])]; + tensor input_225_cast_fp16 = reshape(shape = var_8199, x = attn_53_cast_fp16)[name = string("input_225_cast_fp16")]; + string obj_239_pad_type_0 = const()[name = string("obj_239_pad_type_0"), val = string("valid")]; + tensor obj_239_strides_0 = const()[name = string("obj_239_strides_0"), val = tensor([1, 1])]; + tensor obj_239_pad_0 = const()[name = string("obj_239_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_239_dilations_0 = const()[name = string("obj_239_dilations_0"), val = tensor([1, 1])]; + int32 obj_239_groups_0 = const()[name = string("obj_239_groups_0"), val = int32(1)]; + tensor obj_239_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_239_dilations_0, groups = obj_239_groups_0, pad = obj_239_pad_0, pad_type = obj_239_pad_type_0, strides = obj_239_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_225_cast_fp16)[name = string("obj_239_cast_fp16")]; + tensor inputs_223_cast_fp16 = add(x = inputs_217_cast_fp16, y = obj_239_cast_fp16)[name = string("inputs_223_cast_fp16")]; + tensor inputs_sq_223_cast_fp16 = mul(x = inputs_223_cast_fp16, y = inputs_223_cast_fp16)[name = string("inputs_sq_223_cast_fp16")]; + tensor variance_223_axes_0 = const()[name = string("variance_223_axes_0"), val = tensor([1])]; + bool variance_223_keep_dims_0 = const()[name = string("variance_223_keep_dims_0"), val = bool(true)]; + tensor variance_223_cast_fp16 = reduce_mean(axes = variance_223_axes_0, keep_dims = variance_223_keep_dims_0, x = inputs_sq_223_cast_fp16)[name = string("variance_223_cast_fp16")]; + fp16 var_8217_to_fp16 = const()[name = string("op_8217_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8218_cast_fp16 = add(x = variance_223_cast_fp16, y = var_8217_to_fp16)[name = string("op_8218_cast_fp16")]; + fp32 var_8219_epsilon_0 = const()[name = string("op_8219_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8219_cast_fp16 = rsqrt(epsilon = var_8219_epsilon_0, x = var_8218_cast_fp16)[name = string("op_8219_cast_fp16")]; + tensor hidden_states_275_cast_fp16 = mul(x = inputs_223_cast_fp16, y = var_8219_cast_fp16)[name = string("hidden_states_275_cast_fp16")]; + tensor input_227_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_275_cast_fp16)[name = string("input_227_cast_fp16")]; + string input_229_pad_type_0 = const()[name = string("input_229_pad_type_0"), val = string("valid")]; + tensor input_229_strides_0 = const()[name = string("input_229_strides_0"), val = tensor([1, 1])]; + tensor input_229_pad_0 = const()[name = string("input_229_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_229_dilations_0 = const()[name = string("input_229_dilations_0"), val = tensor([1, 1])]; + int32 input_229_groups_0 = const()[name = string("input_229_groups_0"), val = int32(1)]; + tensor input_229_cast_fp16 = conv(dilations = input_229_dilations_0, groups = input_229_groups_0, pad = input_229_pad_0, pad_type = input_229_pad_type_0, strides = input_229_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_227_cast_fp16)[name = string("input_229_cast_fp16")]; + tensor var_8233_cast_fp16 = silu(x = input_229_cast_fp16)[name = string("op_8233_cast_fp16")]; + string var_8239_pad_type_0 = const()[name = string("op_8239_pad_type_0"), val = string("valid")]; + tensor var_8239_strides_0 = const()[name = string("op_8239_strides_0"), val = tensor([1, 1])]; + tensor var_8239_pad_0 = const()[name = string("op_8239_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8239_dilations_0 = const()[name = string("op_8239_dilations_0"), val = tensor([1, 1])]; + int32 var_8239_groups_0 = const()[name = string("op_8239_groups_0"), val = int32(1)]; + tensor var_8239_cast_fp16 = conv(dilations = var_8239_dilations_0, groups = var_8239_groups_0, pad = var_8239_pad_0, pad_type = var_8239_pad_type_0, strides = var_8239_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_227_cast_fp16)[name = string("op_8239_cast_fp16")]; + tensor input_231_cast_fp16 = mul(x = var_8233_cast_fp16, y = var_8239_cast_fp16)[name = string("input_231_cast_fp16")]; + string hidden_states_277_pad_type_0 = const()[name = string("hidden_states_277_pad_type_0"), val = string("valid")]; + tensor hidden_states_277_strides_0 = const()[name = string("hidden_states_277_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_277_pad_0 = const()[name = string("hidden_states_277_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_277_dilations_0 = const()[name = string("hidden_states_277_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_277_groups_0 = const()[name = string("hidden_states_277_groups_0"), val = int32(1)]; + tensor hidden_states_277_cast_fp16 = conv(dilations = hidden_states_277_dilations_0, groups = hidden_states_277_groups_0, pad = hidden_states_277_pad_0, pad_type = hidden_states_277_pad_type_0, strides = hidden_states_277_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_231_cast_fp16)[name = string("hidden_states_277_cast_fp16")]; + tensor inputs_225_cast_fp16 = add(x = inputs_223_cast_fp16, y = hidden_states_277_cast_fp16)[name = string("inputs_225_cast_fp16")]; + tensor obj_243_begin_0 = const()[name = string("obj_243_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_243_end_0 = const()[name = string("obj_243_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_243_end_mask_0 = const()[name = string("obj_243_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_243_cast_fp16 = slice_by_index(begin = obj_243_begin_0, end = obj_243_end_0, end_mask = obj_243_end_mask_0, x = key_caches_11_cast_fp16)[name = string("obj_243_cast_fp16")]; + tensor obj_245_begin_0 = const()[name = string("obj_245_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_245_end_0 = const()[name = string("obj_245_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_245_end_mask_0 = const()[name = string("obj_245_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_245_cast_fp16 = slice_by_index(begin = obj_245_begin_0, end = obj_245_end_0, end_mask = obj_245_end_mask_0, x = value_caches_11_cast_fp16)[name = string("obj_245_cast_fp16")]; + int32 var_8287 = const()[name = string("op_8287"), val = int32(3)]; + int32 var_8297 = const()[name = string("op_8297"), val = int32(-2)]; + tensor inputs_sq_225_cast_fp16 = mul(x = inputs_225_cast_fp16, y = inputs_225_cast_fp16)[name = string("inputs_sq_225_cast_fp16")]; + tensor variance_225_axes_0 = const()[name = string("variance_225_axes_0"), val = tensor([1])]; + bool variance_225_keep_dims_0 = const()[name = string("variance_225_keep_dims_0"), val = bool(true)]; + tensor variance_225_cast_fp16 = reduce_mean(axes = variance_225_axes_0, keep_dims = variance_225_keep_dims_0, x = inputs_sq_225_cast_fp16)[name = string("variance_225_cast_fp16")]; + fp16 var_8311_to_fp16 = const()[name = string("op_8311_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8312_cast_fp16 = add(x = variance_225_cast_fp16, y = var_8311_to_fp16)[name = string("op_8312_cast_fp16")]; + fp32 var_8313_epsilon_0 = const()[name = string("op_8313_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8313_cast_fp16 = rsqrt(epsilon = var_8313_epsilon_0, x = var_8312_cast_fp16)[name = string("op_8313_cast_fp16")]; + tensor hidden_states_279_cast_fp16 = mul(x = inputs_225_cast_fp16, y = var_8313_cast_fp16)[name = string("hidden_states_279_cast_fp16")]; + tensor obj_241_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_279_cast_fp16)[name = string("obj_241_cast_fp16")]; + string query_163_pad_type_0 = const()[name = string("query_163_pad_type_0"), val = string("valid")]; + tensor query_163_strides_0 = const()[name = string("query_163_strides_0"), val = tensor([1, 1])]; + tensor query_163_pad_0 = const()[name = string("query_163_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_163_dilations_0 = const()[name = string("query_163_dilations_0"), val = tensor([1, 1])]; + int32 query_163_groups_0 = const()[name = string("query_163_groups_0"), val = int32(1)]; + tensor query_163_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_163_dilations_0, groups = query_163_groups_0, pad = query_163_pad_0, pad_type = query_163_pad_type_0, strides = query_163_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_241_cast_fp16)[name = string("query_163_cast_fp16")]; + string current_key_109_pad_type_0 = const()[name = string("current_key_109_pad_type_0"), val = string("valid")]; + tensor current_key_109_strides_0 = const()[name = string("current_key_109_strides_0"), val = tensor([1, 1])]; + tensor current_key_109_pad_0 = const()[name = string("current_key_109_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_109_dilations_0 = const()[name = string("current_key_109_dilations_0"), val = tensor([1, 1])]; + int32 current_key_109_groups_0 = const()[name = string("current_key_109_groups_0"), val = int32(1)]; + tensor current_key_109_cast_fp16 = conv(dilations = current_key_109_dilations_0, groups = current_key_109_groups_0, pad = current_key_109_pad_0, pad_type = current_key_109_pad_type_0, strides = current_key_109_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_241_cast_fp16)[name = string("current_key_109_cast_fp16")]; + string current_value_55_pad_type_0 = const()[name = string("current_value_55_pad_type_0"), val = string("valid")]; + tensor current_value_55_strides_0 = const()[name = string("current_value_55_strides_0"), val = tensor([1, 1])]; + tensor current_value_55_pad_0 = const()[name = string("current_value_55_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_55_dilations_0 = const()[name = string("current_value_55_dilations_0"), val = tensor([1, 1])]; + int32 current_value_55_groups_0 = const()[name = string("current_value_55_groups_0"), val = int32(1)]; + tensor current_value_55_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_55_dilations_0, groups = current_value_55_groups_0, pad = current_value_55_pad_0, pad_type = current_value_55_pad_type_0, strides = current_value_55_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_241_cast_fp16)[name = string("current_value_55_cast_fp16")]; + tensor var_8350 = const()[name = string("op_8350"), val = tensor([16, 128, 1, 1])]; + tensor inputs_227_cast_fp16 = reshape(shape = var_8350, x = query_163_cast_fp16)[name = string("inputs_227_cast_fp16")]; + tensor inputs_sq_227_cast_fp16 = mul(x = inputs_227_cast_fp16, y = inputs_227_cast_fp16)[name = string("inputs_sq_227_cast_fp16")]; + tensor variance_227_axes_0 = const()[name = string("variance_227_axes_0"), val = tensor([1])]; + bool variance_227_keep_dims_0 = const()[name = string("variance_227_keep_dims_0"), val = bool(true)]; + tensor variance_227_cast_fp16 = reduce_mean(axes = variance_227_axes_0, keep_dims = variance_227_keep_dims_0, x = inputs_sq_227_cast_fp16)[name = string("variance_227_cast_fp16")]; + fp16 var_8356_to_fp16 = const()[name = string("op_8356_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8357_cast_fp16 = add(x = variance_227_cast_fp16, y = var_8356_to_fp16)[name = string("op_8357_cast_fp16")]; + fp32 var_8358_epsilon_0 = const()[name = string("op_8358_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8358_cast_fp16 = rsqrt(epsilon = var_8358_epsilon_0, x = var_8357_cast_fp16)[name = string("op_8358_cast_fp16")]; + tensor hidden_states_281_cast_fp16 = mul(x = inputs_227_cast_fp16, y = var_8358_cast_fp16)[name = string("hidden_states_281_cast_fp16")]; + tensor query_normed_55_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_281_cast_fp16)[name = string("query_normed_55_cast_fp16")]; + tensor var_8366 = const()[name = string("op_8366"), val = tensor([8, 128, 1, 1])]; + tensor inputs_229_cast_fp16 = reshape(shape = var_8366, x = current_key_109_cast_fp16)[name = string("inputs_229_cast_fp16")]; + tensor inputs_sq_229_cast_fp16 = mul(x = inputs_229_cast_fp16, y = inputs_229_cast_fp16)[name = string("inputs_sq_229_cast_fp16")]; + tensor variance_229_axes_0 = const()[name = string("variance_229_axes_0"), val = tensor([1])]; + bool variance_229_keep_dims_0 = const()[name = string("variance_229_keep_dims_0"), val = bool(true)]; + tensor variance_229_cast_fp16 = reduce_mean(axes = variance_229_axes_0, keep_dims = variance_229_keep_dims_0, x = inputs_sq_229_cast_fp16)[name = string("variance_229_cast_fp16")]; + fp16 var_8372_to_fp16 = const()[name = string("op_8372_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8373_cast_fp16 = add(x = variance_229_cast_fp16, y = var_8372_to_fp16)[name = string("op_8373_cast_fp16")]; + fp32 var_8374_epsilon_0 = const()[name = string("op_8374_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8374_cast_fp16 = rsqrt(epsilon = var_8374_epsilon_0, x = var_8373_cast_fp16)[name = string("op_8374_cast_fp16")]; + tensor hidden_states_283_cast_fp16 = mul(x = inputs_229_cast_fp16, y = var_8374_cast_fp16)[name = string("hidden_states_283_cast_fp16")]; + tensor current_key_normed_55_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_283_cast_fp16)[name = string("current_key_normed_55_cast_fp16")]; + tensor var_8392 = const()[name = string("op_8392"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_217_cast_fp16 = reshape(shape = var_8392, x = query_normed_55_cast_fp16)[name = string("mh_q_217_cast_fp16")]; + tensor var_8394 = const()[name = string("op_8394"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_217_cast_fp16 = reshape(shape = var_8394, x = current_key_normed_55_cast_fp16)[name = string("mh_k_217_cast_fp16")]; + tensor var_8398_cast_fp16 = mul(x = mh_q_217_cast_fp16, y = cos_51_to_fp16)[name = string("op_8398_cast_fp16")]; + tensor var_8403_begin_0 = const()[name = string("op_8403_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8403_end_0 = const()[name = string("op_8403_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_8403_end_mask_0 = const()[name = string("op_8403_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_8403_cast_fp16 = slice_by_index(begin = var_8403_begin_0, end = var_8403_end_0, end_mask = var_8403_end_mask_0, x = mh_q_217_cast_fp16)[name = string("op_8403_cast_fp16")]; + tensor var_8409_begin_0 = const()[name = string("op_8409_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_8409_end_0 = const()[name = string("op_8409_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_8409_end_mask_0 = const()[name = string("op_8409_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_8409_cast_fp16 = slice_by_index(begin = var_8409_begin_0, end = var_8409_end_0, end_mask = var_8409_end_mask_0, x = mh_q_217_cast_fp16)[name = string("op_8409_cast_fp16")]; + fp16 const_559_promoted_to_fp16 = const()[name = string("const_559_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8411_cast_fp16 = mul(x = var_8409_cast_fp16, y = const_559_promoted_to_fp16)[name = string("op_8411_cast_fp16")]; + bool var_8413_interleave_0 = const()[name = string("op_8413_interleave_0"), val = bool(false)]; + tensor var_8413_cast_fp16 = concat(axis = var_8297, interleave = var_8413_interleave_0, values = (var_8411_cast_fp16, var_8403_cast_fp16))[name = string("op_8413_cast_fp16")]; + tensor var_8414_cast_fp16 = mul(x = var_8413_cast_fp16, y = sin_51_to_fp16)[name = string("op_8414_cast_fp16")]; + tensor mh_q_219_cast_fp16 = add(x = var_8398_cast_fp16, y = var_8414_cast_fp16)[name = string("mh_q_219_cast_fp16")]; + tensor var_8416_cast_fp16 = mul(x = mh_k_217_cast_fp16, y = cos_51_to_fp16)[name = string("op_8416_cast_fp16")]; + tensor var_8421_begin_0 = const()[name = string("op_8421_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8421_end_0 = const()[name = string("op_8421_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_8421_end_mask_0 = const()[name = string("op_8421_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_8421_cast_fp16 = slice_by_index(begin = var_8421_begin_0, end = var_8421_end_0, end_mask = var_8421_end_mask_0, x = mh_k_217_cast_fp16)[name = string("op_8421_cast_fp16")]; + tensor var_8427_begin_0 = const()[name = string("op_8427_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_8427_end_0 = const()[name = string("op_8427_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_8427_end_mask_0 = const()[name = string("op_8427_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_8427_cast_fp16 = slice_by_index(begin = var_8427_begin_0, end = var_8427_end_0, end_mask = var_8427_end_mask_0, x = mh_k_217_cast_fp16)[name = string("op_8427_cast_fp16")]; + fp16 const_562_promoted_to_fp16 = const()[name = string("const_562_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8429_cast_fp16 = mul(x = var_8427_cast_fp16, y = const_562_promoted_to_fp16)[name = string("op_8429_cast_fp16")]; + bool var_8431_interleave_0 = const()[name = string("op_8431_interleave_0"), val = bool(false)]; + tensor var_8431_cast_fp16 = concat(axis = var_8297, interleave = var_8431_interleave_0, values = (var_8429_cast_fp16, var_8421_cast_fp16))[name = string("op_8431_cast_fp16")]; + tensor var_8432_cast_fp16 = mul(x = var_8431_cast_fp16, y = sin_51_to_fp16)[name = string("op_8432_cast_fp16")]; + tensor mh_k_219_cast_fp16 = add(x = var_8416_cast_fp16, y = var_8432_cast_fp16)[name = string("mh_k_219_cast_fp16")]; + tensor var_8436 = const()[name = string("op_8436"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_111_cast_fp16 = reshape(shape = var_8436, x = mh_k_219_cast_fp16)[name = string("current_key_111_cast_fp16")]; + tensor var_8442_to_fp16 = const()[name = string("op_8442_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194304)))]; + tensor var_8443_cast_fp16 = mul(x = obj_243_cast_fp16, y = var_8442_to_fp16)[name = string("op_8443_cast_fp16")]; + tensor var_8440_to_fp16 = const()[name = string("op_8440_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194432)))]; + tensor var_8444_cast_fp16 = mul(x = current_key_111_cast_fp16, y = var_8440_to_fp16)[name = string("op_8444_cast_fp16")]; + tensor key_111_cast_fp16 = add(x = var_8443_cast_fp16, y = var_8444_cast_fp16)[name = string("key_111_cast_fp16")]; + tensor var_8446_to_fp16 = const()[name = string("op_8446_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194304)))]; + tensor var_8447_cast_fp16 = mul(x = obj_245_cast_fp16, y = var_8446_to_fp16)[name = string("op_8447_cast_fp16")]; + tensor var_8448_cast_fp16 = mul(x = current_value_55_cast_fp16, y = var_8440_to_fp16)[name = string("op_8448_cast_fp16")]; + tensor value_55_cast_fp16 = add(x = var_8447_cast_fp16, y = var_8448_cast_fp16)[name = string("value_55_cast_fp16")]; + fp16 var_8455_to_fp16 = const()[name = string("op_8455_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_223_cast_fp16 = mul(x = mh_q_219_cast_fp16, y = var_8455_to_fp16)[name = string("mh_q_223_cast_fp16")]; + tensor var_8457 = const()[name = string("op_8457"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_221_cast_fp16 = reshape(shape = var_8457, x = key_111_cast_fp16)[name = string("mh_k_221_cast_fp16")]; + tensor var_8459 = const()[name = string("op_8459"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_109_cast_fp16 = reshape(shape = var_8459, x = value_55_cast_fp16)[name = string("mh_v_109_cast_fp16")]; + tensor transpose_108_perm_0 = const()[name = string("transpose_108_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_54_reps_0 = const()[name = string("tile_54_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_108_cast_fp16 = transpose(perm = transpose_108_perm_0, x = mh_k_221_cast_fp16)[name = string("transpose_317")]; + tensor tile_54_cast_fp16 = tile(reps = tile_54_reps_0, x = transpose_108_cast_fp16)[name = string("tile_54_cast_fp16")]; + tensor concat_136 = const()[name = string("concat_136"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_108_cast_fp16 = reshape(shape = concat_136, x = tile_54_cast_fp16)[name = string("reshape_108_cast_fp16")]; + tensor transpose_109_perm_0 = const()[name = string("transpose_109_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_137 = const()[name = string("concat_137"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_109_cast_fp16 = transpose(perm = transpose_109_perm_0, x = reshape_108_cast_fp16)[name = string("transpose_316")]; + tensor reshape_109_cast_fp16 = reshape(shape = concat_137, x = transpose_109_cast_fp16)[name = string("reshape_109_cast_fp16")]; + tensor transpose_110_perm_0 = const()[name = string("transpose_110_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_55_reps_0 = const()[name = string("tile_55_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_110_cast_fp16 = transpose(perm = transpose_110_perm_0, x = mh_v_109_cast_fp16)[name = string("transpose_315")]; + tensor tile_55_cast_fp16 = tile(reps = tile_55_reps_0, x = transpose_110_cast_fp16)[name = string("tile_55_cast_fp16")]; + tensor concat_138 = const()[name = string("concat_138"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_110_cast_fp16 = reshape(shape = concat_138, x = tile_55_cast_fp16)[name = string("reshape_110_cast_fp16")]; + tensor transpose_111_perm_0 = const()[name = string("transpose_111_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_139 = const()[name = string("concat_139"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_111_cast_fp16 = transpose(perm = transpose_111_perm_0, x = reshape_110_cast_fp16)[name = string("transpose_314")]; + tensor reshape_111_cast_fp16 = reshape(shape = concat_139, x = transpose_111_cast_fp16)[name = string("reshape_111_cast_fp16")]; + tensor transpose_425_perm_0 = const()[name = string("transpose_425_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_163_transpose_x_1 = const()[name = string("mh_w_163_transpose_x_1"), val = bool(true)]; + bool mh_w_163_transpose_y_1 = const()[name = string("mh_w_163_transpose_y_1"), val = bool(false)]; + tensor transpose_425_cast_fp16 = transpose(perm = transpose_425_perm_0, x = reshape_109_cast_fp16)[name = string("transpose_313")]; + tensor mh_w_163_cast_fp16 = matmul(transpose_x = mh_w_163_transpose_x_1, transpose_y = mh_w_163_transpose_y_1, x = mh_q_223_cast_fp16, y = transpose_425_cast_fp16)[name = string("mh_w_163_cast_fp16")]; + tensor var_8467_to_fp16 = const()[name = string("op_8467_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194560)))]; + tensor mh_w_165_cast_fp16 = add(x = mh_w_163_cast_fp16, y = var_8467_to_fp16)[name = string("mh_w_165_cast_fp16")]; + tensor mh_w_167_cast_fp16 = softmax(axis = var_8287, x = mh_w_165_cast_fp16)[name = string("mh_w_167_cast_fp16")]; + tensor transpose_426_perm_0 = const()[name = string("transpose_426_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_55_transpose_x_1 = const()[name = string("attn_55_transpose_x_1"), val = bool(false)]; + bool attn_55_transpose_y_1 = const()[name = string("attn_55_transpose_y_1"), val = bool(true)]; + tensor transpose_426_cast_fp16 = transpose(perm = transpose_426_perm_0, x = reshape_111_cast_fp16)[name = string("transpose_312")]; + tensor attn_55_cast_fp16 = matmul(transpose_x = attn_55_transpose_x_1, transpose_y = attn_55_transpose_y_1, x = transpose_426_cast_fp16, y = mh_w_167_cast_fp16)[name = string("attn_55_cast_fp16")]; + tensor var_8473 = const()[name = string("op_8473"), val = tensor([1, 2048, 1, 1])]; + tensor input_233_cast_fp16 = reshape(shape = var_8473, x = attn_55_cast_fp16)[name = string("input_233_cast_fp16")]; + string obj_247_pad_type_0 = const()[name = string("obj_247_pad_type_0"), val = string("valid")]; + tensor obj_247_strides_0 = const()[name = string("obj_247_strides_0"), val = tensor([1, 1])]; + tensor obj_247_pad_0 = const()[name = string("obj_247_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_247_dilations_0 = const()[name = string("obj_247_dilations_0"), val = tensor([1, 1])]; + int32 obj_247_groups_0 = const()[name = string("obj_247_groups_0"), val = int32(1)]; + tensor obj_247_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_247_dilations_0, groups = obj_247_groups_0, pad = obj_247_pad_0, pad_type = obj_247_pad_type_0, strides = obj_247_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_233_cast_fp16)[name = string("obj_247_cast_fp16")]; + tensor inputs_231_cast_fp16 = add(x = inputs_225_cast_fp16, y = obj_247_cast_fp16)[name = string("inputs_231_cast_fp16")]; + tensor inputs_sq_231_cast_fp16 = mul(x = inputs_231_cast_fp16, y = inputs_231_cast_fp16)[name = string("inputs_sq_231_cast_fp16")]; + tensor variance_231_axes_0 = const()[name = string("variance_231_axes_0"), val = tensor([1])]; + bool variance_231_keep_dims_0 = const()[name = string("variance_231_keep_dims_0"), val = bool(true)]; + tensor variance_231_cast_fp16 = reduce_mean(axes = variance_231_axes_0, keep_dims = variance_231_keep_dims_0, x = inputs_sq_231_cast_fp16)[name = string("variance_231_cast_fp16")]; + fp16 var_8491_to_fp16 = const()[name = string("op_8491_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8492_cast_fp16 = add(x = variance_231_cast_fp16, y = var_8491_to_fp16)[name = string("op_8492_cast_fp16")]; + fp32 var_8493_epsilon_0 = const()[name = string("op_8493_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8493_cast_fp16 = rsqrt(epsilon = var_8493_epsilon_0, x = var_8492_cast_fp16)[name = string("op_8493_cast_fp16")]; + tensor hidden_states_285_cast_fp16 = mul(x = inputs_231_cast_fp16, y = var_8493_cast_fp16)[name = string("hidden_states_285_cast_fp16")]; + tensor input_235_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_285_cast_fp16)[name = string("input_235_cast_fp16")]; + string input_237_pad_type_0 = const()[name = string("input_237_pad_type_0"), val = string("valid")]; + tensor input_237_strides_0 = const()[name = string("input_237_strides_0"), val = tensor([1, 1])]; + tensor input_237_pad_0 = const()[name = string("input_237_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_237_dilations_0 = const()[name = string("input_237_dilations_0"), val = tensor([1, 1])]; + int32 input_237_groups_0 = const()[name = string("input_237_groups_0"), val = int32(1)]; + tensor input_237_cast_fp16 = conv(dilations = input_237_dilations_0, groups = input_237_groups_0, pad = input_237_pad_0, pad_type = input_237_pad_type_0, strides = input_237_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_235_cast_fp16)[name = string("input_237_cast_fp16")]; + tensor var_8507_cast_fp16 = silu(x = input_237_cast_fp16)[name = string("op_8507_cast_fp16")]; + string var_8513_pad_type_0 = const()[name = string("op_8513_pad_type_0"), val = string("valid")]; + tensor var_8513_strides_0 = const()[name = string("op_8513_strides_0"), val = tensor([1, 1])]; + tensor var_8513_pad_0 = const()[name = string("op_8513_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8513_dilations_0 = const()[name = string("op_8513_dilations_0"), val = tensor([1, 1])]; + int32 var_8513_groups_0 = const()[name = string("op_8513_groups_0"), val = int32(1)]; + tensor var_8513_cast_fp16 = conv(dilations = var_8513_dilations_0, groups = var_8513_groups_0, pad = var_8513_pad_0, pad_type = var_8513_pad_type_0, strides = var_8513_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_235_cast_fp16)[name = string("op_8513_cast_fp16")]; + tensor input_239_cast_fp16 = mul(x = var_8507_cast_fp16, y = var_8513_cast_fp16)[name = string("input_239_cast_fp16")]; + string hidden_states_287_pad_type_0 = const()[name = string("hidden_states_287_pad_type_0"), val = string("valid")]; + tensor hidden_states_287_strides_0 = const()[name = string("hidden_states_287_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_287_pad_0 = const()[name = string("hidden_states_287_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_287_dilations_0 = const()[name = string("hidden_states_287_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_287_groups_0 = const()[name = string("hidden_states_287_groups_0"), val = int32(1)]; + tensor hidden_states_287_cast_fp16 = conv(dilations = hidden_states_287_dilations_0, groups = hidden_states_287_groups_0, pad = hidden_states_287_pad_0, pad_type = hidden_states_287_pad_type_0, strides = hidden_states_287_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_239_cast_fp16)[name = string("hidden_states_287_cast_fp16")]; + tensor inputs_233_cast_fp16 = add(x = inputs_231_cast_fp16, y = hidden_states_287_cast_fp16)[name = string("inputs_233_cast_fp16")]; + tensor obj_251_begin_0 = const()[name = string("obj_251_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_251_end_0 = const()[name = string("obj_251_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_251_end_mask_0 = const()[name = string("obj_251_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_251_cast_fp16 = slice_by_index(begin = obj_251_begin_0, end = obj_251_end_0, end_mask = obj_251_end_mask_0, x = key_caches_11_cast_fp16)[name = string("obj_251_cast_fp16")]; + tensor obj_253_begin_0 = const()[name = string("obj_253_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_253_end_0 = const()[name = string("obj_253_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_253_end_mask_0 = const()[name = string("obj_253_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_253_cast_fp16 = slice_by_index(begin = obj_253_begin_0, end = obj_253_end_0, end_mask = obj_253_end_mask_0, x = value_caches_11_cast_fp16)[name = string("obj_253_cast_fp16")]; + int32 var_8561 = const()[name = string("op_8561"), val = int32(3)]; + int32 var_8571 = const()[name = string("op_8571"), val = int32(-2)]; + tensor inputs_sq_233_cast_fp16 = mul(x = inputs_233_cast_fp16, y = inputs_233_cast_fp16)[name = string("inputs_sq_233_cast_fp16")]; + tensor variance_233_axes_0 = const()[name = string("variance_233_axes_0"), val = tensor([1])]; + bool variance_233_keep_dims_0 = const()[name = string("variance_233_keep_dims_0"), val = bool(true)]; + tensor variance_233_cast_fp16 = reduce_mean(axes = variance_233_axes_0, keep_dims = variance_233_keep_dims_0, x = inputs_sq_233_cast_fp16)[name = string("variance_233_cast_fp16")]; + fp16 var_8585_to_fp16 = const()[name = string("op_8585_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8586_cast_fp16 = add(x = variance_233_cast_fp16, y = var_8585_to_fp16)[name = string("op_8586_cast_fp16")]; + fp32 var_8587_epsilon_0 = const()[name = string("op_8587_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8587_cast_fp16 = rsqrt(epsilon = var_8587_epsilon_0, x = var_8586_cast_fp16)[name = string("op_8587_cast_fp16")]; + tensor hidden_states_289_cast_fp16 = mul(x = inputs_233_cast_fp16, y = var_8587_cast_fp16)[name = string("hidden_states_289_cast_fp16")]; + tensor obj_249_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_289_cast_fp16)[name = string("obj_249_cast_fp16")]; + string query_169_pad_type_0 = const()[name = string("query_169_pad_type_0"), val = string("valid")]; + tensor query_169_strides_0 = const()[name = string("query_169_strides_0"), val = tensor([1, 1])]; + tensor query_169_pad_0 = const()[name = string("query_169_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_169_dilations_0 = const()[name = string("query_169_dilations_0"), val = tensor([1, 1])]; + int32 query_169_groups_0 = const()[name = string("query_169_groups_0"), val = int32(1)]; + tensor query_169_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_169_dilations_0, groups = query_169_groups_0, pad = query_169_pad_0, pad_type = query_169_pad_type_0, strides = query_169_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_249_cast_fp16)[name = string("query_169_cast_fp16")]; + string current_key_113_pad_type_0 = const()[name = string("current_key_113_pad_type_0"), val = string("valid")]; + tensor current_key_113_strides_0 = const()[name = string("current_key_113_strides_0"), val = tensor([1, 1])]; + tensor current_key_113_pad_0 = const()[name = string("current_key_113_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_113_dilations_0 = const()[name = string("current_key_113_dilations_0"), val = tensor([1, 1])]; + int32 current_key_113_groups_0 = const()[name = string("current_key_113_groups_0"), val = int32(1)]; + tensor current_key_113_cast_fp16 = conv(dilations = current_key_113_dilations_0, groups = current_key_113_groups_0, pad = current_key_113_pad_0, pad_type = current_key_113_pad_type_0, strides = current_key_113_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_249_cast_fp16)[name = string("current_key_113_cast_fp16")]; + string current_value_57_pad_type_0 = const()[name = string("current_value_57_pad_type_0"), val = string("valid")]; + tensor current_value_57_strides_0 = const()[name = string("current_value_57_strides_0"), val = tensor([1, 1])]; + tensor current_value_57_pad_0 = const()[name = string("current_value_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_57_dilations_0 = const()[name = string("current_value_57_dilations_0"), val = tensor([1, 1])]; + int32 current_value_57_groups_0 = const()[name = string("current_value_57_groups_0"), val = int32(1)]; + tensor current_value_57_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_57_dilations_0, groups = current_value_57_groups_0, pad = current_value_57_pad_0, pad_type = current_value_57_pad_type_0, strides = current_value_57_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_249_cast_fp16)[name = string("current_value_57_cast_fp16")]; + tensor var_8624 = const()[name = string("op_8624"), val = tensor([16, 128, 1, 1])]; + tensor inputs_235_cast_fp16 = reshape(shape = var_8624, x = query_169_cast_fp16)[name = string("inputs_235_cast_fp16")]; + tensor inputs_sq_235_cast_fp16 = mul(x = inputs_235_cast_fp16, y = inputs_235_cast_fp16)[name = string("inputs_sq_235_cast_fp16")]; + tensor variance_235_axes_0 = const()[name = string("variance_235_axes_0"), val = tensor([1])]; + bool variance_235_keep_dims_0 = const()[name = string("variance_235_keep_dims_0"), val = bool(true)]; + tensor variance_235_cast_fp16 = reduce_mean(axes = variance_235_axes_0, keep_dims = variance_235_keep_dims_0, x = inputs_sq_235_cast_fp16)[name = string("variance_235_cast_fp16")]; + fp16 var_8630_to_fp16 = const()[name = string("op_8630_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8631_cast_fp16 = add(x = variance_235_cast_fp16, y = var_8630_to_fp16)[name = string("op_8631_cast_fp16")]; + fp32 var_8632_epsilon_0 = const()[name = string("op_8632_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8632_cast_fp16 = rsqrt(epsilon = var_8632_epsilon_0, x = var_8631_cast_fp16)[name = string("op_8632_cast_fp16")]; + tensor hidden_states_291_cast_fp16 = mul(x = inputs_235_cast_fp16, y = var_8632_cast_fp16)[name = string("hidden_states_291_cast_fp16")]; + tensor query_normed_57_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_291_cast_fp16)[name = string("query_normed_57_cast_fp16")]; + tensor var_8640 = const()[name = string("op_8640"), val = tensor([8, 128, 1, 1])]; + tensor inputs_237_cast_fp16 = reshape(shape = var_8640, x = current_key_113_cast_fp16)[name = string("inputs_237_cast_fp16")]; + tensor inputs_sq_237_cast_fp16 = mul(x = inputs_237_cast_fp16, y = inputs_237_cast_fp16)[name = string("inputs_sq_237_cast_fp16")]; + tensor variance_237_axes_0 = const()[name = string("variance_237_axes_0"), val = tensor([1])]; + bool variance_237_keep_dims_0 = const()[name = string("variance_237_keep_dims_0"), val = bool(true)]; + tensor variance_237_cast_fp16 = reduce_mean(axes = variance_237_axes_0, keep_dims = variance_237_keep_dims_0, x = inputs_sq_237_cast_fp16)[name = string("variance_237_cast_fp16")]; + fp16 var_8646_to_fp16 = const()[name = string("op_8646_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8647_cast_fp16 = add(x = variance_237_cast_fp16, y = var_8646_to_fp16)[name = string("op_8647_cast_fp16")]; + fp32 var_8648_epsilon_0 = const()[name = string("op_8648_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8648_cast_fp16 = rsqrt(epsilon = var_8648_epsilon_0, x = var_8647_cast_fp16)[name = string("op_8648_cast_fp16")]; + tensor hidden_states_293_cast_fp16 = mul(x = inputs_237_cast_fp16, y = var_8648_cast_fp16)[name = string("hidden_states_293_cast_fp16")]; + tensor current_key_normed_57_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_293_cast_fp16)[name = string("current_key_normed_57_cast_fp16")]; + tensor var_8666 = const()[name = string("op_8666"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_225_cast_fp16 = reshape(shape = var_8666, x = query_normed_57_cast_fp16)[name = string("mh_q_225_cast_fp16")]; + tensor var_8668 = const()[name = string("op_8668"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_225_cast_fp16 = reshape(shape = var_8668, x = current_key_normed_57_cast_fp16)[name = string("mh_k_225_cast_fp16")]; + tensor var_8672_cast_fp16 = mul(x = mh_q_225_cast_fp16, y = cos_51_to_fp16)[name = string("op_8672_cast_fp16")]; + tensor var_8677_begin_0 = const()[name = string("op_8677_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8677_end_0 = const()[name = string("op_8677_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_8677_end_mask_0 = const()[name = string("op_8677_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_8677_cast_fp16 = slice_by_index(begin = var_8677_begin_0, end = var_8677_end_0, end_mask = var_8677_end_mask_0, x = mh_q_225_cast_fp16)[name = string("op_8677_cast_fp16")]; + tensor var_8683_begin_0 = const()[name = string("op_8683_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_8683_end_0 = const()[name = string("op_8683_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_8683_end_mask_0 = const()[name = string("op_8683_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_8683_cast_fp16 = slice_by_index(begin = var_8683_begin_0, end = var_8683_end_0, end_mask = var_8683_end_mask_0, x = mh_q_225_cast_fp16)[name = string("op_8683_cast_fp16")]; + fp16 const_579_promoted_to_fp16 = const()[name = string("const_579_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8685_cast_fp16 = mul(x = var_8683_cast_fp16, y = const_579_promoted_to_fp16)[name = string("op_8685_cast_fp16")]; + bool var_8687_interleave_0 = const()[name = string("op_8687_interleave_0"), val = bool(false)]; + tensor var_8687_cast_fp16 = concat(axis = var_8571, interleave = var_8687_interleave_0, values = (var_8685_cast_fp16, var_8677_cast_fp16))[name = string("op_8687_cast_fp16")]; + tensor var_8688_cast_fp16 = mul(x = var_8687_cast_fp16, y = sin_51_to_fp16)[name = string("op_8688_cast_fp16")]; + tensor mh_q_227_cast_fp16 = add(x = var_8672_cast_fp16, y = var_8688_cast_fp16)[name = string("mh_q_227_cast_fp16")]; + tensor var_8690_cast_fp16 = mul(x = mh_k_225_cast_fp16, y = cos_51_to_fp16)[name = string("op_8690_cast_fp16")]; + tensor var_8695_begin_0 = const()[name = string("op_8695_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8695_end_0 = const()[name = string("op_8695_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_8695_end_mask_0 = const()[name = string("op_8695_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_8695_cast_fp16 = slice_by_index(begin = var_8695_begin_0, end = var_8695_end_0, end_mask = var_8695_end_mask_0, x = mh_k_225_cast_fp16)[name = string("op_8695_cast_fp16")]; + tensor var_8701_begin_0 = const()[name = string("op_8701_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_8701_end_0 = const()[name = string("op_8701_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_8701_end_mask_0 = const()[name = string("op_8701_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_8701_cast_fp16 = slice_by_index(begin = var_8701_begin_0, end = var_8701_end_0, end_mask = var_8701_end_mask_0, x = mh_k_225_cast_fp16)[name = string("op_8701_cast_fp16")]; + fp16 const_582_promoted_to_fp16 = const()[name = string("const_582_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8703_cast_fp16 = mul(x = var_8701_cast_fp16, y = const_582_promoted_to_fp16)[name = string("op_8703_cast_fp16")]; + bool var_8705_interleave_0 = const()[name = string("op_8705_interleave_0"), val = bool(false)]; + tensor var_8705_cast_fp16 = concat(axis = var_8571, interleave = var_8705_interleave_0, values = (var_8703_cast_fp16, var_8695_cast_fp16))[name = string("op_8705_cast_fp16")]; + tensor var_8706_cast_fp16 = mul(x = var_8705_cast_fp16, y = sin_51_to_fp16)[name = string("op_8706_cast_fp16")]; + tensor mh_k_227_cast_fp16 = add(x = var_8690_cast_fp16, y = var_8706_cast_fp16)[name = string("mh_k_227_cast_fp16")]; + tensor var_8710 = const()[name = string("op_8710"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_115_cast_fp16 = reshape(shape = var_8710, x = mh_k_227_cast_fp16)[name = string("current_key_115_cast_fp16")]; + tensor var_8716_to_fp16 = const()[name = string("op_8716_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194304)))]; + tensor var_8717_cast_fp16 = mul(x = obj_251_cast_fp16, y = var_8716_to_fp16)[name = string("op_8717_cast_fp16")]; + tensor var_8714_to_fp16 = const()[name = string("op_8714_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194432)))]; + tensor var_8718_cast_fp16 = mul(x = current_key_115_cast_fp16, y = var_8714_to_fp16)[name = string("op_8718_cast_fp16")]; + tensor key_115_cast_fp16 = add(x = var_8717_cast_fp16, y = var_8718_cast_fp16)[name = string("key_115_cast_fp16")]; + tensor var_8720_to_fp16 = const()[name = string("op_8720_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194304)))]; + tensor var_8721_cast_fp16 = mul(x = obj_253_cast_fp16, y = var_8720_to_fp16)[name = string("op_8721_cast_fp16")]; + tensor var_8722_cast_fp16 = mul(x = current_value_57_cast_fp16, y = var_8714_to_fp16)[name = string("op_8722_cast_fp16")]; + tensor value_57_cast_fp16 = add(x = var_8721_cast_fp16, y = var_8722_cast_fp16)[name = string("value_57_cast_fp16")]; + fp16 var_8729_to_fp16 = const()[name = string("op_8729_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_231_cast_fp16 = mul(x = mh_q_227_cast_fp16, y = var_8729_to_fp16)[name = string("mh_q_231_cast_fp16")]; + tensor var_8731 = const()[name = string("op_8731"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_229_cast_fp16 = reshape(shape = var_8731, x = key_115_cast_fp16)[name = string("mh_k_229_cast_fp16")]; + tensor var_8733 = const()[name = string("op_8733"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_113_cast_fp16 = reshape(shape = var_8733, x = value_57_cast_fp16)[name = string("mh_v_113_cast_fp16")]; + tensor transpose_112_perm_0 = const()[name = string("transpose_112_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_56_reps_0 = const()[name = string("tile_56_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_112_cast_fp16 = transpose(perm = transpose_112_perm_0, x = mh_k_229_cast_fp16)[name = string("transpose_311")]; + tensor tile_56_cast_fp16 = tile(reps = tile_56_reps_0, x = transpose_112_cast_fp16)[name = string("tile_56_cast_fp16")]; + tensor concat_140 = const()[name = string("concat_140"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_112_cast_fp16 = reshape(shape = concat_140, x = tile_56_cast_fp16)[name = string("reshape_112_cast_fp16")]; + tensor transpose_113_perm_0 = const()[name = string("transpose_113_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_141 = const()[name = string("concat_141"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_113_cast_fp16 = transpose(perm = transpose_113_perm_0, x = reshape_112_cast_fp16)[name = string("transpose_310")]; + tensor reshape_113_cast_fp16 = reshape(shape = concat_141, x = transpose_113_cast_fp16)[name = string("reshape_113_cast_fp16")]; + tensor transpose_114_perm_0 = const()[name = string("transpose_114_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_57_reps_0 = const()[name = string("tile_57_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_114_cast_fp16 = transpose(perm = transpose_114_perm_0, x = mh_v_113_cast_fp16)[name = string("transpose_309")]; + tensor tile_57_cast_fp16 = tile(reps = tile_57_reps_0, x = transpose_114_cast_fp16)[name = string("tile_57_cast_fp16")]; + tensor concat_142 = const()[name = string("concat_142"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_114_cast_fp16 = reshape(shape = concat_142, x = tile_57_cast_fp16)[name = string("reshape_114_cast_fp16")]; + tensor transpose_115_perm_0 = const()[name = string("transpose_115_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_143 = const()[name = string("concat_143"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_115_cast_fp16 = transpose(perm = transpose_115_perm_0, x = reshape_114_cast_fp16)[name = string("transpose_308")]; + tensor reshape_115_cast_fp16 = reshape(shape = concat_143, x = transpose_115_cast_fp16)[name = string("reshape_115_cast_fp16")]; + tensor transpose_429_perm_0 = const()[name = string("transpose_429_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_169_transpose_x_1 = const()[name = string("mh_w_169_transpose_x_1"), val = bool(true)]; + bool mh_w_169_transpose_y_1 = const()[name = string("mh_w_169_transpose_y_1"), val = bool(false)]; + tensor transpose_429_cast_fp16 = transpose(perm = transpose_429_perm_0, x = reshape_113_cast_fp16)[name = string("transpose_307")]; + tensor mh_w_169_cast_fp16 = matmul(transpose_x = mh_w_169_transpose_x_1, transpose_y = mh_w_169_transpose_y_1, x = mh_q_231_cast_fp16, y = transpose_429_cast_fp16)[name = string("mh_w_169_cast_fp16")]; + tensor var_8741_to_fp16 = const()[name = string("op_8741_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194560)))]; + tensor mh_w_171_cast_fp16 = add(x = mh_w_169_cast_fp16, y = var_8741_to_fp16)[name = string("mh_w_171_cast_fp16")]; + tensor mh_w_173_cast_fp16 = softmax(axis = var_8561, x = mh_w_171_cast_fp16)[name = string("mh_w_173_cast_fp16")]; + tensor transpose_430_perm_0 = const()[name = string("transpose_430_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_57_transpose_x_1 = const()[name = string("attn_57_transpose_x_1"), val = bool(false)]; + bool attn_57_transpose_y_1 = const()[name = string("attn_57_transpose_y_1"), val = bool(true)]; + tensor transpose_430_cast_fp16 = transpose(perm = transpose_430_perm_0, x = reshape_115_cast_fp16)[name = string("transpose_306")]; + tensor attn_57_cast_fp16 = matmul(transpose_x = attn_57_transpose_x_1, transpose_y = attn_57_transpose_y_1, x = transpose_430_cast_fp16, y = mh_w_173_cast_fp16)[name = string("attn_57_cast_fp16")]; + tensor var_8747 = const()[name = string("op_8747"), val = tensor([1, 2048, 1, 1])]; + tensor input_241_cast_fp16 = reshape(shape = var_8747, x = attn_57_cast_fp16)[name = string("input_241_cast_fp16")]; + string obj_255_pad_type_0 = const()[name = string("obj_255_pad_type_0"), val = string("valid")]; + tensor obj_255_strides_0 = const()[name = string("obj_255_strides_0"), val = tensor([1, 1])]; + tensor obj_255_pad_0 = const()[name = string("obj_255_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_255_dilations_0 = const()[name = string("obj_255_dilations_0"), val = tensor([1, 1])]; + int32 obj_255_groups_0 = const()[name = string("obj_255_groups_0"), val = int32(1)]; + tensor obj_255_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_255_dilations_0, groups = obj_255_groups_0, pad = obj_255_pad_0, pad_type = obj_255_pad_type_0, strides = obj_255_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_241_cast_fp16)[name = string("obj_255_cast_fp16")]; + tensor inputs_239_cast_fp16 = add(x = inputs_233_cast_fp16, y = obj_255_cast_fp16)[name = string("inputs_239_cast_fp16")]; + tensor inputs_sq_239_cast_fp16 = mul(x = inputs_239_cast_fp16, y = inputs_239_cast_fp16)[name = string("inputs_sq_239_cast_fp16")]; + tensor variance_239_axes_0 = const()[name = string("variance_239_axes_0"), val = tensor([1])]; + bool variance_239_keep_dims_0 = const()[name = string("variance_239_keep_dims_0"), val = bool(true)]; + tensor variance_239_cast_fp16 = reduce_mean(axes = variance_239_axes_0, keep_dims = variance_239_keep_dims_0, x = inputs_sq_239_cast_fp16)[name = string("variance_239_cast_fp16")]; + fp16 var_8765_to_fp16 = const()[name = string("op_8765_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8766_cast_fp16 = add(x = variance_239_cast_fp16, y = var_8765_to_fp16)[name = string("op_8766_cast_fp16")]; + fp32 var_8767_epsilon_0 = const()[name = string("op_8767_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8767_cast_fp16 = rsqrt(epsilon = var_8767_epsilon_0, x = var_8766_cast_fp16)[name = string("op_8767_cast_fp16")]; + tensor hidden_states_295_cast_fp16 = mul(x = inputs_239_cast_fp16, y = var_8767_cast_fp16)[name = string("hidden_states_295_cast_fp16")]; + tensor input_243_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_295_cast_fp16)[name = string("input_243_cast_fp16")]; + string input_245_pad_type_0 = const()[name = string("input_245_pad_type_0"), val = string("valid")]; + tensor input_245_strides_0 = const()[name = string("input_245_strides_0"), val = tensor([1, 1])]; + tensor input_245_pad_0 = const()[name = string("input_245_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_245_dilations_0 = const()[name = string("input_245_dilations_0"), val = tensor([1, 1])]; + int32 input_245_groups_0 = const()[name = string("input_245_groups_0"), val = int32(1)]; + tensor input_245_cast_fp16 = conv(dilations = input_245_dilations_0, groups = input_245_groups_0, pad = input_245_pad_0, pad_type = input_245_pad_type_0, strides = input_245_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_243_cast_fp16)[name = string("input_245_cast_fp16")]; + tensor var_8781_cast_fp16 = silu(x = input_245_cast_fp16)[name = string("op_8781_cast_fp16")]; + string var_8787_pad_type_0 = const()[name = string("op_8787_pad_type_0"), val = string("valid")]; + tensor var_8787_strides_0 = const()[name = string("op_8787_strides_0"), val = tensor([1, 1])]; + tensor var_8787_pad_0 = const()[name = string("op_8787_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8787_dilations_0 = const()[name = string("op_8787_dilations_0"), val = tensor([1, 1])]; + int32 var_8787_groups_0 = const()[name = string("op_8787_groups_0"), val = int32(1)]; + tensor var_8787_cast_fp16 = conv(dilations = var_8787_dilations_0, groups = var_8787_groups_0, pad = var_8787_pad_0, pad_type = var_8787_pad_type_0, strides = var_8787_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_243_cast_fp16)[name = string("op_8787_cast_fp16")]; + tensor input_247_cast_fp16 = mul(x = var_8781_cast_fp16, y = var_8787_cast_fp16)[name = string("input_247_cast_fp16")]; + string hidden_states_297_pad_type_0 = const()[name = string("hidden_states_297_pad_type_0"), val = string("valid")]; + tensor hidden_states_297_strides_0 = const()[name = string("hidden_states_297_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_297_pad_0 = const()[name = string("hidden_states_297_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_297_dilations_0 = const()[name = string("hidden_states_297_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_297_groups_0 = const()[name = string("hidden_states_297_groups_0"), val = int32(1)]; + tensor hidden_states_297_cast_fp16 = conv(dilations = hidden_states_297_dilations_0, groups = hidden_states_297_groups_0, pad = hidden_states_297_pad_0, pad_type = hidden_states_297_pad_type_0, strides = hidden_states_297_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_247_cast_fp16)[name = string("hidden_states_297_cast_fp16")]; + tensor inputs_241_cast_fp16 = add(x = inputs_239_cast_fp16, y = hidden_states_297_cast_fp16)[name = string("inputs_241_cast_fp16")]; + tensor obj_259_begin_0 = const()[name = string("obj_259_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_259_end_0 = const()[name = string("obj_259_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_259_end_mask_0 = const()[name = string("obj_259_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_259_cast_fp16 = slice_by_index(begin = obj_259_begin_0, end = obj_259_end_0, end_mask = obj_259_end_mask_0, x = key_caches_11_cast_fp16)[name = string("obj_259_cast_fp16")]; + tensor obj_261_begin_0 = const()[name = string("obj_261_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_261_end_0 = const()[name = string("obj_261_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_261_end_mask_0 = const()[name = string("obj_261_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_261_cast_fp16 = slice_by_index(begin = obj_261_begin_0, end = obj_261_end_0, end_mask = obj_261_end_mask_0, x = value_caches_11_cast_fp16)[name = string("obj_261_cast_fp16")]; + int32 var_8835 = const()[name = string("op_8835"), val = int32(3)]; + int32 var_8845 = const()[name = string("op_8845"), val = int32(-2)]; + tensor inputs_sq_241_cast_fp16 = mul(x = inputs_241_cast_fp16, y = inputs_241_cast_fp16)[name = string("inputs_sq_241_cast_fp16")]; + tensor variance_241_axes_0 = const()[name = string("variance_241_axes_0"), val = tensor([1])]; + bool variance_241_keep_dims_0 = const()[name = string("variance_241_keep_dims_0"), val = bool(true)]; + tensor variance_241_cast_fp16 = reduce_mean(axes = variance_241_axes_0, keep_dims = variance_241_keep_dims_0, x = inputs_sq_241_cast_fp16)[name = string("variance_241_cast_fp16")]; + fp16 var_8859_to_fp16 = const()[name = string("op_8859_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8860_cast_fp16 = add(x = variance_241_cast_fp16, y = var_8859_to_fp16)[name = string("op_8860_cast_fp16")]; + fp32 var_8861_epsilon_0 = const()[name = string("op_8861_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8861_cast_fp16 = rsqrt(epsilon = var_8861_epsilon_0, x = var_8860_cast_fp16)[name = string("op_8861_cast_fp16")]; + tensor hidden_states_299_cast_fp16 = mul(x = inputs_241_cast_fp16, y = var_8861_cast_fp16)[name = string("hidden_states_299_cast_fp16")]; + tensor obj_257_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_299_cast_fp16)[name = string("obj_257_cast_fp16")]; + string query_175_pad_type_0 = const()[name = string("query_175_pad_type_0"), val = string("valid")]; + tensor query_175_strides_0 = const()[name = string("query_175_strides_0"), val = tensor([1, 1])]; + tensor query_175_pad_0 = const()[name = string("query_175_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_175_dilations_0 = const()[name = string("query_175_dilations_0"), val = tensor([1, 1])]; + int32 query_175_groups_0 = const()[name = string("query_175_groups_0"), val = int32(1)]; + tensor query_175_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_175_dilations_0, groups = query_175_groups_0, pad = query_175_pad_0, pad_type = query_175_pad_type_0, strides = query_175_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_257_cast_fp16)[name = string("query_175_cast_fp16")]; + string current_key_117_pad_type_0 = const()[name = string("current_key_117_pad_type_0"), val = string("valid")]; + tensor current_key_117_strides_0 = const()[name = string("current_key_117_strides_0"), val = tensor([1, 1])]; + tensor current_key_117_pad_0 = const()[name = string("current_key_117_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_117_dilations_0 = const()[name = string("current_key_117_dilations_0"), val = tensor([1, 1])]; + int32 current_key_117_groups_0 = const()[name = string("current_key_117_groups_0"), val = int32(1)]; + tensor current_key_117_cast_fp16 = conv(dilations = current_key_117_dilations_0, groups = current_key_117_groups_0, pad = current_key_117_pad_0, pad_type = current_key_117_pad_type_0, strides = current_key_117_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_257_cast_fp16)[name = string("current_key_117_cast_fp16")]; + string current_value_59_pad_type_0 = const()[name = string("current_value_59_pad_type_0"), val = string("valid")]; + tensor current_value_59_strides_0 = const()[name = string("current_value_59_strides_0"), val = tensor([1, 1])]; + tensor current_value_59_pad_0 = const()[name = string("current_value_59_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_59_dilations_0 = const()[name = string("current_value_59_dilations_0"), val = tensor([1, 1])]; + int32 current_value_59_groups_0 = const()[name = string("current_value_59_groups_0"), val = int32(1)]; + tensor current_value_59_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_59_dilations_0, groups = current_value_59_groups_0, pad = current_value_59_pad_0, pad_type = current_value_59_pad_type_0, strides = current_value_59_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_257_cast_fp16)[name = string("current_value_59_cast_fp16")]; + tensor var_8898 = const()[name = string("op_8898"), val = tensor([16, 128, 1, 1])]; + tensor inputs_243_cast_fp16 = reshape(shape = var_8898, x = query_175_cast_fp16)[name = string("inputs_243_cast_fp16")]; + tensor inputs_sq_243_cast_fp16 = mul(x = inputs_243_cast_fp16, y = inputs_243_cast_fp16)[name = string("inputs_sq_243_cast_fp16")]; + tensor variance_243_axes_0 = const()[name = string("variance_243_axes_0"), val = tensor([1])]; + bool variance_243_keep_dims_0 = const()[name = string("variance_243_keep_dims_0"), val = bool(true)]; + tensor variance_243_cast_fp16 = reduce_mean(axes = variance_243_axes_0, keep_dims = variance_243_keep_dims_0, x = inputs_sq_243_cast_fp16)[name = string("variance_243_cast_fp16")]; + fp16 var_8904_to_fp16 = const()[name = string("op_8904_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8905_cast_fp16 = add(x = variance_243_cast_fp16, y = var_8904_to_fp16)[name = string("op_8905_cast_fp16")]; + fp32 var_8906_epsilon_0 = const()[name = string("op_8906_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8906_cast_fp16 = rsqrt(epsilon = var_8906_epsilon_0, x = var_8905_cast_fp16)[name = string("op_8906_cast_fp16")]; + tensor hidden_states_301_cast_fp16 = mul(x = inputs_243_cast_fp16, y = var_8906_cast_fp16)[name = string("hidden_states_301_cast_fp16")]; + tensor query_normed_59_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_301_cast_fp16)[name = string("query_normed_59_cast_fp16")]; + tensor var_8914 = const()[name = string("op_8914"), val = tensor([8, 128, 1, 1])]; + tensor inputs_245_cast_fp16 = reshape(shape = var_8914, x = current_key_117_cast_fp16)[name = string("inputs_245_cast_fp16")]; + tensor inputs_sq_245_cast_fp16 = mul(x = inputs_245_cast_fp16, y = inputs_245_cast_fp16)[name = string("inputs_sq_245_cast_fp16")]; + tensor variance_245_axes_0 = const()[name = string("variance_245_axes_0"), val = tensor([1])]; + bool variance_245_keep_dims_0 = const()[name = string("variance_245_keep_dims_0"), val = bool(true)]; + tensor variance_245_cast_fp16 = reduce_mean(axes = variance_245_axes_0, keep_dims = variance_245_keep_dims_0, x = inputs_sq_245_cast_fp16)[name = string("variance_245_cast_fp16")]; + fp16 var_8920_to_fp16 = const()[name = string("op_8920_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_8921_cast_fp16 = add(x = variance_245_cast_fp16, y = var_8920_to_fp16)[name = string("op_8921_cast_fp16")]; + fp32 var_8922_epsilon_0 = const()[name = string("op_8922_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_8922_cast_fp16 = rsqrt(epsilon = var_8922_epsilon_0, x = var_8921_cast_fp16)[name = string("op_8922_cast_fp16")]; + tensor hidden_states_303_cast_fp16 = mul(x = inputs_245_cast_fp16, y = var_8922_cast_fp16)[name = string("hidden_states_303_cast_fp16")]; + tensor current_key_normed_59_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_303_cast_fp16)[name = string("current_key_normed_59_cast_fp16")]; + tensor var_8940 = const()[name = string("op_8940"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_233_cast_fp16 = reshape(shape = var_8940, x = query_normed_59_cast_fp16)[name = string("mh_q_233_cast_fp16")]; + tensor var_8942 = const()[name = string("op_8942"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_233_cast_fp16 = reshape(shape = var_8942, x = current_key_normed_59_cast_fp16)[name = string("mh_k_233_cast_fp16")]; + tensor var_8946_cast_fp16 = mul(x = mh_q_233_cast_fp16, y = cos_51_to_fp16)[name = string("op_8946_cast_fp16")]; + tensor var_8951_begin_0 = const()[name = string("op_8951_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8951_end_0 = const()[name = string("op_8951_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_8951_end_mask_0 = const()[name = string("op_8951_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_8951_cast_fp16 = slice_by_index(begin = var_8951_begin_0, end = var_8951_end_0, end_mask = var_8951_end_mask_0, x = mh_q_233_cast_fp16)[name = string("op_8951_cast_fp16")]; + tensor var_8957_begin_0 = const()[name = string("op_8957_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_8957_end_0 = const()[name = string("op_8957_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_8957_end_mask_0 = const()[name = string("op_8957_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_8957_cast_fp16 = slice_by_index(begin = var_8957_begin_0, end = var_8957_end_0, end_mask = var_8957_end_mask_0, x = mh_q_233_cast_fp16)[name = string("op_8957_cast_fp16")]; + fp16 const_599_promoted_to_fp16 = const()[name = string("const_599_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8959_cast_fp16 = mul(x = var_8957_cast_fp16, y = const_599_promoted_to_fp16)[name = string("op_8959_cast_fp16")]; + bool var_8961_interleave_0 = const()[name = string("op_8961_interleave_0"), val = bool(false)]; + tensor var_8961_cast_fp16 = concat(axis = var_8845, interleave = var_8961_interleave_0, values = (var_8959_cast_fp16, var_8951_cast_fp16))[name = string("op_8961_cast_fp16")]; + tensor var_8962_cast_fp16 = mul(x = var_8961_cast_fp16, y = sin_51_to_fp16)[name = string("op_8962_cast_fp16")]; + tensor mh_q_235_cast_fp16 = add(x = var_8946_cast_fp16, y = var_8962_cast_fp16)[name = string("mh_q_235_cast_fp16")]; + tensor var_8964_cast_fp16 = mul(x = mh_k_233_cast_fp16, y = cos_51_to_fp16)[name = string("op_8964_cast_fp16")]; + tensor var_8969_begin_0 = const()[name = string("op_8969_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_8969_end_0 = const()[name = string("op_8969_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_8969_end_mask_0 = const()[name = string("op_8969_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_8969_cast_fp16 = slice_by_index(begin = var_8969_begin_0, end = var_8969_end_0, end_mask = var_8969_end_mask_0, x = mh_k_233_cast_fp16)[name = string("op_8969_cast_fp16")]; + tensor var_8975_begin_0 = const()[name = string("op_8975_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_8975_end_0 = const()[name = string("op_8975_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_8975_end_mask_0 = const()[name = string("op_8975_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_8975_cast_fp16 = slice_by_index(begin = var_8975_begin_0, end = var_8975_end_0, end_mask = var_8975_end_mask_0, x = mh_k_233_cast_fp16)[name = string("op_8975_cast_fp16")]; + fp16 const_602_promoted_to_fp16 = const()[name = string("const_602_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8977_cast_fp16 = mul(x = var_8975_cast_fp16, y = const_602_promoted_to_fp16)[name = string("op_8977_cast_fp16")]; + bool var_8979_interleave_0 = const()[name = string("op_8979_interleave_0"), val = bool(false)]; + tensor var_8979_cast_fp16 = concat(axis = var_8845, interleave = var_8979_interleave_0, values = (var_8977_cast_fp16, var_8969_cast_fp16))[name = string("op_8979_cast_fp16")]; + tensor var_8980_cast_fp16 = mul(x = var_8979_cast_fp16, y = sin_51_to_fp16)[name = string("op_8980_cast_fp16")]; + tensor mh_k_235_cast_fp16 = add(x = var_8964_cast_fp16, y = var_8980_cast_fp16)[name = string("mh_k_235_cast_fp16")]; + tensor var_8984 = const()[name = string("op_8984"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_119_cast_fp16 = reshape(shape = var_8984, x = mh_k_235_cast_fp16)[name = string("current_key_119_cast_fp16")]; + tensor var_8990_to_fp16 = const()[name = string("op_8990_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194304)))]; + tensor var_8991_cast_fp16 = mul(x = obj_259_cast_fp16, y = var_8990_to_fp16)[name = string("op_8991_cast_fp16")]; + tensor var_8988_to_fp16 = const()[name = string("op_8988_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194432)))]; + tensor var_8992_cast_fp16 = mul(x = current_key_119_cast_fp16, y = var_8988_to_fp16)[name = string("op_8992_cast_fp16")]; + tensor key_119_cast_fp16 = add(x = var_8991_cast_fp16, y = var_8992_cast_fp16)[name = string("key_119_cast_fp16")]; + tensor var_8994_to_fp16 = const()[name = string("op_8994_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194304)))]; + tensor var_8995_cast_fp16 = mul(x = obj_261_cast_fp16, y = var_8994_to_fp16)[name = string("op_8995_cast_fp16")]; + tensor var_8996_cast_fp16 = mul(x = current_value_59_cast_fp16, y = var_8988_to_fp16)[name = string("op_8996_cast_fp16")]; + tensor value_59_cast_fp16 = add(x = var_8995_cast_fp16, y = var_8996_cast_fp16)[name = string("value_59_cast_fp16")]; + fp16 var_9003_to_fp16 = const()[name = string("op_9003_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_239_cast_fp16 = mul(x = mh_q_235_cast_fp16, y = var_9003_to_fp16)[name = string("mh_q_239_cast_fp16")]; + tensor var_9005 = const()[name = string("op_9005"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_237_cast_fp16 = reshape(shape = var_9005, x = key_119_cast_fp16)[name = string("mh_k_237_cast_fp16")]; + tensor var_9007 = const()[name = string("op_9007"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_117_cast_fp16 = reshape(shape = var_9007, x = value_59_cast_fp16)[name = string("mh_v_117_cast_fp16")]; + tensor transpose_116_perm_0 = const()[name = string("transpose_116_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_58_reps_0 = const()[name = string("tile_58_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_116_cast_fp16 = transpose(perm = transpose_116_perm_0, x = mh_k_237_cast_fp16)[name = string("transpose_305")]; + tensor tile_58_cast_fp16 = tile(reps = tile_58_reps_0, x = transpose_116_cast_fp16)[name = string("tile_58_cast_fp16")]; + tensor concat_144 = const()[name = string("concat_144"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_116_cast_fp16 = reshape(shape = concat_144, x = tile_58_cast_fp16)[name = string("reshape_116_cast_fp16")]; + tensor transpose_117_perm_0 = const()[name = string("transpose_117_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_145 = const()[name = string("concat_145"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_117_cast_fp16 = transpose(perm = transpose_117_perm_0, x = reshape_116_cast_fp16)[name = string("transpose_304")]; + tensor reshape_117_cast_fp16 = reshape(shape = concat_145, x = transpose_117_cast_fp16)[name = string("reshape_117_cast_fp16")]; + tensor transpose_118_perm_0 = const()[name = string("transpose_118_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_59_reps_0 = const()[name = string("tile_59_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_118_cast_fp16 = transpose(perm = transpose_118_perm_0, x = mh_v_117_cast_fp16)[name = string("transpose_303")]; + tensor tile_59_cast_fp16 = tile(reps = tile_59_reps_0, x = transpose_118_cast_fp16)[name = string("tile_59_cast_fp16")]; + tensor concat_146 = const()[name = string("concat_146"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_118_cast_fp16 = reshape(shape = concat_146, x = tile_59_cast_fp16)[name = string("reshape_118_cast_fp16")]; + tensor transpose_119_perm_0 = const()[name = string("transpose_119_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_147 = const()[name = string("concat_147"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_119_cast_fp16 = transpose(perm = transpose_119_perm_0, x = reshape_118_cast_fp16)[name = string("transpose_302")]; + tensor reshape_119_cast_fp16 = reshape(shape = concat_147, x = transpose_119_cast_fp16)[name = string("reshape_119_cast_fp16")]; + tensor transpose_433_perm_0 = const()[name = string("transpose_433_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_175_transpose_x_1 = const()[name = string("mh_w_175_transpose_x_1"), val = bool(true)]; + bool mh_w_175_transpose_y_1 = const()[name = string("mh_w_175_transpose_y_1"), val = bool(false)]; + tensor transpose_433_cast_fp16 = transpose(perm = transpose_433_perm_0, x = reshape_117_cast_fp16)[name = string("transpose_301")]; + tensor mh_w_175_cast_fp16 = matmul(transpose_x = mh_w_175_transpose_x_1, transpose_y = mh_w_175_transpose_y_1, x = mh_q_239_cast_fp16, y = transpose_433_cast_fp16)[name = string("mh_w_175_cast_fp16")]; + tensor var_9015_to_fp16 = const()[name = string("op_9015_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194560)))]; + tensor mh_w_177_cast_fp16 = add(x = mh_w_175_cast_fp16, y = var_9015_to_fp16)[name = string("mh_w_177_cast_fp16")]; + tensor mh_w_179_cast_fp16 = softmax(axis = var_8835, x = mh_w_177_cast_fp16)[name = string("mh_w_179_cast_fp16")]; + tensor transpose_434_perm_0 = const()[name = string("transpose_434_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_59_transpose_x_1 = const()[name = string("attn_59_transpose_x_1"), val = bool(false)]; + bool attn_59_transpose_y_1 = const()[name = string("attn_59_transpose_y_1"), val = bool(true)]; + tensor transpose_434_cast_fp16 = transpose(perm = transpose_434_perm_0, x = reshape_119_cast_fp16)[name = string("transpose_300")]; + tensor attn_59_cast_fp16 = matmul(transpose_x = attn_59_transpose_x_1, transpose_y = attn_59_transpose_y_1, x = transpose_434_cast_fp16, y = mh_w_179_cast_fp16)[name = string("attn_59_cast_fp16")]; + tensor var_9021 = const()[name = string("op_9021"), val = tensor([1, 2048, 1, 1])]; + tensor input_249_cast_fp16 = reshape(shape = var_9021, x = attn_59_cast_fp16)[name = string("input_249_cast_fp16")]; + string obj_263_pad_type_0 = const()[name = string("obj_263_pad_type_0"), val = string("valid")]; + tensor obj_263_strides_0 = const()[name = string("obj_263_strides_0"), val = tensor([1, 1])]; + tensor obj_263_pad_0 = const()[name = string("obj_263_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_263_dilations_0 = const()[name = string("obj_263_dilations_0"), val = tensor([1, 1])]; + int32 obj_263_groups_0 = const()[name = string("obj_263_groups_0"), val = int32(1)]; + tensor obj_263_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_263_dilations_0, groups = obj_263_groups_0, pad = obj_263_pad_0, pad_type = obj_263_pad_type_0, strides = obj_263_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_249_cast_fp16)[name = string("obj_263_cast_fp16")]; + tensor inputs_247_cast_fp16 = add(x = inputs_241_cast_fp16, y = obj_263_cast_fp16)[name = string("inputs_247_cast_fp16")]; + tensor inputs_sq_247_cast_fp16 = mul(x = inputs_247_cast_fp16, y = inputs_247_cast_fp16)[name = string("inputs_sq_247_cast_fp16")]; + tensor variance_247_axes_0 = const()[name = string("variance_247_axes_0"), val = tensor([1])]; + bool variance_247_keep_dims_0 = const()[name = string("variance_247_keep_dims_0"), val = bool(true)]; + tensor variance_247_cast_fp16 = reduce_mean(axes = variance_247_axes_0, keep_dims = variance_247_keep_dims_0, x = inputs_sq_247_cast_fp16)[name = string("variance_247_cast_fp16")]; + fp16 var_9039_to_fp16 = const()[name = string("op_9039_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9040_cast_fp16 = add(x = variance_247_cast_fp16, y = var_9039_to_fp16)[name = string("op_9040_cast_fp16")]; + fp32 var_9041_epsilon_0 = const()[name = string("op_9041_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9041_cast_fp16 = rsqrt(epsilon = var_9041_epsilon_0, x = var_9040_cast_fp16)[name = string("op_9041_cast_fp16")]; + tensor hidden_states_305_cast_fp16 = mul(x = inputs_247_cast_fp16, y = var_9041_cast_fp16)[name = string("hidden_states_305_cast_fp16")]; + tensor input_251_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_305_cast_fp16)[name = string("input_251_cast_fp16")]; + string input_253_pad_type_0 = const()[name = string("input_253_pad_type_0"), val = string("valid")]; + tensor input_253_strides_0 = const()[name = string("input_253_strides_0"), val = tensor([1, 1])]; + tensor input_253_pad_0 = const()[name = string("input_253_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_253_dilations_0 = const()[name = string("input_253_dilations_0"), val = tensor([1, 1])]; + int32 input_253_groups_0 = const()[name = string("input_253_groups_0"), val = int32(1)]; + tensor input_253_cast_fp16 = conv(dilations = input_253_dilations_0, groups = input_253_groups_0, pad = input_253_pad_0, pad_type = input_253_pad_type_0, strides = input_253_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_251_cast_fp16)[name = string("input_253_cast_fp16")]; + tensor var_9055_cast_fp16 = silu(x = input_253_cast_fp16)[name = string("op_9055_cast_fp16")]; + string var_9061_pad_type_0 = const()[name = string("op_9061_pad_type_0"), val = string("valid")]; + tensor var_9061_strides_0 = const()[name = string("op_9061_strides_0"), val = tensor([1, 1])]; + tensor var_9061_pad_0 = const()[name = string("op_9061_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9061_dilations_0 = const()[name = string("op_9061_dilations_0"), val = tensor([1, 1])]; + int32 var_9061_groups_0 = const()[name = string("op_9061_groups_0"), val = int32(1)]; + tensor var_9061_cast_fp16 = conv(dilations = var_9061_dilations_0, groups = var_9061_groups_0, pad = var_9061_pad_0, pad_type = var_9061_pad_type_0, strides = var_9061_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_251_cast_fp16)[name = string("op_9061_cast_fp16")]; + tensor input_255_cast_fp16 = mul(x = var_9055_cast_fp16, y = var_9061_cast_fp16)[name = string("input_255_cast_fp16")]; + string hidden_states_307_pad_type_0 = const()[name = string("hidden_states_307_pad_type_0"), val = string("valid")]; + tensor hidden_states_307_strides_0 = const()[name = string("hidden_states_307_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_307_pad_0 = const()[name = string("hidden_states_307_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_307_dilations_0 = const()[name = string("hidden_states_307_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_307_groups_0 = const()[name = string("hidden_states_307_groups_0"), val = int32(1)]; + tensor hidden_states_307_cast_fp16 = conv(dilations = hidden_states_307_dilations_0, groups = hidden_states_307_groups_0, pad = hidden_states_307_pad_0, pad_type = hidden_states_307_pad_type_0, strides = hidden_states_307_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_255_cast_fp16)[name = string("hidden_states_307_cast_fp16")]; + tensor inputs_249_cast_fp16 = add(x = inputs_247_cast_fp16, y = hidden_states_307_cast_fp16)[name = string("inputs_249_cast_fp16")]; + int32 var_9089 = const()[name = string("op_9089"), val = int32(1)]; + bool key_caches_13_interleave_0 = const()[name = string("key_caches_13_interleave_0"), val = bool(false)]; + tensor key_caches_13_cast_fp16 = concat(axis = var_9089, interleave = key_caches_13_interleave_0, values = (key_103_cast_fp16, key_107_cast_fp16, key_111_cast_fp16, key_115_cast_fp16, key_119_cast_fp16))[name = string("key_caches_13_cast_fp16")]; + int32 var_9092 = const()[name = string("op_9092"), val = int32(1)]; + bool value_caches_13_interleave_0 = const()[name = string("value_caches_13_interleave_0"), val = bool(false)]; + tensor value_caches_13_cast_fp16 = concat(axis = var_9092, interleave = value_caches_13_interleave_0, values = (value_51_cast_fp16, value_53_cast_fp16, value_55_cast_fp16, value_57_cast_fp16, value_59_cast_fp16))[name = string("value_caches_13_cast_fp16")]; + tensor inputs_sq_249_cast_fp16 = mul(x = inputs_249_cast_fp16, y = inputs_249_cast_fp16)[name = string("inputs_sq_249_cast_fp16")]; + tensor variance_249_axes_0 = const()[name = string("variance_249_axes_0"), val = tensor([1])]; + bool variance_249_keep_dims_0 = const()[name = string("variance_249_keep_dims_0"), val = bool(true)]; + tensor variance_249_cast_fp16 = reduce_mean(axes = variance_249_axes_0, keep_dims = variance_249_keep_dims_0, x = inputs_sq_249_cast_fp16)[name = string("variance_249_cast_fp16")]; + fp16 var_9102_to_fp16 = const()[name = string("op_9102_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9103_cast_fp16 = add(x = variance_249_cast_fp16, y = var_9102_to_fp16)[name = string("op_9103_cast_fp16")]; + fp32 var_9104_epsilon_0 = const()[name = string("op_9104_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9104_cast_fp16 = rsqrt(epsilon = var_9104_epsilon_0, x = var_9103_cast_fp16)[name = string("op_9104_cast_fp16")]; + tensor hidden_states_309_cast_fp16 = mul(x = inputs_249_cast_fp16, y = var_9104_cast_fp16)[name = string("hidden_states_309_cast_fp16")]; + tensor input_257_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_309_cast_fp16)[name = string("input_257_cast_fp16")]; + string logits_17_pad_type_0 = const()[name = string("logits_17_pad_type_0"), val = string("valid")]; + tensor logits_17_strides_0 = const()[name = string("logits_17_strides_0"), val = tensor([1, 1])]; + tensor logits_17_pad_0 = const()[name = string("logits_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_17_dilations_0 = const()[name = string("logits_17_dilations_0"), val = tensor([1, 1])]; + int32 logits_17_groups_0 = const()[name = string("logits_17_groups_0"), val = int32(1)]; + tensor lm_heads_4_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89197760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91294976))))[name = string("lm_heads_4_weight_to_fp16_palettized")]; + tensor logits_17_cast_fp16 = conv(dilations = logits_17_dilations_0, groups = logits_17_groups_0, pad = logits_17_pad_0, pad_type = logits_17_pad_type_0, strides = logits_17_strides_0, weight = lm_heads_4_weight_to_fp16_palettized, x = input_257_cast_fp16)[name = string("logits_17_cast_fp16")]; + tensor var_9122 = const()[name = string("op_9122"), val = tensor([1, 2048])]; + tensor logits_19_cast_fp16 = reshape(shape = var_9122, x = logits_17_cast_fp16)[name = string("logits_19_cast_fp16")]; + tensor scaled_logits_9_cast_fp16 = real_div(x = logits_19_cast_fp16, y = temperature)[name = string("scaled_logits_9_cast_fp16")]; + int32 var_9132 = const()[name = string("op_9132"), val = int32(100)]; + int32 top_values_9_axis_0 = const()[name = string("top_values_9_axis_0"), val = int32(1)]; + bool top_values_9_ascending_0 = const()[name = string("top_values_9_ascending_0"), val = bool(false)]; + bool top_values_9_sort_0 = const()[name = string("top_values_9_sort_0"), val = bool(true)]; + bool top_values_9_return_indices_0 = const()[name = string("top_values_9_return_indices_0"), val = bool(true)]; + string top_values_9_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_9_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_9_cast_fp16_cast_uint16_0, tensor top_values_9_cast_fp16_cast_uint16_1 = topk(ascending = top_values_9_ascending_0, axis = top_values_9_axis_0, k = var_9132, output_indices_dtype = top_values_9_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_9_return_indices_0, sort = top_values_9_sort_0, x = scaled_logits_9_cast_fp16)[name = string("top_values_9_cast_fp16_cast_uint16")]; + tensor var_9138_cast_fp16 = mul(x = top_values_9_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_9138_cast_fp16")]; + tensor var_9142_cast_fp16 = add(x = var_9138_cast_fp16, y = var_3001_cast_fp16)[name = string("op_9142_cast_fp16")]; + tensor reduce_min_4_axes_0 = const()[name = string("reduce_min_4_axes_0"), val = tensor([1])]; + bool reduce_min_4_keep_dims_0 = const()[name = string("reduce_min_4_keep_dims_0"), val = bool(true)]; + tensor reduce_min_4_cast_fp16 = reduce_min(axes = reduce_min_4_axes_0, keep_dims = reduce_min_4_keep_dims_0, x = var_9142_cast_fp16)[name = string("reduce_min_4_cast_fp16")]; + tensor var_9145_cast_fp16 = greater_equal(x = scaled_logits_9_cast_fp16, y = reduce_min_4_cast_fp16)[name = string("op_9145_cast_fp16")]; + fp16 var_9146_value_0_to_fp16 = const()[name = string("op_9146_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_9146_cast_fp16 = fill_like(ref_tensor = scaled_logits_9_cast_fp16, value = var_9146_value_0_to_fp16)[name = string("op_9146_cast_fp16")]; + tensor masked_logits_9_cast_fp16 = select(a = scaled_logits_9_cast_fp16, b = var_9146_cast_fp16, cond = var_9145_cast_fp16)[name = string("masked_logits_9_cast_fp16")]; + tensor var_9150_begin_0 = const()[name = string("op_9150_begin_0"), val = tensor([4, 0])]; + tensor var_9150_end_0 = const()[name = string("op_9150_end_0"), val = tensor([5, 2048])]; + tensor var_9150_end_mask_0 = const()[name = string("op_9150_end_mask_0"), val = tensor([false, true])]; + tensor var_9150_squeeze_mask_0 = const()[name = string("op_9150_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_9150_cast_fp16 = slice_by_index(begin = var_9150_begin_0, end = var_9150_end_0, end_mask = var_9150_end_mask_0, squeeze_mask = var_9150_squeeze_mask_0, x = gumbel)[name = string("op_9150_cast_fp16")]; + tensor var_9153 = const()[name = string("op_9153"), val = tensor([1, 2048])]; + tensor var_9154_cast_fp16 = reshape(shape = var_9153, x = var_9150_cast_fp16)[name = string("op_9154_cast_fp16")]; + tensor noisy_logits_9_cast_fp16 = add(x = masked_logits_9_cast_fp16, y = var_9154_cast_fp16)[name = string("noisy_logits_9_cast_fp16")]; + int32 code_9_axis_0 = const()[name = string("code_9_axis_0"), val = int32(1)]; + bool code_9_keep_dims_0 = const()[name = string("code_9_keep_dims_0"), val = bool(false)]; + string code_9_output_dtype_0 = const()[name = string("code_9_output_dtype_0"), val = string("int32")]; + tensor code_9_cast_fp16 = reduce_argmax(axis = code_9_axis_0, keep_dims = code_9_keep_dims_0, output_dtype = code_9_output_dtype_0, x = noisy_logits_9_cast_fp16)[name = string("code_9_cast_fp16")]; + int32 var_9165 = const()[name = string("op_9165"), val = int32(8192)]; + tensor input_259 = add(x = code_9_cast_fp16, y = var_9165)[name = string("input_259")]; + int32 code_embed_17_axis_0 = const()[name = string("code_embed_17_axis_0"), val = int32(0)]; + int32 code_embed_17_batch_dims_0 = const()[name = string("code_embed_17_batch_dims_0"), val = int32(0)]; + bool code_embed_17_validate_indices_0 = const()[name = string("code_embed_17_validate_indices_0"), val = bool(false)]; + string input_259_to_uint16_dtype_0 = const()[name = string("input_259_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_259_to_uint16 = cast(dtype = input_259_to_uint16_dtype_0, x = input_259)[name = string("cast_10")]; + tensor code_embed_17_cast_fp16_cast_uint16 = gather(axis = code_embed_17_axis_0, batch_dims = code_embed_17_batch_dims_0, indices = input_259_to_uint16, validate_indices = code_embed_17_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_17_cast_fp16_cast_uint16")]; + tensor var_9169 = const()[name = string("op_9169"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_19_cast_fp16 = reshape(shape = var_9169, x = code_embed_17_cast_fp16_cast_uint16)[name = string("code_embed_19_cast_fp16")]; + tensor embed_sum_11_cast_fp16 = add(x = embed_sum_9_cast_fp16, y = code_embed_19_cast_fp16)[name = string("embed_sum_11_cast_fp16")]; + string inputs_251_pad_type_0 = const()[name = string("inputs_251_pad_type_0"), val = string("valid")]; + tensor inputs_251_strides_0 = const()[name = string("inputs_251_strides_0"), val = tensor([1, 1])]; + tensor inputs_251_pad_0 = const()[name = string("inputs_251_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_251_dilations_0 = const()[name = string("inputs_251_dilations_0"), val = tensor([1, 1])]; + int32 inputs_251_groups_0 = const()[name = string("inputs_251_groups_0"), val = int32(1)]; + tensor inputs_251_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_251_dilations_0, groups = inputs_251_groups_0, pad = inputs_251_pad_0, pad_type = inputs_251_pad_type_0, strides = inputs_251_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_19_cast_fp16)[name = string("inputs_251_cast_fp16")]; + tensor obj_267_begin_0 = const()[name = string("obj_267_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_267_end_0 = const()[name = string("obj_267_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_267_end_mask_0 = const()[name = string("obj_267_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_267_cast_fp16 = slice_by_index(begin = obj_267_begin_0, end = obj_267_end_0, end_mask = obj_267_end_mask_0, x = key_caches_13_cast_fp16)[name = string("obj_267_cast_fp16")]; + tensor obj_269_begin_0 = const()[name = string("obj_269_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_269_end_0 = const()[name = string("obj_269_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_269_end_mask_0 = const()[name = string("obj_269_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_269_cast_fp16 = slice_by_index(begin = obj_269_begin_0, end = obj_269_end_0, end_mask = obj_269_end_mask_0, x = value_caches_13_cast_fp16)[name = string("obj_269_cast_fp16")]; + int32 var_9274 = const()[name = string("op_9274"), val = int32(3)]; + int32 var_9284 = const()[name = string("op_9284"), val = int32(-2)]; + tensor inputs_sq_251_cast_fp16 = mul(x = inputs_251_cast_fp16, y = inputs_251_cast_fp16)[name = string("inputs_sq_251_cast_fp16")]; + tensor variance_251_axes_0 = const()[name = string("variance_251_axes_0"), val = tensor([1])]; + bool variance_251_keep_dims_0 = const()[name = string("variance_251_keep_dims_0"), val = bool(true)]; + tensor variance_251_cast_fp16 = reduce_mean(axes = variance_251_axes_0, keep_dims = variance_251_keep_dims_0, x = inputs_sq_251_cast_fp16)[name = string("variance_251_cast_fp16")]; + fp16 var_9298_to_fp16 = const()[name = string("op_9298_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9299_cast_fp16 = add(x = variance_251_cast_fp16, y = var_9298_to_fp16)[name = string("op_9299_cast_fp16")]; + fp32 var_9300_epsilon_0 = const()[name = string("op_9300_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9300_cast_fp16 = rsqrt(epsilon = var_9300_epsilon_0, x = var_9299_cast_fp16)[name = string("op_9300_cast_fp16")]; + tensor hidden_states_311_cast_fp16 = mul(x = inputs_251_cast_fp16, y = var_9300_cast_fp16)[name = string("hidden_states_311_cast_fp16")]; + tensor obj_265_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_311_cast_fp16)[name = string("obj_265_cast_fp16")]; + string query_181_pad_type_0 = const()[name = string("query_181_pad_type_0"), val = string("valid")]; + tensor query_181_strides_0 = const()[name = string("query_181_strides_0"), val = tensor([1, 1])]; + tensor query_181_pad_0 = const()[name = string("query_181_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_181_dilations_0 = const()[name = string("query_181_dilations_0"), val = tensor([1, 1])]; + int32 query_181_groups_0 = const()[name = string("query_181_groups_0"), val = int32(1)]; + tensor query_181_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_181_dilations_0, groups = query_181_groups_0, pad = query_181_pad_0, pad_type = query_181_pad_type_0, strides = query_181_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_265_cast_fp16)[name = string("query_181_cast_fp16")]; + string current_key_121_pad_type_0 = const()[name = string("current_key_121_pad_type_0"), val = string("valid")]; + tensor current_key_121_strides_0 = const()[name = string("current_key_121_strides_0"), val = tensor([1, 1])]; + tensor current_key_121_pad_0 = const()[name = string("current_key_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_121_dilations_0 = const()[name = string("current_key_121_dilations_0"), val = tensor([1, 1])]; + int32 current_key_121_groups_0 = const()[name = string("current_key_121_groups_0"), val = int32(1)]; + tensor current_key_121_cast_fp16 = conv(dilations = current_key_121_dilations_0, groups = current_key_121_groups_0, pad = current_key_121_pad_0, pad_type = current_key_121_pad_type_0, strides = current_key_121_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_265_cast_fp16)[name = string("current_key_121_cast_fp16")]; + string current_value_61_pad_type_0 = const()[name = string("current_value_61_pad_type_0"), val = string("valid")]; + tensor current_value_61_strides_0 = const()[name = string("current_value_61_strides_0"), val = tensor([1, 1])]; + tensor current_value_61_pad_0 = const()[name = string("current_value_61_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_61_dilations_0 = const()[name = string("current_value_61_dilations_0"), val = tensor([1, 1])]; + int32 current_value_61_groups_0 = const()[name = string("current_value_61_groups_0"), val = int32(1)]; + tensor current_value_61_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_61_dilations_0, groups = current_value_61_groups_0, pad = current_value_61_pad_0, pad_type = current_value_61_pad_type_0, strides = current_value_61_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_265_cast_fp16)[name = string("current_value_61_cast_fp16")]; + tensor var_9337 = const()[name = string("op_9337"), val = tensor([16, 128, 1, 1])]; + tensor inputs_253_cast_fp16 = reshape(shape = var_9337, x = query_181_cast_fp16)[name = string("inputs_253_cast_fp16")]; + tensor inputs_sq_253_cast_fp16 = mul(x = inputs_253_cast_fp16, y = inputs_253_cast_fp16)[name = string("inputs_sq_253_cast_fp16")]; + tensor variance_253_axes_0 = const()[name = string("variance_253_axes_0"), val = tensor([1])]; + bool variance_253_keep_dims_0 = const()[name = string("variance_253_keep_dims_0"), val = bool(true)]; + tensor variance_253_cast_fp16 = reduce_mean(axes = variance_253_axes_0, keep_dims = variance_253_keep_dims_0, x = inputs_sq_253_cast_fp16)[name = string("variance_253_cast_fp16")]; + fp16 var_9343_to_fp16 = const()[name = string("op_9343_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9344_cast_fp16 = add(x = variance_253_cast_fp16, y = var_9343_to_fp16)[name = string("op_9344_cast_fp16")]; + fp32 var_9345_epsilon_0 = const()[name = string("op_9345_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9345_cast_fp16 = rsqrt(epsilon = var_9345_epsilon_0, x = var_9344_cast_fp16)[name = string("op_9345_cast_fp16")]; + tensor hidden_states_313_cast_fp16 = mul(x = inputs_253_cast_fp16, y = var_9345_cast_fp16)[name = string("hidden_states_313_cast_fp16")]; + tensor query_normed_61_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_313_cast_fp16)[name = string("query_normed_61_cast_fp16")]; + tensor var_9353 = const()[name = string("op_9353"), val = tensor([8, 128, 1, 1])]; + tensor inputs_255_cast_fp16 = reshape(shape = var_9353, x = current_key_121_cast_fp16)[name = string("inputs_255_cast_fp16")]; + tensor inputs_sq_255_cast_fp16 = mul(x = inputs_255_cast_fp16, y = inputs_255_cast_fp16)[name = string("inputs_sq_255_cast_fp16")]; + tensor variance_255_axes_0 = const()[name = string("variance_255_axes_0"), val = tensor([1])]; + bool variance_255_keep_dims_0 = const()[name = string("variance_255_keep_dims_0"), val = bool(true)]; + tensor variance_255_cast_fp16 = reduce_mean(axes = variance_255_axes_0, keep_dims = variance_255_keep_dims_0, x = inputs_sq_255_cast_fp16)[name = string("variance_255_cast_fp16")]; + fp16 var_9359_to_fp16 = const()[name = string("op_9359_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9360_cast_fp16 = add(x = variance_255_cast_fp16, y = var_9359_to_fp16)[name = string("op_9360_cast_fp16")]; + fp32 var_9361_epsilon_0 = const()[name = string("op_9361_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9361_cast_fp16 = rsqrt(epsilon = var_9361_epsilon_0, x = var_9360_cast_fp16)[name = string("op_9361_cast_fp16")]; + tensor hidden_states_315_cast_fp16 = mul(x = inputs_255_cast_fp16, y = var_9361_cast_fp16)[name = string("hidden_states_315_cast_fp16")]; + tensor current_key_normed_61_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_315_cast_fp16)[name = string("current_key_normed_61_cast_fp16")]; + tensor var_9379 = const()[name = string("op_9379"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_241_cast_fp16 = reshape(shape = var_9379, x = query_normed_61_cast_fp16)[name = string("mh_q_241_cast_fp16")]; + tensor var_9381 = const()[name = string("op_9381"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_241_cast_fp16 = reshape(shape = var_9381, x = current_key_normed_61_cast_fp16)[name = string("mh_k_241_cast_fp16")]; + tensor cos_61_to_fp16 = const()[name = string("cos_61_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175194688)))]; + tensor var_9385_cast_fp16 = mul(x = mh_q_241_cast_fp16, y = cos_61_to_fp16)[name = string("op_9385_cast_fp16")]; + tensor var_9390_begin_0 = const()[name = string("op_9390_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9390_end_0 = const()[name = string("op_9390_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_9390_end_mask_0 = const()[name = string("op_9390_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_9390_cast_fp16 = slice_by_index(begin = var_9390_begin_0, end = var_9390_end_0, end_mask = var_9390_end_mask_0, x = mh_q_241_cast_fp16)[name = string("op_9390_cast_fp16")]; + tensor var_9396_begin_0 = const()[name = string("op_9396_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_9396_end_0 = const()[name = string("op_9396_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_9396_end_mask_0 = const()[name = string("op_9396_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_9396_cast_fp16 = slice_by_index(begin = var_9396_begin_0, end = var_9396_end_0, end_mask = var_9396_end_mask_0, x = mh_q_241_cast_fp16)[name = string("op_9396_cast_fp16")]; + fp16 const_620_promoted_to_fp16 = const()[name = string("const_620_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9398_cast_fp16 = mul(x = var_9396_cast_fp16, y = const_620_promoted_to_fp16)[name = string("op_9398_cast_fp16")]; + bool var_9400_interleave_0 = const()[name = string("op_9400_interleave_0"), val = bool(false)]; + tensor var_9400_cast_fp16 = concat(axis = var_9284, interleave = var_9400_interleave_0, values = (var_9398_cast_fp16, var_9390_cast_fp16))[name = string("op_9400_cast_fp16")]; + tensor sin_61_to_fp16 = const()[name = string("sin_61_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195008)))]; + tensor var_9401_cast_fp16 = mul(x = var_9400_cast_fp16, y = sin_61_to_fp16)[name = string("op_9401_cast_fp16")]; + tensor mh_q_243_cast_fp16 = add(x = var_9385_cast_fp16, y = var_9401_cast_fp16)[name = string("mh_q_243_cast_fp16")]; + tensor var_9403_cast_fp16 = mul(x = mh_k_241_cast_fp16, y = cos_61_to_fp16)[name = string("op_9403_cast_fp16")]; + tensor var_9408_begin_0 = const()[name = string("op_9408_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9408_end_0 = const()[name = string("op_9408_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_9408_end_mask_0 = const()[name = string("op_9408_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_9408_cast_fp16 = slice_by_index(begin = var_9408_begin_0, end = var_9408_end_0, end_mask = var_9408_end_mask_0, x = mh_k_241_cast_fp16)[name = string("op_9408_cast_fp16")]; + tensor var_9414_begin_0 = const()[name = string("op_9414_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_9414_end_0 = const()[name = string("op_9414_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_9414_end_mask_0 = const()[name = string("op_9414_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_9414_cast_fp16 = slice_by_index(begin = var_9414_begin_0, end = var_9414_end_0, end_mask = var_9414_end_mask_0, x = mh_k_241_cast_fp16)[name = string("op_9414_cast_fp16")]; + fp16 const_623_promoted_to_fp16 = const()[name = string("const_623_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9416_cast_fp16 = mul(x = var_9414_cast_fp16, y = const_623_promoted_to_fp16)[name = string("op_9416_cast_fp16")]; + bool var_9418_interleave_0 = const()[name = string("op_9418_interleave_0"), val = bool(false)]; + tensor var_9418_cast_fp16 = concat(axis = var_9284, interleave = var_9418_interleave_0, values = (var_9416_cast_fp16, var_9408_cast_fp16))[name = string("op_9418_cast_fp16")]; + tensor var_9419_cast_fp16 = mul(x = var_9418_cast_fp16, y = sin_61_to_fp16)[name = string("op_9419_cast_fp16")]; + tensor mh_k_243_cast_fp16 = add(x = var_9403_cast_fp16, y = var_9419_cast_fp16)[name = string("mh_k_243_cast_fp16")]; + tensor var_9423 = const()[name = string("op_9423"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_123_cast_fp16 = reshape(shape = var_9423, x = mh_k_243_cast_fp16)[name = string("current_key_123_cast_fp16")]; + tensor var_9429_to_fp16 = const()[name = string("op_9429_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195328)))]; + tensor var_9430_cast_fp16 = mul(x = obj_267_cast_fp16, y = var_9429_to_fp16)[name = string("op_9430_cast_fp16")]; + tensor var_9427_to_fp16 = const()[name = string("op_9427_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195456)))]; + tensor var_9431_cast_fp16 = mul(x = current_key_123_cast_fp16, y = var_9427_to_fp16)[name = string("op_9431_cast_fp16")]; + tensor key_123_cast_fp16 = add(x = var_9430_cast_fp16, y = var_9431_cast_fp16)[name = string("key_123_cast_fp16")]; + tensor var_9433_to_fp16 = const()[name = string("op_9433_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195328)))]; + tensor var_9434_cast_fp16 = mul(x = obj_269_cast_fp16, y = var_9433_to_fp16)[name = string("op_9434_cast_fp16")]; + tensor var_9435_cast_fp16 = mul(x = current_value_61_cast_fp16, y = var_9427_to_fp16)[name = string("op_9435_cast_fp16")]; + tensor value_61_cast_fp16 = add(x = var_9434_cast_fp16, y = var_9435_cast_fp16)[name = string("value_61_cast_fp16")]; + fp16 var_9442_to_fp16 = const()[name = string("op_9442_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_247_cast_fp16 = mul(x = mh_q_243_cast_fp16, y = var_9442_to_fp16)[name = string("mh_q_247_cast_fp16")]; + tensor var_9444 = const()[name = string("op_9444"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_245_cast_fp16 = reshape(shape = var_9444, x = key_123_cast_fp16)[name = string("mh_k_245_cast_fp16")]; + tensor var_9446 = const()[name = string("op_9446"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_121_cast_fp16 = reshape(shape = var_9446, x = value_61_cast_fp16)[name = string("mh_v_121_cast_fp16")]; + tensor transpose_120_perm_0 = const()[name = string("transpose_120_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_60_reps_0 = const()[name = string("tile_60_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_120_cast_fp16 = transpose(perm = transpose_120_perm_0, x = mh_k_245_cast_fp16)[name = string("transpose_299")]; + tensor tile_60_cast_fp16 = tile(reps = tile_60_reps_0, x = transpose_120_cast_fp16)[name = string("tile_60_cast_fp16")]; + tensor concat_153 = const()[name = string("concat_153"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_120_cast_fp16 = reshape(shape = concat_153, x = tile_60_cast_fp16)[name = string("reshape_120_cast_fp16")]; + tensor transpose_121_perm_0 = const()[name = string("transpose_121_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_154 = const()[name = string("concat_154"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_121_cast_fp16 = transpose(perm = transpose_121_perm_0, x = reshape_120_cast_fp16)[name = string("transpose_298")]; + tensor reshape_121_cast_fp16 = reshape(shape = concat_154, x = transpose_121_cast_fp16)[name = string("reshape_121_cast_fp16")]; + tensor transpose_122_perm_0 = const()[name = string("transpose_122_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_61_reps_0 = const()[name = string("tile_61_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_122_cast_fp16 = transpose(perm = transpose_122_perm_0, x = mh_v_121_cast_fp16)[name = string("transpose_297")]; + tensor tile_61_cast_fp16 = tile(reps = tile_61_reps_0, x = transpose_122_cast_fp16)[name = string("tile_61_cast_fp16")]; + tensor concat_155 = const()[name = string("concat_155"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_122_cast_fp16 = reshape(shape = concat_155, x = tile_61_cast_fp16)[name = string("reshape_122_cast_fp16")]; + tensor transpose_123_perm_0 = const()[name = string("transpose_123_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_156 = const()[name = string("concat_156"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_123_cast_fp16 = transpose(perm = transpose_123_perm_0, x = reshape_122_cast_fp16)[name = string("transpose_296")]; + tensor reshape_123_cast_fp16 = reshape(shape = concat_156, x = transpose_123_cast_fp16)[name = string("reshape_123_cast_fp16")]; + tensor transpose_437_perm_0 = const()[name = string("transpose_437_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_181_transpose_x_1 = const()[name = string("mh_w_181_transpose_x_1"), val = bool(true)]; + bool mh_w_181_transpose_y_1 = const()[name = string("mh_w_181_transpose_y_1"), val = bool(false)]; + tensor transpose_437_cast_fp16 = transpose(perm = transpose_437_perm_0, x = reshape_121_cast_fp16)[name = string("transpose_295")]; + tensor mh_w_181_cast_fp16 = matmul(transpose_x = mh_w_181_transpose_x_1, transpose_y = mh_w_181_transpose_y_1, x = mh_q_247_cast_fp16, y = transpose_437_cast_fp16)[name = string("mh_w_181_cast_fp16")]; + tensor var_9454_to_fp16 = const()[name = string("op_9454_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195584)))]; + tensor mh_w_183_cast_fp16 = add(x = mh_w_181_cast_fp16, y = var_9454_to_fp16)[name = string("mh_w_183_cast_fp16")]; + tensor mh_w_185_cast_fp16 = softmax(axis = var_9274, x = mh_w_183_cast_fp16)[name = string("mh_w_185_cast_fp16")]; + tensor transpose_438_perm_0 = const()[name = string("transpose_438_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_61_transpose_x_1 = const()[name = string("attn_61_transpose_x_1"), val = bool(false)]; + bool attn_61_transpose_y_1 = const()[name = string("attn_61_transpose_y_1"), val = bool(true)]; + tensor transpose_438_cast_fp16 = transpose(perm = transpose_438_perm_0, x = reshape_123_cast_fp16)[name = string("transpose_294")]; + tensor attn_61_cast_fp16 = matmul(transpose_x = attn_61_transpose_x_1, transpose_y = attn_61_transpose_y_1, x = transpose_438_cast_fp16, y = mh_w_185_cast_fp16)[name = string("attn_61_cast_fp16")]; + tensor var_9460 = const()[name = string("op_9460"), val = tensor([1, 2048, 1, 1])]; + tensor input_261_cast_fp16 = reshape(shape = var_9460, x = attn_61_cast_fp16)[name = string("input_261_cast_fp16")]; + string obj_275_pad_type_0 = const()[name = string("obj_275_pad_type_0"), val = string("valid")]; + tensor obj_275_strides_0 = const()[name = string("obj_275_strides_0"), val = tensor([1, 1])]; + tensor obj_275_pad_0 = const()[name = string("obj_275_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_275_dilations_0 = const()[name = string("obj_275_dilations_0"), val = tensor([1, 1])]; + int32 obj_275_groups_0 = const()[name = string("obj_275_groups_0"), val = int32(1)]; + tensor obj_275_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_275_dilations_0, groups = obj_275_groups_0, pad = obj_275_pad_0, pad_type = obj_275_pad_type_0, strides = obj_275_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_261_cast_fp16)[name = string("obj_275_cast_fp16")]; + tensor inputs_257_cast_fp16 = add(x = inputs_251_cast_fp16, y = obj_275_cast_fp16)[name = string("inputs_257_cast_fp16")]; + tensor inputs_sq_257_cast_fp16 = mul(x = inputs_257_cast_fp16, y = inputs_257_cast_fp16)[name = string("inputs_sq_257_cast_fp16")]; + tensor variance_257_axes_0 = const()[name = string("variance_257_axes_0"), val = tensor([1])]; + bool variance_257_keep_dims_0 = const()[name = string("variance_257_keep_dims_0"), val = bool(true)]; + tensor variance_257_cast_fp16 = reduce_mean(axes = variance_257_axes_0, keep_dims = variance_257_keep_dims_0, x = inputs_sq_257_cast_fp16)[name = string("variance_257_cast_fp16")]; + fp16 var_9478_to_fp16 = const()[name = string("op_9478_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9479_cast_fp16 = add(x = variance_257_cast_fp16, y = var_9478_to_fp16)[name = string("op_9479_cast_fp16")]; + fp32 var_9480_epsilon_0 = const()[name = string("op_9480_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9480_cast_fp16 = rsqrt(epsilon = var_9480_epsilon_0, x = var_9479_cast_fp16)[name = string("op_9480_cast_fp16")]; + tensor hidden_states_317_cast_fp16 = mul(x = inputs_257_cast_fp16, y = var_9480_cast_fp16)[name = string("hidden_states_317_cast_fp16")]; + tensor input_263_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_317_cast_fp16)[name = string("input_263_cast_fp16")]; + string input_265_pad_type_0 = const()[name = string("input_265_pad_type_0"), val = string("valid")]; + tensor input_265_strides_0 = const()[name = string("input_265_strides_0"), val = tensor([1, 1])]; + tensor input_265_pad_0 = const()[name = string("input_265_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_265_dilations_0 = const()[name = string("input_265_dilations_0"), val = tensor([1, 1])]; + int32 input_265_groups_0 = const()[name = string("input_265_groups_0"), val = int32(1)]; + tensor input_265_cast_fp16 = conv(dilations = input_265_dilations_0, groups = input_265_groups_0, pad = input_265_pad_0, pad_type = input_265_pad_type_0, strides = input_265_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_263_cast_fp16)[name = string("input_265_cast_fp16")]; + tensor var_9494_cast_fp16 = silu(x = input_265_cast_fp16)[name = string("op_9494_cast_fp16")]; + string var_9500_pad_type_0 = const()[name = string("op_9500_pad_type_0"), val = string("valid")]; + tensor var_9500_strides_0 = const()[name = string("op_9500_strides_0"), val = tensor([1, 1])]; + tensor var_9500_pad_0 = const()[name = string("op_9500_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9500_dilations_0 = const()[name = string("op_9500_dilations_0"), val = tensor([1, 1])]; + int32 var_9500_groups_0 = const()[name = string("op_9500_groups_0"), val = int32(1)]; + tensor var_9500_cast_fp16 = conv(dilations = var_9500_dilations_0, groups = var_9500_groups_0, pad = var_9500_pad_0, pad_type = var_9500_pad_type_0, strides = var_9500_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_263_cast_fp16)[name = string("op_9500_cast_fp16")]; + tensor input_267_cast_fp16 = mul(x = var_9494_cast_fp16, y = var_9500_cast_fp16)[name = string("input_267_cast_fp16")]; + string hidden_states_319_pad_type_0 = const()[name = string("hidden_states_319_pad_type_0"), val = string("valid")]; + tensor hidden_states_319_strides_0 = const()[name = string("hidden_states_319_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_319_pad_0 = const()[name = string("hidden_states_319_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_319_dilations_0 = const()[name = string("hidden_states_319_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_319_groups_0 = const()[name = string("hidden_states_319_groups_0"), val = int32(1)]; + tensor hidden_states_319_cast_fp16 = conv(dilations = hidden_states_319_dilations_0, groups = hidden_states_319_groups_0, pad = hidden_states_319_pad_0, pad_type = hidden_states_319_pad_type_0, strides = hidden_states_319_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_267_cast_fp16)[name = string("hidden_states_319_cast_fp16")]; + tensor inputs_259_cast_fp16 = add(x = inputs_257_cast_fp16, y = hidden_states_319_cast_fp16)[name = string("inputs_259_cast_fp16")]; + tensor obj_279_begin_0 = const()[name = string("obj_279_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_279_end_0 = const()[name = string("obj_279_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_279_end_mask_0 = const()[name = string("obj_279_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_279_cast_fp16 = slice_by_index(begin = obj_279_begin_0, end = obj_279_end_0, end_mask = obj_279_end_mask_0, x = key_caches_13_cast_fp16)[name = string("obj_279_cast_fp16")]; + tensor obj_281_begin_0 = const()[name = string("obj_281_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_281_end_0 = const()[name = string("obj_281_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_281_end_mask_0 = const()[name = string("obj_281_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_281_cast_fp16 = slice_by_index(begin = obj_281_begin_0, end = obj_281_end_0, end_mask = obj_281_end_mask_0, x = value_caches_13_cast_fp16)[name = string("obj_281_cast_fp16")]; + int32 var_9548 = const()[name = string("op_9548"), val = int32(3)]; + int32 var_9558 = const()[name = string("op_9558"), val = int32(-2)]; + tensor inputs_sq_259_cast_fp16 = mul(x = inputs_259_cast_fp16, y = inputs_259_cast_fp16)[name = string("inputs_sq_259_cast_fp16")]; + tensor variance_259_axes_0 = const()[name = string("variance_259_axes_0"), val = tensor([1])]; + bool variance_259_keep_dims_0 = const()[name = string("variance_259_keep_dims_0"), val = bool(true)]; + tensor variance_259_cast_fp16 = reduce_mean(axes = variance_259_axes_0, keep_dims = variance_259_keep_dims_0, x = inputs_sq_259_cast_fp16)[name = string("variance_259_cast_fp16")]; + fp16 var_9572_to_fp16 = const()[name = string("op_9572_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9573_cast_fp16 = add(x = variance_259_cast_fp16, y = var_9572_to_fp16)[name = string("op_9573_cast_fp16")]; + fp32 var_9574_epsilon_0 = const()[name = string("op_9574_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9574_cast_fp16 = rsqrt(epsilon = var_9574_epsilon_0, x = var_9573_cast_fp16)[name = string("op_9574_cast_fp16")]; + tensor hidden_states_321_cast_fp16 = mul(x = inputs_259_cast_fp16, y = var_9574_cast_fp16)[name = string("hidden_states_321_cast_fp16")]; + tensor obj_277_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_321_cast_fp16)[name = string("obj_277_cast_fp16")]; + string query_187_pad_type_0 = const()[name = string("query_187_pad_type_0"), val = string("valid")]; + tensor query_187_strides_0 = const()[name = string("query_187_strides_0"), val = tensor([1, 1])]; + tensor query_187_pad_0 = const()[name = string("query_187_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_187_dilations_0 = const()[name = string("query_187_dilations_0"), val = tensor([1, 1])]; + int32 query_187_groups_0 = const()[name = string("query_187_groups_0"), val = int32(1)]; + tensor query_187_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_187_dilations_0, groups = query_187_groups_0, pad = query_187_pad_0, pad_type = query_187_pad_type_0, strides = query_187_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_277_cast_fp16)[name = string("query_187_cast_fp16")]; + string current_key_125_pad_type_0 = const()[name = string("current_key_125_pad_type_0"), val = string("valid")]; + tensor current_key_125_strides_0 = const()[name = string("current_key_125_strides_0"), val = tensor([1, 1])]; + tensor current_key_125_pad_0 = const()[name = string("current_key_125_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_125_dilations_0 = const()[name = string("current_key_125_dilations_0"), val = tensor([1, 1])]; + int32 current_key_125_groups_0 = const()[name = string("current_key_125_groups_0"), val = int32(1)]; + tensor current_key_125_cast_fp16 = conv(dilations = current_key_125_dilations_0, groups = current_key_125_groups_0, pad = current_key_125_pad_0, pad_type = current_key_125_pad_type_0, strides = current_key_125_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_277_cast_fp16)[name = string("current_key_125_cast_fp16")]; + string current_value_63_pad_type_0 = const()[name = string("current_value_63_pad_type_0"), val = string("valid")]; + tensor current_value_63_strides_0 = const()[name = string("current_value_63_strides_0"), val = tensor([1, 1])]; + tensor current_value_63_pad_0 = const()[name = string("current_value_63_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_63_dilations_0 = const()[name = string("current_value_63_dilations_0"), val = tensor([1, 1])]; + int32 current_value_63_groups_0 = const()[name = string("current_value_63_groups_0"), val = int32(1)]; + tensor current_value_63_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_63_dilations_0, groups = current_value_63_groups_0, pad = current_value_63_pad_0, pad_type = current_value_63_pad_type_0, strides = current_value_63_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_277_cast_fp16)[name = string("current_value_63_cast_fp16")]; + tensor var_9611 = const()[name = string("op_9611"), val = tensor([16, 128, 1, 1])]; + tensor inputs_261_cast_fp16 = reshape(shape = var_9611, x = query_187_cast_fp16)[name = string("inputs_261_cast_fp16")]; + tensor inputs_sq_261_cast_fp16 = mul(x = inputs_261_cast_fp16, y = inputs_261_cast_fp16)[name = string("inputs_sq_261_cast_fp16")]; + tensor variance_261_axes_0 = const()[name = string("variance_261_axes_0"), val = tensor([1])]; + bool variance_261_keep_dims_0 = const()[name = string("variance_261_keep_dims_0"), val = bool(true)]; + tensor variance_261_cast_fp16 = reduce_mean(axes = variance_261_axes_0, keep_dims = variance_261_keep_dims_0, x = inputs_sq_261_cast_fp16)[name = string("variance_261_cast_fp16")]; + fp16 var_9617_to_fp16 = const()[name = string("op_9617_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9618_cast_fp16 = add(x = variance_261_cast_fp16, y = var_9617_to_fp16)[name = string("op_9618_cast_fp16")]; + fp32 var_9619_epsilon_0 = const()[name = string("op_9619_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9619_cast_fp16 = rsqrt(epsilon = var_9619_epsilon_0, x = var_9618_cast_fp16)[name = string("op_9619_cast_fp16")]; + tensor hidden_states_323_cast_fp16 = mul(x = inputs_261_cast_fp16, y = var_9619_cast_fp16)[name = string("hidden_states_323_cast_fp16")]; + tensor query_normed_63_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_323_cast_fp16)[name = string("query_normed_63_cast_fp16")]; + tensor var_9627 = const()[name = string("op_9627"), val = tensor([8, 128, 1, 1])]; + tensor inputs_263_cast_fp16 = reshape(shape = var_9627, x = current_key_125_cast_fp16)[name = string("inputs_263_cast_fp16")]; + tensor inputs_sq_263_cast_fp16 = mul(x = inputs_263_cast_fp16, y = inputs_263_cast_fp16)[name = string("inputs_sq_263_cast_fp16")]; + tensor variance_263_axes_0 = const()[name = string("variance_263_axes_0"), val = tensor([1])]; + bool variance_263_keep_dims_0 = const()[name = string("variance_263_keep_dims_0"), val = bool(true)]; + tensor variance_263_cast_fp16 = reduce_mean(axes = variance_263_axes_0, keep_dims = variance_263_keep_dims_0, x = inputs_sq_263_cast_fp16)[name = string("variance_263_cast_fp16")]; + fp16 var_9633_to_fp16 = const()[name = string("op_9633_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9634_cast_fp16 = add(x = variance_263_cast_fp16, y = var_9633_to_fp16)[name = string("op_9634_cast_fp16")]; + fp32 var_9635_epsilon_0 = const()[name = string("op_9635_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9635_cast_fp16 = rsqrt(epsilon = var_9635_epsilon_0, x = var_9634_cast_fp16)[name = string("op_9635_cast_fp16")]; + tensor hidden_states_325_cast_fp16 = mul(x = inputs_263_cast_fp16, y = var_9635_cast_fp16)[name = string("hidden_states_325_cast_fp16")]; + tensor current_key_normed_63_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_325_cast_fp16)[name = string("current_key_normed_63_cast_fp16")]; + tensor var_9653 = const()[name = string("op_9653"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_249_cast_fp16 = reshape(shape = var_9653, x = query_normed_63_cast_fp16)[name = string("mh_q_249_cast_fp16")]; + tensor var_9655 = const()[name = string("op_9655"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_249_cast_fp16 = reshape(shape = var_9655, x = current_key_normed_63_cast_fp16)[name = string("mh_k_249_cast_fp16")]; + tensor var_9659_cast_fp16 = mul(x = mh_q_249_cast_fp16, y = cos_61_to_fp16)[name = string("op_9659_cast_fp16")]; + tensor var_9664_begin_0 = const()[name = string("op_9664_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9664_end_0 = const()[name = string("op_9664_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_9664_end_mask_0 = const()[name = string("op_9664_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_9664_cast_fp16 = slice_by_index(begin = var_9664_begin_0, end = var_9664_end_0, end_mask = var_9664_end_mask_0, x = mh_q_249_cast_fp16)[name = string("op_9664_cast_fp16")]; + tensor var_9670_begin_0 = const()[name = string("op_9670_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_9670_end_0 = const()[name = string("op_9670_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_9670_end_mask_0 = const()[name = string("op_9670_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_9670_cast_fp16 = slice_by_index(begin = var_9670_begin_0, end = var_9670_end_0, end_mask = var_9670_end_mask_0, x = mh_q_249_cast_fp16)[name = string("op_9670_cast_fp16")]; + fp16 const_640_promoted_to_fp16 = const()[name = string("const_640_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9672_cast_fp16 = mul(x = var_9670_cast_fp16, y = const_640_promoted_to_fp16)[name = string("op_9672_cast_fp16")]; + bool var_9674_interleave_0 = const()[name = string("op_9674_interleave_0"), val = bool(false)]; + tensor var_9674_cast_fp16 = concat(axis = var_9558, interleave = var_9674_interleave_0, values = (var_9672_cast_fp16, var_9664_cast_fp16))[name = string("op_9674_cast_fp16")]; + tensor var_9675_cast_fp16 = mul(x = var_9674_cast_fp16, y = sin_61_to_fp16)[name = string("op_9675_cast_fp16")]; + tensor mh_q_251_cast_fp16 = add(x = var_9659_cast_fp16, y = var_9675_cast_fp16)[name = string("mh_q_251_cast_fp16")]; + tensor var_9677_cast_fp16 = mul(x = mh_k_249_cast_fp16, y = cos_61_to_fp16)[name = string("op_9677_cast_fp16")]; + tensor var_9682_begin_0 = const()[name = string("op_9682_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9682_end_0 = const()[name = string("op_9682_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_9682_end_mask_0 = const()[name = string("op_9682_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_9682_cast_fp16 = slice_by_index(begin = var_9682_begin_0, end = var_9682_end_0, end_mask = var_9682_end_mask_0, x = mh_k_249_cast_fp16)[name = string("op_9682_cast_fp16")]; + tensor var_9688_begin_0 = const()[name = string("op_9688_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_9688_end_0 = const()[name = string("op_9688_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_9688_end_mask_0 = const()[name = string("op_9688_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_9688_cast_fp16 = slice_by_index(begin = var_9688_begin_0, end = var_9688_end_0, end_mask = var_9688_end_mask_0, x = mh_k_249_cast_fp16)[name = string("op_9688_cast_fp16")]; + fp16 const_643_promoted_to_fp16 = const()[name = string("const_643_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9690_cast_fp16 = mul(x = var_9688_cast_fp16, y = const_643_promoted_to_fp16)[name = string("op_9690_cast_fp16")]; + bool var_9692_interleave_0 = const()[name = string("op_9692_interleave_0"), val = bool(false)]; + tensor var_9692_cast_fp16 = concat(axis = var_9558, interleave = var_9692_interleave_0, values = (var_9690_cast_fp16, var_9682_cast_fp16))[name = string("op_9692_cast_fp16")]; + tensor var_9693_cast_fp16 = mul(x = var_9692_cast_fp16, y = sin_61_to_fp16)[name = string("op_9693_cast_fp16")]; + tensor mh_k_251_cast_fp16 = add(x = var_9677_cast_fp16, y = var_9693_cast_fp16)[name = string("mh_k_251_cast_fp16")]; + tensor var_9697 = const()[name = string("op_9697"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_127_cast_fp16 = reshape(shape = var_9697, x = mh_k_251_cast_fp16)[name = string("current_key_127_cast_fp16")]; + tensor var_9703_to_fp16 = const()[name = string("op_9703_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195328)))]; + tensor var_9704_cast_fp16 = mul(x = obj_279_cast_fp16, y = var_9703_to_fp16)[name = string("op_9704_cast_fp16")]; + tensor var_9701_to_fp16 = const()[name = string("op_9701_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195456)))]; + tensor var_9705_cast_fp16 = mul(x = current_key_127_cast_fp16, y = var_9701_to_fp16)[name = string("op_9705_cast_fp16")]; + tensor key_127_cast_fp16 = add(x = var_9704_cast_fp16, y = var_9705_cast_fp16)[name = string("key_127_cast_fp16")]; + tensor var_9707_to_fp16 = const()[name = string("op_9707_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195328)))]; + tensor var_9708_cast_fp16 = mul(x = obj_281_cast_fp16, y = var_9707_to_fp16)[name = string("op_9708_cast_fp16")]; + tensor var_9709_cast_fp16 = mul(x = current_value_63_cast_fp16, y = var_9701_to_fp16)[name = string("op_9709_cast_fp16")]; + tensor value_63_cast_fp16 = add(x = var_9708_cast_fp16, y = var_9709_cast_fp16)[name = string("value_63_cast_fp16")]; + fp16 var_9716_to_fp16 = const()[name = string("op_9716_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_255_cast_fp16 = mul(x = mh_q_251_cast_fp16, y = var_9716_to_fp16)[name = string("mh_q_255_cast_fp16")]; + tensor var_9718 = const()[name = string("op_9718"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_253_cast_fp16 = reshape(shape = var_9718, x = key_127_cast_fp16)[name = string("mh_k_253_cast_fp16")]; + tensor var_9720 = const()[name = string("op_9720"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_125_cast_fp16 = reshape(shape = var_9720, x = value_63_cast_fp16)[name = string("mh_v_125_cast_fp16")]; + tensor transpose_124_perm_0 = const()[name = string("transpose_124_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_62_reps_0 = const()[name = string("tile_62_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_124_cast_fp16 = transpose(perm = transpose_124_perm_0, x = mh_k_253_cast_fp16)[name = string("transpose_293")]; + tensor tile_62_cast_fp16 = tile(reps = tile_62_reps_0, x = transpose_124_cast_fp16)[name = string("tile_62_cast_fp16")]; + tensor concat_157 = const()[name = string("concat_157"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_124_cast_fp16 = reshape(shape = concat_157, x = tile_62_cast_fp16)[name = string("reshape_124_cast_fp16")]; + tensor transpose_125_perm_0 = const()[name = string("transpose_125_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_158 = const()[name = string("concat_158"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_125_cast_fp16 = transpose(perm = transpose_125_perm_0, x = reshape_124_cast_fp16)[name = string("transpose_292")]; + tensor reshape_125_cast_fp16 = reshape(shape = concat_158, x = transpose_125_cast_fp16)[name = string("reshape_125_cast_fp16")]; + tensor transpose_126_perm_0 = const()[name = string("transpose_126_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_63_reps_0 = const()[name = string("tile_63_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_126_cast_fp16 = transpose(perm = transpose_126_perm_0, x = mh_v_125_cast_fp16)[name = string("transpose_291")]; + tensor tile_63_cast_fp16 = tile(reps = tile_63_reps_0, x = transpose_126_cast_fp16)[name = string("tile_63_cast_fp16")]; + tensor concat_159 = const()[name = string("concat_159"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_126_cast_fp16 = reshape(shape = concat_159, x = tile_63_cast_fp16)[name = string("reshape_126_cast_fp16")]; + tensor transpose_127_perm_0 = const()[name = string("transpose_127_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_160 = const()[name = string("concat_160"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_127_cast_fp16 = transpose(perm = transpose_127_perm_0, x = reshape_126_cast_fp16)[name = string("transpose_290")]; + tensor reshape_127_cast_fp16 = reshape(shape = concat_160, x = transpose_127_cast_fp16)[name = string("reshape_127_cast_fp16")]; + tensor transpose_441_perm_0 = const()[name = string("transpose_441_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_187_transpose_x_1 = const()[name = string("mh_w_187_transpose_x_1"), val = bool(true)]; + bool mh_w_187_transpose_y_1 = const()[name = string("mh_w_187_transpose_y_1"), val = bool(false)]; + tensor transpose_441_cast_fp16 = transpose(perm = transpose_441_perm_0, x = reshape_125_cast_fp16)[name = string("transpose_289")]; + tensor mh_w_187_cast_fp16 = matmul(transpose_x = mh_w_187_transpose_x_1, transpose_y = mh_w_187_transpose_y_1, x = mh_q_255_cast_fp16, y = transpose_441_cast_fp16)[name = string("mh_w_187_cast_fp16")]; + tensor var_9728_to_fp16 = const()[name = string("op_9728_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195584)))]; + tensor mh_w_189_cast_fp16 = add(x = mh_w_187_cast_fp16, y = var_9728_to_fp16)[name = string("mh_w_189_cast_fp16")]; + tensor mh_w_191_cast_fp16 = softmax(axis = var_9548, x = mh_w_189_cast_fp16)[name = string("mh_w_191_cast_fp16")]; + tensor transpose_442_perm_0 = const()[name = string("transpose_442_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_63_transpose_x_1 = const()[name = string("attn_63_transpose_x_1"), val = bool(false)]; + bool attn_63_transpose_y_1 = const()[name = string("attn_63_transpose_y_1"), val = bool(true)]; + tensor transpose_442_cast_fp16 = transpose(perm = transpose_442_perm_0, x = reshape_127_cast_fp16)[name = string("transpose_288")]; + tensor attn_63_cast_fp16 = matmul(transpose_x = attn_63_transpose_x_1, transpose_y = attn_63_transpose_y_1, x = transpose_442_cast_fp16, y = mh_w_191_cast_fp16)[name = string("attn_63_cast_fp16")]; + tensor var_9734 = const()[name = string("op_9734"), val = tensor([1, 2048, 1, 1])]; + tensor input_269_cast_fp16 = reshape(shape = var_9734, x = attn_63_cast_fp16)[name = string("input_269_cast_fp16")]; + string obj_283_pad_type_0 = const()[name = string("obj_283_pad_type_0"), val = string("valid")]; + tensor obj_283_strides_0 = const()[name = string("obj_283_strides_0"), val = tensor([1, 1])]; + tensor obj_283_pad_0 = const()[name = string("obj_283_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_283_dilations_0 = const()[name = string("obj_283_dilations_0"), val = tensor([1, 1])]; + int32 obj_283_groups_0 = const()[name = string("obj_283_groups_0"), val = int32(1)]; + tensor obj_283_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_283_dilations_0, groups = obj_283_groups_0, pad = obj_283_pad_0, pad_type = obj_283_pad_type_0, strides = obj_283_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_269_cast_fp16)[name = string("obj_283_cast_fp16")]; + tensor inputs_265_cast_fp16 = add(x = inputs_259_cast_fp16, y = obj_283_cast_fp16)[name = string("inputs_265_cast_fp16")]; + tensor inputs_sq_265_cast_fp16 = mul(x = inputs_265_cast_fp16, y = inputs_265_cast_fp16)[name = string("inputs_sq_265_cast_fp16")]; + tensor variance_265_axes_0 = const()[name = string("variance_265_axes_0"), val = tensor([1])]; + bool variance_265_keep_dims_0 = const()[name = string("variance_265_keep_dims_0"), val = bool(true)]; + tensor variance_265_cast_fp16 = reduce_mean(axes = variance_265_axes_0, keep_dims = variance_265_keep_dims_0, x = inputs_sq_265_cast_fp16)[name = string("variance_265_cast_fp16")]; + fp16 var_9752_to_fp16 = const()[name = string("op_9752_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9753_cast_fp16 = add(x = variance_265_cast_fp16, y = var_9752_to_fp16)[name = string("op_9753_cast_fp16")]; + fp32 var_9754_epsilon_0 = const()[name = string("op_9754_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9754_cast_fp16 = rsqrt(epsilon = var_9754_epsilon_0, x = var_9753_cast_fp16)[name = string("op_9754_cast_fp16")]; + tensor hidden_states_327_cast_fp16 = mul(x = inputs_265_cast_fp16, y = var_9754_cast_fp16)[name = string("hidden_states_327_cast_fp16")]; + tensor input_271_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_327_cast_fp16)[name = string("input_271_cast_fp16")]; + string input_273_pad_type_0 = const()[name = string("input_273_pad_type_0"), val = string("valid")]; + tensor input_273_strides_0 = const()[name = string("input_273_strides_0"), val = tensor([1, 1])]; + tensor input_273_pad_0 = const()[name = string("input_273_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_273_dilations_0 = const()[name = string("input_273_dilations_0"), val = tensor([1, 1])]; + int32 input_273_groups_0 = const()[name = string("input_273_groups_0"), val = int32(1)]; + tensor input_273_cast_fp16 = conv(dilations = input_273_dilations_0, groups = input_273_groups_0, pad = input_273_pad_0, pad_type = input_273_pad_type_0, strides = input_273_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_271_cast_fp16)[name = string("input_273_cast_fp16")]; + tensor var_9768_cast_fp16 = silu(x = input_273_cast_fp16)[name = string("op_9768_cast_fp16")]; + string var_9774_pad_type_0 = const()[name = string("op_9774_pad_type_0"), val = string("valid")]; + tensor var_9774_strides_0 = const()[name = string("op_9774_strides_0"), val = tensor([1, 1])]; + tensor var_9774_pad_0 = const()[name = string("op_9774_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9774_dilations_0 = const()[name = string("op_9774_dilations_0"), val = tensor([1, 1])]; + int32 var_9774_groups_0 = const()[name = string("op_9774_groups_0"), val = int32(1)]; + tensor var_9774_cast_fp16 = conv(dilations = var_9774_dilations_0, groups = var_9774_groups_0, pad = var_9774_pad_0, pad_type = var_9774_pad_type_0, strides = var_9774_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_271_cast_fp16)[name = string("op_9774_cast_fp16")]; + tensor input_275_cast_fp16 = mul(x = var_9768_cast_fp16, y = var_9774_cast_fp16)[name = string("input_275_cast_fp16")]; + string hidden_states_329_pad_type_0 = const()[name = string("hidden_states_329_pad_type_0"), val = string("valid")]; + tensor hidden_states_329_strides_0 = const()[name = string("hidden_states_329_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_329_pad_0 = const()[name = string("hidden_states_329_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_329_dilations_0 = const()[name = string("hidden_states_329_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_329_groups_0 = const()[name = string("hidden_states_329_groups_0"), val = int32(1)]; + tensor hidden_states_329_cast_fp16 = conv(dilations = hidden_states_329_dilations_0, groups = hidden_states_329_groups_0, pad = hidden_states_329_pad_0, pad_type = hidden_states_329_pad_type_0, strides = hidden_states_329_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_275_cast_fp16)[name = string("hidden_states_329_cast_fp16")]; + tensor inputs_267_cast_fp16 = add(x = inputs_265_cast_fp16, y = hidden_states_329_cast_fp16)[name = string("inputs_267_cast_fp16")]; + tensor obj_287_begin_0 = const()[name = string("obj_287_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_287_end_0 = const()[name = string("obj_287_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_287_end_mask_0 = const()[name = string("obj_287_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_287_cast_fp16 = slice_by_index(begin = obj_287_begin_0, end = obj_287_end_0, end_mask = obj_287_end_mask_0, x = key_caches_13_cast_fp16)[name = string("obj_287_cast_fp16")]; + tensor obj_289_begin_0 = const()[name = string("obj_289_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_289_end_0 = const()[name = string("obj_289_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_289_end_mask_0 = const()[name = string("obj_289_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_289_cast_fp16 = slice_by_index(begin = obj_289_begin_0, end = obj_289_end_0, end_mask = obj_289_end_mask_0, x = value_caches_13_cast_fp16)[name = string("obj_289_cast_fp16")]; + int32 var_9822 = const()[name = string("op_9822"), val = int32(3)]; + int32 var_9832 = const()[name = string("op_9832"), val = int32(-2)]; + tensor inputs_sq_267_cast_fp16 = mul(x = inputs_267_cast_fp16, y = inputs_267_cast_fp16)[name = string("inputs_sq_267_cast_fp16")]; + tensor variance_267_axes_0 = const()[name = string("variance_267_axes_0"), val = tensor([1])]; + bool variance_267_keep_dims_0 = const()[name = string("variance_267_keep_dims_0"), val = bool(true)]; + tensor variance_267_cast_fp16 = reduce_mean(axes = variance_267_axes_0, keep_dims = variance_267_keep_dims_0, x = inputs_sq_267_cast_fp16)[name = string("variance_267_cast_fp16")]; + fp16 var_9846_to_fp16 = const()[name = string("op_9846_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9847_cast_fp16 = add(x = variance_267_cast_fp16, y = var_9846_to_fp16)[name = string("op_9847_cast_fp16")]; + fp32 var_9848_epsilon_0 = const()[name = string("op_9848_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9848_cast_fp16 = rsqrt(epsilon = var_9848_epsilon_0, x = var_9847_cast_fp16)[name = string("op_9848_cast_fp16")]; + tensor hidden_states_331_cast_fp16 = mul(x = inputs_267_cast_fp16, y = var_9848_cast_fp16)[name = string("hidden_states_331_cast_fp16")]; + tensor obj_285_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_331_cast_fp16)[name = string("obj_285_cast_fp16")]; + string query_193_pad_type_0 = const()[name = string("query_193_pad_type_0"), val = string("valid")]; + tensor query_193_strides_0 = const()[name = string("query_193_strides_0"), val = tensor([1, 1])]; + tensor query_193_pad_0 = const()[name = string("query_193_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_193_dilations_0 = const()[name = string("query_193_dilations_0"), val = tensor([1, 1])]; + int32 query_193_groups_0 = const()[name = string("query_193_groups_0"), val = int32(1)]; + tensor query_193_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_193_dilations_0, groups = query_193_groups_0, pad = query_193_pad_0, pad_type = query_193_pad_type_0, strides = query_193_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_285_cast_fp16)[name = string("query_193_cast_fp16")]; + string current_key_129_pad_type_0 = const()[name = string("current_key_129_pad_type_0"), val = string("valid")]; + tensor current_key_129_strides_0 = const()[name = string("current_key_129_strides_0"), val = tensor([1, 1])]; + tensor current_key_129_pad_0 = const()[name = string("current_key_129_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_129_dilations_0 = const()[name = string("current_key_129_dilations_0"), val = tensor([1, 1])]; + int32 current_key_129_groups_0 = const()[name = string("current_key_129_groups_0"), val = int32(1)]; + tensor current_key_129_cast_fp16 = conv(dilations = current_key_129_dilations_0, groups = current_key_129_groups_0, pad = current_key_129_pad_0, pad_type = current_key_129_pad_type_0, strides = current_key_129_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_285_cast_fp16)[name = string("current_key_129_cast_fp16")]; + string current_value_65_pad_type_0 = const()[name = string("current_value_65_pad_type_0"), val = string("valid")]; + tensor current_value_65_strides_0 = const()[name = string("current_value_65_strides_0"), val = tensor([1, 1])]; + tensor current_value_65_pad_0 = const()[name = string("current_value_65_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_65_dilations_0 = const()[name = string("current_value_65_dilations_0"), val = tensor([1, 1])]; + int32 current_value_65_groups_0 = const()[name = string("current_value_65_groups_0"), val = int32(1)]; + tensor current_value_65_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_65_dilations_0, groups = current_value_65_groups_0, pad = current_value_65_pad_0, pad_type = current_value_65_pad_type_0, strides = current_value_65_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_285_cast_fp16)[name = string("current_value_65_cast_fp16")]; + tensor var_9885 = const()[name = string("op_9885"), val = tensor([16, 128, 1, 1])]; + tensor inputs_269_cast_fp16 = reshape(shape = var_9885, x = query_193_cast_fp16)[name = string("inputs_269_cast_fp16")]; + tensor inputs_sq_269_cast_fp16 = mul(x = inputs_269_cast_fp16, y = inputs_269_cast_fp16)[name = string("inputs_sq_269_cast_fp16")]; + tensor variance_269_axes_0 = const()[name = string("variance_269_axes_0"), val = tensor([1])]; + bool variance_269_keep_dims_0 = const()[name = string("variance_269_keep_dims_0"), val = bool(true)]; + tensor variance_269_cast_fp16 = reduce_mean(axes = variance_269_axes_0, keep_dims = variance_269_keep_dims_0, x = inputs_sq_269_cast_fp16)[name = string("variance_269_cast_fp16")]; + fp16 var_9891_to_fp16 = const()[name = string("op_9891_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9892_cast_fp16 = add(x = variance_269_cast_fp16, y = var_9891_to_fp16)[name = string("op_9892_cast_fp16")]; + fp32 var_9893_epsilon_0 = const()[name = string("op_9893_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9893_cast_fp16 = rsqrt(epsilon = var_9893_epsilon_0, x = var_9892_cast_fp16)[name = string("op_9893_cast_fp16")]; + tensor hidden_states_333_cast_fp16 = mul(x = inputs_269_cast_fp16, y = var_9893_cast_fp16)[name = string("hidden_states_333_cast_fp16")]; + tensor query_normed_65_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_333_cast_fp16)[name = string("query_normed_65_cast_fp16")]; + tensor var_9901 = const()[name = string("op_9901"), val = tensor([8, 128, 1, 1])]; + tensor inputs_271_cast_fp16 = reshape(shape = var_9901, x = current_key_129_cast_fp16)[name = string("inputs_271_cast_fp16")]; + tensor inputs_sq_271_cast_fp16 = mul(x = inputs_271_cast_fp16, y = inputs_271_cast_fp16)[name = string("inputs_sq_271_cast_fp16")]; + tensor variance_271_axes_0 = const()[name = string("variance_271_axes_0"), val = tensor([1])]; + bool variance_271_keep_dims_0 = const()[name = string("variance_271_keep_dims_0"), val = bool(true)]; + tensor variance_271_cast_fp16 = reduce_mean(axes = variance_271_axes_0, keep_dims = variance_271_keep_dims_0, x = inputs_sq_271_cast_fp16)[name = string("variance_271_cast_fp16")]; + fp16 var_9907_to_fp16 = const()[name = string("op_9907_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_9908_cast_fp16 = add(x = variance_271_cast_fp16, y = var_9907_to_fp16)[name = string("op_9908_cast_fp16")]; + fp32 var_9909_epsilon_0 = const()[name = string("op_9909_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_9909_cast_fp16 = rsqrt(epsilon = var_9909_epsilon_0, x = var_9908_cast_fp16)[name = string("op_9909_cast_fp16")]; + tensor hidden_states_335_cast_fp16 = mul(x = inputs_271_cast_fp16, y = var_9909_cast_fp16)[name = string("hidden_states_335_cast_fp16")]; + tensor current_key_normed_65_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_335_cast_fp16)[name = string("current_key_normed_65_cast_fp16")]; + tensor var_9927 = const()[name = string("op_9927"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_257_cast_fp16 = reshape(shape = var_9927, x = query_normed_65_cast_fp16)[name = string("mh_q_257_cast_fp16")]; + tensor var_9929 = const()[name = string("op_9929"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_257_cast_fp16 = reshape(shape = var_9929, x = current_key_normed_65_cast_fp16)[name = string("mh_k_257_cast_fp16")]; + tensor var_9933_cast_fp16 = mul(x = mh_q_257_cast_fp16, y = cos_61_to_fp16)[name = string("op_9933_cast_fp16")]; + tensor var_9938_begin_0 = const()[name = string("op_9938_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9938_end_0 = const()[name = string("op_9938_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_9938_end_mask_0 = const()[name = string("op_9938_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_9938_cast_fp16 = slice_by_index(begin = var_9938_begin_0, end = var_9938_end_0, end_mask = var_9938_end_mask_0, x = mh_q_257_cast_fp16)[name = string("op_9938_cast_fp16")]; + tensor var_9944_begin_0 = const()[name = string("op_9944_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_9944_end_0 = const()[name = string("op_9944_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_9944_end_mask_0 = const()[name = string("op_9944_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_9944_cast_fp16 = slice_by_index(begin = var_9944_begin_0, end = var_9944_end_0, end_mask = var_9944_end_mask_0, x = mh_q_257_cast_fp16)[name = string("op_9944_cast_fp16")]; + fp16 const_660_promoted_to_fp16 = const()[name = string("const_660_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9946_cast_fp16 = mul(x = var_9944_cast_fp16, y = const_660_promoted_to_fp16)[name = string("op_9946_cast_fp16")]; + bool var_9948_interleave_0 = const()[name = string("op_9948_interleave_0"), val = bool(false)]; + tensor var_9948_cast_fp16 = concat(axis = var_9832, interleave = var_9948_interleave_0, values = (var_9946_cast_fp16, var_9938_cast_fp16))[name = string("op_9948_cast_fp16")]; + tensor var_9949_cast_fp16 = mul(x = var_9948_cast_fp16, y = sin_61_to_fp16)[name = string("op_9949_cast_fp16")]; + tensor mh_q_259_cast_fp16 = add(x = var_9933_cast_fp16, y = var_9949_cast_fp16)[name = string("mh_q_259_cast_fp16")]; + tensor var_9951_cast_fp16 = mul(x = mh_k_257_cast_fp16, y = cos_61_to_fp16)[name = string("op_9951_cast_fp16")]; + tensor var_9956_begin_0 = const()[name = string("op_9956_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_9956_end_0 = const()[name = string("op_9956_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_9956_end_mask_0 = const()[name = string("op_9956_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_9956_cast_fp16 = slice_by_index(begin = var_9956_begin_0, end = var_9956_end_0, end_mask = var_9956_end_mask_0, x = mh_k_257_cast_fp16)[name = string("op_9956_cast_fp16")]; + tensor var_9962_begin_0 = const()[name = string("op_9962_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_9962_end_0 = const()[name = string("op_9962_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_9962_end_mask_0 = const()[name = string("op_9962_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_9962_cast_fp16 = slice_by_index(begin = var_9962_begin_0, end = var_9962_end_0, end_mask = var_9962_end_mask_0, x = mh_k_257_cast_fp16)[name = string("op_9962_cast_fp16")]; + fp16 const_663_promoted_to_fp16 = const()[name = string("const_663_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_9964_cast_fp16 = mul(x = var_9962_cast_fp16, y = const_663_promoted_to_fp16)[name = string("op_9964_cast_fp16")]; + bool var_9966_interleave_0 = const()[name = string("op_9966_interleave_0"), val = bool(false)]; + tensor var_9966_cast_fp16 = concat(axis = var_9832, interleave = var_9966_interleave_0, values = (var_9964_cast_fp16, var_9956_cast_fp16))[name = string("op_9966_cast_fp16")]; + tensor var_9967_cast_fp16 = mul(x = var_9966_cast_fp16, y = sin_61_to_fp16)[name = string("op_9967_cast_fp16")]; + tensor mh_k_259_cast_fp16 = add(x = var_9951_cast_fp16, y = var_9967_cast_fp16)[name = string("mh_k_259_cast_fp16")]; + tensor var_9971 = const()[name = string("op_9971"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_131_cast_fp16 = reshape(shape = var_9971, x = mh_k_259_cast_fp16)[name = string("current_key_131_cast_fp16")]; + tensor var_9977_to_fp16 = const()[name = string("op_9977_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195328)))]; + tensor var_9978_cast_fp16 = mul(x = obj_287_cast_fp16, y = var_9977_to_fp16)[name = string("op_9978_cast_fp16")]; + tensor var_9975_to_fp16 = const()[name = string("op_9975_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195456)))]; + tensor var_9979_cast_fp16 = mul(x = current_key_131_cast_fp16, y = var_9975_to_fp16)[name = string("op_9979_cast_fp16")]; + tensor key_131_cast_fp16 = add(x = var_9978_cast_fp16, y = var_9979_cast_fp16)[name = string("key_131_cast_fp16")]; + tensor var_9981_to_fp16 = const()[name = string("op_9981_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195328)))]; + tensor var_9982_cast_fp16 = mul(x = obj_289_cast_fp16, y = var_9981_to_fp16)[name = string("op_9982_cast_fp16")]; + tensor var_9983_cast_fp16 = mul(x = current_value_65_cast_fp16, y = var_9975_to_fp16)[name = string("op_9983_cast_fp16")]; + tensor value_65_cast_fp16 = add(x = var_9982_cast_fp16, y = var_9983_cast_fp16)[name = string("value_65_cast_fp16")]; + fp16 var_9990_to_fp16 = const()[name = string("op_9990_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_263_cast_fp16 = mul(x = mh_q_259_cast_fp16, y = var_9990_to_fp16)[name = string("mh_q_263_cast_fp16")]; + tensor var_9992 = const()[name = string("op_9992"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_261_cast_fp16 = reshape(shape = var_9992, x = key_131_cast_fp16)[name = string("mh_k_261_cast_fp16")]; + tensor var_9994 = const()[name = string("op_9994"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_129_cast_fp16 = reshape(shape = var_9994, x = value_65_cast_fp16)[name = string("mh_v_129_cast_fp16")]; + tensor transpose_128_perm_0 = const()[name = string("transpose_128_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_64_reps_0 = const()[name = string("tile_64_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_128_cast_fp16 = transpose(perm = transpose_128_perm_0, x = mh_k_261_cast_fp16)[name = string("transpose_287")]; + tensor tile_64_cast_fp16 = tile(reps = tile_64_reps_0, x = transpose_128_cast_fp16)[name = string("tile_64_cast_fp16")]; + tensor concat_161 = const()[name = string("concat_161"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_128_cast_fp16 = reshape(shape = concat_161, x = tile_64_cast_fp16)[name = string("reshape_128_cast_fp16")]; + tensor transpose_129_perm_0 = const()[name = string("transpose_129_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_162 = const()[name = string("concat_162"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_129_cast_fp16 = transpose(perm = transpose_129_perm_0, x = reshape_128_cast_fp16)[name = string("transpose_286")]; + tensor reshape_129_cast_fp16 = reshape(shape = concat_162, x = transpose_129_cast_fp16)[name = string("reshape_129_cast_fp16")]; + tensor transpose_130_perm_0 = const()[name = string("transpose_130_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_65_reps_0 = const()[name = string("tile_65_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_130_cast_fp16 = transpose(perm = transpose_130_perm_0, x = mh_v_129_cast_fp16)[name = string("transpose_285")]; + tensor tile_65_cast_fp16 = tile(reps = tile_65_reps_0, x = transpose_130_cast_fp16)[name = string("tile_65_cast_fp16")]; + tensor concat_163 = const()[name = string("concat_163"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_130_cast_fp16 = reshape(shape = concat_163, x = tile_65_cast_fp16)[name = string("reshape_130_cast_fp16")]; + tensor transpose_131_perm_0 = const()[name = string("transpose_131_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_164 = const()[name = string("concat_164"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_131_cast_fp16 = transpose(perm = transpose_131_perm_0, x = reshape_130_cast_fp16)[name = string("transpose_284")]; + tensor reshape_131_cast_fp16 = reshape(shape = concat_164, x = transpose_131_cast_fp16)[name = string("reshape_131_cast_fp16")]; + tensor transpose_445_perm_0 = const()[name = string("transpose_445_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_193_transpose_x_1 = const()[name = string("mh_w_193_transpose_x_1"), val = bool(true)]; + bool mh_w_193_transpose_y_1 = const()[name = string("mh_w_193_transpose_y_1"), val = bool(false)]; + tensor transpose_445_cast_fp16 = transpose(perm = transpose_445_perm_0, x = reshape_129_cast_fp16)[name = string("transpose_283")]; + tensor mh_w_193_cast_fp16 = matmul(transpose_x = mh_w_193_transpose_x_1, transpose_y = mh_w_193_transpose_y_1, x = mh_q_263_cast_fp16, y = transpose_445_cast_fp16)[name = string("mh_w_193_cast_fp16")]; + tensor var_10002_to_fp16 = const()[name = string("op_10002_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195584)))]; + tensor mh_w_195_cast_fp16 = add(x = mh_w_193_cast_fp16, y = var_10002_to_fp16)[name = string("mh_w_195_cast_fp16")]; + tensor mh_w_197_cast_fp16 = softmax(axis = var_9822, x = mh_w_195_cast_fp16)[name = string("mh_w_197_cast_fp16")]; + tensor transpose_446_perm_0 = const()[name = string("transpose_446_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_65_transpose_x_1 = const()[name = string("attn_65_transpose_x_1"), val = bool(false)]; + bool attn_65_transpose_y_1 = const()[name = string("attn_65_transpose_y_1"), val = bool(true)]; + tensor transpose_446_cast_fp16 = transpose(perm = transpose_446_perm_0, x = reshape_131_cast_fp16)[name = string("transpose_282")]; + tensor attn_65_cast_fp16 = matmul(transpose_x = attn_65_transpose_x_1, transpose_y = attn_65_transpose_y_1, x = transpose_446_cast_fp16, y = mh_w_197_cast_fp16)[name = string("attn_65_cast_fp16")]; + tensor var_10008 = const()[name = string("op_10008"), val = tensor([1, 2048, 1, 1])]; + tensor input_277_cast_fp16 = reshape(shape = var_10008, x = attn_65_cast_fp16)[name = string("input_277_cast_fp16")]; + string obj_291_pad_type_0 = const()[name = string("obj_291_pad_type_0"), val = string("valid")]; + tensor obj_291_strides_0 = const()[name = string("obj_291_strides_0"), val = tensor([1, 1])]; + tensor obj_291_pad_0 = const()[name = string("obj_291_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_291_dilations_0 = const()[name = string("obj_291_dilations_0"), val = tensor([1, 1])]; + int32 obj_291_groups_0 = const()[name = string("obj_291_groups_0"), val = int32(1)]; + tensor obj_291_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_291_dilations_0, groups = obj_291_groups_0, pad = obj_291_pad_0, pad_type = obj_291_pad_type_0, strides = obj_291_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_277_cast_fp16)[name = string("obj_291_cast_fp16")]; + tensor inputs_273_cast_fp16 = add(x = inputs_267_cast_fp16, y = obj_291_cast_fp16)[name = string("inputs_273_cast_fp16")]; + tensor inputs_sq_273_cast_fp16 = mul(x = inputs_273_cast_fp16, y = inputs_273_cast_fp16)[name = string("inputs_sq_273_cast_fp16")]; + tensor variance_273_axes_0 = const()[name = string("variance_273_axes_0"), val = tensor([1])]; + bool variance_273_keep_dims_0 = const()[name = string("variance_273_keep_dims_0"), val = bool(true)]; + tensor variance_273_cast_fp16 = reduce_mean(axes = variance_273_axes_0, keep_dims = variance_273_keep_dims_0, x = inputs_sq_273_cast_fp16)[name = string("variance_273_cast_fp16")]; + fp16 var_10026_to_fp16 = const()[name = string("op_10026_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10027_cast_fp16 = add(x = variance_273_cast_fp16, y = var_10026_to_fp16)[name = string("op_10027_cast_fp16")]; + fp32 var_10028_epsilon_0 = const()[name = string("op_10028_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10028_cast_fp16 = rsqrt(epsilon = var_10028_epsilon_0, x = var_10027_cast_fp16)[name = string("op_10028_cast_fp16")]; + tensor hidden_states_337_cast_fp16 = mul(x = inputs_273_cast_fp16, y = var_10028_cast_fp16)[name = string("hidden_states_337_cast_fp16")]; + tensor input_279_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_337_cast_fp16)[name = string("input_279_cast_fp16")]; + string input_281_pad_type_0 = const()[name = string("input_281_pad_type_0"), val = string("valid")]; + tensor input_281_strides_0 = const()[name = string("input_281_strides_0"), val = tensor([1, 1])]; + tensor input_281_pad_0 = const()[name = string("input_281_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_281_dilations_0 = const()[name = string("input_281_dilations_0"), val = tensor([1, 1])]; + int32 input_281_groups_0 = const()[name = string("input_281_groups_0"), val = int32(1)]; + tensor input_281_cast_fp16 = conv(dilations = input_281_dilations_0, groups = input_281_groups_0, pad = input_281_pad_0, pad_type = input_281_pad_type_0, strides = input_281_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_279_cast_fp16)[name = string("input_281_cast_fp16")]; + tensor var_10042_cast_fp16 = silu(x = input_281_cast_fp16)[name = string("op_10042_cast_fp16")]; + string var_10048_pad_type_0 = const()[name = string("op_10048_pad_type_0"), val = string("valid")]; + tensor var_10048_strides_0 = const()[name = string("op_10048_strides_0"), val = tensor([1, 1])]; + tensor var_10048_pad_0 = const()[name = string("op_10048_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10048_dilations_0 = const()[name = string("op_10048_dilations_0"), val = tensor([1, 1])]; + int32 var_10048_groups_0 = const()[name = string("op_10048_groups_0"), val = int32(1)]; + tensor var_10048_cast_fp16 = conv(dilations = var_10048_dilations_0, groups = var_10048_groups_0, pad = var_10048_pad_0, pad_type = var_10048_pad_type_0, strides = var_10048_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_279_cast_fp16)[name = string("op_10048_cast_fp16")]; + tensor input_283_cast_fp16 = mul(x = var_10042_cast_fp16, y = var_10048_cast_fp16)[name = string("input_283_cast_fp16")]; + string hidden_states_339_pad_type_0 = const()[name = string("hidden_states_339_pad_type_0"), val = string("valid")]; + tensor hidden_states_339_strides_0 = const()[name = string("hidden_states_339_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_339_pad_0 = const()[name = string("hidden_states_339_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_339_dilations_0 = const()[name = string("hidden_states_339_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_339_groups_0 = const()[name = string("hidden_states_339_groups_0"), val = int32(1)]; + tensor hidden_states_339_cast_fp16 = conv(dilations = hidden_states_339_dilations_0, groups = hidden_states_339_groups_0, pad = hidden_states_339_pad_0, pad_type = hidden_states_339_pad_type_0, strides = hidden_states_339_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_283_cast_fp16)[name = string("hidden_states_339_cast_fp16")]; + tensor inputs_275_cast_fp16 = add(x = inputs_273_cast_fp16, y = hidden_states_339_cast_fp16)[name = string("inputs_275_cast_fp16")]; + tensor obj_295_begin_0 = const()[name = string("obj_295_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_295_end_0 = const()[name = string("obj_295_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_295_end_mask_0 = const()[name = string("obj_295_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_295_cast_fp16 = slice_by_index(begin = obj_295_begin_0, end = obj_295_end_0, end_mask = obj_295_end_mask_0, x = key_caches_13_cast_fp16)[name = string("obj_295_cast_fp16")]; + tensor obj_297_begin_0 = const()[name = string("obj_297_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_297_end_0 = const()[name = string("obj_297_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_297_end_mask_0 = const()[name = string("obj_297_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_297_cast_fp16 = slice_by_index(begin = obj_297_begin_0, end = obj_297_end_0, end_mask = obj_297_end_mask_0, x = value_caches_13_cast_fp16)[name = string("obj_297_cast_fp16")]; + int32 var_10096 = const()[name = string("op_10096"), val = int32(3)]; + int32 var_10106 = const()[name = string("op_10106"), val = int32(-2)]; + tensor inputs_sq_275_cast_fp16 = mul(x = inputs_275_cast_fp16, y = inputs_275_cast_fp16)[name = string("inputs_sq_275_cast_fp16")]; + tensor variance_275_axes_0 = const()[name = string("variance_275_axes_0"), val = tensor([1])]; + bool variance_275_keep_dims_0 = const()[name = string("variance_275_keep_dims_0"), val = bool(true)]; + tensor variance_275_cast_fp16 = reduce_mean(axes = variance_275_axes_0, keep_dims = variance_275_keep_dims_0, x = inputs_sq_275_cast_fp16)[name = string("variance_275_cast_fp16")]; + fp16 var_10120_to_fp16 = const()[name = string("op_10120_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10121_cast_fp16 = add(x = variance_275_cast_fp16, y = var_10120_to_fp16)[name = string("op_10121_cast_fp16")]; + fp32 var_10122_epsilon_0 = const()[name = string("op_10122_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10122_cast_fp16 = rsqrt(epsilon = var_10122_epsilon_0, x = var_10121_cast_fp16)[name = string("op_10122_cast_fp16")]; + tensor hidden_states_341_cast_fp16 = mul(x = inputs_275_cast_fp16, y = var_10122_cast_fp16)[name = string("hidden_states_341_cast_fp16")]; + tensor obj_293_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_341_cast_fp16)[name = string("obj_293_cast_fp16")]; + string query_199_pad_type_0 = const()[name = string("query_199_pad_type_0"), val = string("valid")]; + tensor query_199_strides_0 = const()[name = string("query_199_strides_0"), val = tensor([1, 1])]; + tensor query_199_pad_0 = const()[name = string("query_199_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_199_dilations_0 = const()[name = string("query_199_dilations_0"), val = tensor([1, 1])]; + int32 query_199_groups_0 = const()[name = string("query_199_groups_0"), val = int32(1)]; + tensor query_199_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_199_dilations_0, groups = query_199_groups_0, pad = query_199_pad_0, pad_type = query_199_pad_type_0, strides = query_199_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_293_cast_fp16)[name = string("query_199_cast_fp16")]; + string current_key_133_pad_type_0 = const()[name = string("current_key_133_pad_type_0"), val = string("valid")]; + tensor current_key_133_strides_0 = const()[name = string("current_key_133_strides_0"), val = tensor([1, 1])]; + tensor current_key_133_pad_0 = const()[name = string("current_key_133_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_133_dilations_0 = const()[name = string("current_key_133_dilations_0"), val = tensor([1, 1])]; + int32 current_key_133_groups_0 = const()[name = string("current_key_133_groups_0"), val = int32(1)]; + tensor current_key_133_cast_fp16 = conv(dilations = current_key_133_dilations_0, groups = current_key_133_groups_0, pad = current_key_133_pad_0, pad_type = current_key_133_pad_type_0, strides = current_key_133_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_293_cast_fp16)[name = string("current_key_133_cast_fp16")]; + string current_value_67_pad_type_0 = const()[name = string("current_value_67_pad_type_0"), val = string("valid")]; + tensor current_value_67_strides_0 = const()[name = string("current_value_67_strides_0"), val = tensor([1, 1])]; + tensor current_value_67_pad_0 = const()[name = string("current_value_67_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_67_dilations_0 = const()[name = string("current_value_67_dilations_0"), val = tensor([1, 1])]; + int32 current_value_67_groups_0 = const()[name = string("current_value_67_groups_0"), val = int32(1)]; + tensor current_value_67_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_67_dilations_0, groups = current_value_67_groups_0, pad = current_value_67_pad_0, pad_type = current_value_67_pad_type_0, strides = current_value_67_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_293_cast_fp16)[name = string("current_value_67_cast_fp16")]; + tensor var_10159 = const()[name = string("op_10159"), val = tensor([16, 128, 1, 1])]; + tensor inputs_277_cast_fp16 = reshape(shape = var_10159, x = query_199_cast_fp16)[name = string("inputs_277_cast_fp16")]; + tensor inputs_sq_277_cast_fp16 = mul(x = inputs_277_cast_fp16, y = inputs_277_cast_fp16)[name = string("inputs_sq_277_cast_fp16")]; + tensor variance_277_axes_0 = const()[name = string("variance_277_axes_0"), val = tensor([1])]; + bool variance_277_keep_dims_0 = const()[name = string("variance_277_keep_dims_0"), val = bool(true)]; + tensor variance_277_cast_fp16 = reduce_mean(axes = variance_277_axes_0, keep_dims = variance_277_keep_dims_0, x = inputs_sq_277_cast_fp16)[name = string("variance_277_cast_fp16")]; + fp16 var_10165_to_fp16 = const()[name = string("op_10165_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10166_cast_fp16 = add(x = variance_277_cast_fp16, y = var_10165_to_fp16)[name = string("op_10166_cast_fp16")]; + fp32 var_10167_epsilon_0 = const()[name = string("op_10167_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10167_cast_fp16 = rsqrt(epsilon = var_10167_epsilon_0, x = var_10166_cast_fp16)[name = string("op_10167_cast_fp16")]; + tensor hidden_states_343_cast_fp16 = mul(x = inputs_277_cast_fp16, y = var_10167_cast_fp16)[name = string("hidden_states_343_cast_fp16")]; + tensor query_normed_67_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_343_cast_fp16)[name = string("query_normed_67_cast_fp16")]; + tensor var_10175 = const()[name = string("op_10175"), val = tensor([8, 128, 1, 1])]; + tensor inputs_279_cast_fp16 = reshape(shape = var_10175, x = current_key_133_cast_fp16)[name = string("inputs_279_cast_fp16")]; + tensor inputs_sq_279_cast_fp16 = mul(x = inputs_279_cast_fp16, y = inputs_279_cast_fp16)[name = string("inputs_sq_279_cast_fp16")]; + tensor variance_279_axes_0 = const()[name = string("variance_279_axes_0"), val = tensor([1])]; + bool variance_279_keep_dims_0 = const()[name = string("variance_279_keep_dims_0"), val = bool(true)]; + tensor variance_279_cast_fp16 = reduce_mean(axes = variance_279_axes_0, keep_dims = variance_279_keep_dims_0, x = inputs_sq_279_cast_fp16)[name = string("variance_279_cast_fp16")]; + fp16 var_10181_to_fp16 = const()[name = string("op_10181_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10182_cast_fp16 = add(x = variance_279_cast_fp16, y = var_10181_to_fp16)[name = string("op_10182_cast_fp16")]; + fp32 var_10183_epsilon_0 = const()[name = string("op_10183_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10183_cast_fp16 = rsqrt(epsilon = var_10183_epsilon_0, x = var_10182_cast_fp16)[name = string("op_10183_cast_fp16")]; + tensor hidden_states_345_cast_fp16 = mul(x = inputs_279_cast_fp16, y = var_10183_cast_fp16)[name = string("hidden_states_345_cast_fp16")]; + tensor current_key_normed_67_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_345_cast_fp16)[name = string("current_key_normed_67_cast_fp16")]; + tensor var_10201 = const()[name = string("op_10201"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_265_cast_fp16 = reshape(shape = var_10201, x = query_normed_67_cast_fp16)[name = string("mh_q_265_cast_fp16")]; + tensor var_10203 = const()[name = string("op_10203"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_265_cast_fp16 = reshape(shape = var_10203, x = current_key_normed_67_cast_fp16)[name = string("mh_k_265_cast_fp16")]; + tensor var_10207_cast_fp16 = mul(x = mh_q_265_cast_fp16, y = cos_61_to_fp16)[name = string("op_10207_cast_fp16")]; + tensor var_10212_begin_0 = const()[name = string("op_10212_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10212_end_0 = const()[name = string("op_10212_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_10212_end_mask_0 = const()[name = string("op_10212_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_10212_cast_fp16 = slice_by_index(begin = var_10212_begin_0, end = var_10212_end_0, end_mask = var_10212_end_mask_0, x = mh_q_265_cast_fp16)[name = string("op_10212_cast_fp16")]; + tensor var_10218_begin_0 = const()[name = string("op_10218_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_10218_end_0 = const()[name = string("op_10218_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_10218_end_mask_0 = const()[name = string("op_10218_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_10218_cast_fp16 = slice_by_index(begin = var_10218_begin_0, end = var_10218_end_0, end_mask = var_10218_end_mask_0, x = mh_q_265_cast_fp16)[name = string("op_10218_cast_fp16")]; + fp16 const_680_promoted_to_fp16 = const()[name = string("const_680_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10220_cast_fp16 = mul(x = var_10218_cast_fp16, y = const_680_promoted_to_fp16)[name = string("op_10220_cast_fp16")]; + bool var_10222_interleave_0 = const()[name = string("op_10222_interleave_0"), val = bool(false)]; + tensor var_10222_cast_fp16 = concat(axis = var_10106, interleave = var_10222_interleave_0, values = (var_10220_cast_fp16, var_10212_cast_fp16))[name = string("op_10222_cast_fp16")]; + tensor var_10223_cast_fp16 = mul(x = var_10222_cast_fp16, y = sin_61_to_fp16)[name = string("op_10223_cast_fp16")]; + tensor mh_q_267_cast_fp16 = add(x = var_10207_cast_fp16, y = var_10223_cast_fp16)[name = string("mh_q_267_cast_fp16")]; + tensor var_10225_cast_fp16 = mul(x = mh_k_265_cast_fp16, y = cos_61_to_fp16)[name = string("op_10225_cast_fp16")]; + tensor var_10230_begin_0 = const()[name = string("op_10230_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10230_end_0 = const()[name = string("op_10230_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_10230_end_mask_0 = const()[name = string("op_10230_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_10230_cast_fp16 = slice_by_index(begin = var_10230_begin_0, end = var_10230_end_0, end_mask = var_10230_end_mask_0, x = mh_k_265_cast_fp16)[name = string("op_10230_cast_fp16")]; + tensor var_10236_begin_0 = const()[name = string("op_10236_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_10236_end_0 = const()[name = string("op_10236_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_10236_end_mask_0 = const()[name = string("op_10236_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_10236_cast_fp16 = slice_by_index(begin = var_10236_begin_0, end = var_10236_end_0, end_mask = var_10236_end_mask_0, x = mh_k_265_cast_fp16)[name = string("op_10236_cast_fp16")]; + fp16 const_683_promoted_to_fp16 = const()[name = string("const_683_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10238_cast_fp16 = mul(x = var_10236_cast_fp16, y = const_683_promoted_to_fp16)[name = string("op_10238_cast_fp16")]; + bool var_10240_interleave_0 = const()[name = string("op_10240_interleave_0"), val = bool(false)]; + tensor var_10240_cast_fp16 = concat(axis = var_10106, interleave = var_10240_interleave_0, values = (var_10238_cast_fp16, var_10230_cast_fp16))[name = string("op_10240_cast_fp16")]; + tensor var_10241_cast_fp16 = mul(x = var_10240_cast_fp16, y = sin_61_to_fp16)[name = string("op_10241_cast_fp16")]; + tensor mh_k_267_cast_fp16 = add(x = var_10225_cast_fp16, y = var_10241_cast_fp16)[name = string("mh_k_267_cast_fp16")]; + tensor var_10245 = const()[name = string("op_10245"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_135_cast_fp16 = reshape(shape = var_10245, x = mh_k_267_cast_fp16)[name = string("current_key_135_cast_fp16")]; + tensor var_10251_to_fp16 = const()[name = string("op_10251_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195328)))]; + tensor var_10252_cast_fp16 = mul(x = obj_295_cast_fp16, y = var_10251_to_fp16)[name = string("op_10252_cast_fp16")]; + tensor var_10249_to_fp16 = const()[name = string("op_10249_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195456)))]; + tensor var_10253_cast_fp16 = mul(x = current_key_135_cast_fp16, y = var_10249_to_fp16)[name = string("op_10253_cast_fp16")]; + tensor key_135_cast_fp16 = add(x = var_10252_cast_fp16, y = var_10253_cast_fp16)[name = string("key_135_cast_fp16")]; + tensor var_10255_to_fp16 = const()[name = string("op_10255_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195328)))]; + tensor var_10256_cast_fp16 = mul(x = obj_297_cast_fp16, y = var_10255_to_fp16)[name = string("op_10256_cast_fp16")]; + tensor var_10257_cast_fp16 = mul(x = current_value_67_cast_fp16, y = var_10249_to_fp16)[name = string("op_10257_cast_fp16")]; + tensor value_67_cast_fp16 = add(x = var_10256_cast_fp16, y = var_10257_cast_fp16)[name = string("value_67_cast_fp16")]; + fp16 var_10264_to_fp16 = const()[name = string("op_10264_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_271_cast_fp16 = mul(x = mh_q_267_cast_fp16, y = var_10264_to_fp16)[name = string("mh_q_271_cast_fp16")]; + tensor var_10266 = const()[name = string("op_10266"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_269_cast_fp16 = reshape(shape = var_10266, x = key_135_cast_fp16)[name = string("mh_k_269_cast_fp16")]; + tensor var_10268 = const()[name = string("op_10268"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_133_cast_fp16 = reshape(shape = var_10268, x = value_67_cast_fp16)[name = string("mh_v_133_cast_fp16")]; + tensor transpose_132_perm_0 = const()[name = string("transpose_132_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_66_reps_0 = const()[name = string("tile_66_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_132_cast_fp16 = transpose(perm = transpose_132_perm_0, x = mh_k_269_cast_fp16)[name = string("transpose_281")]; + tensor tile_66_cast_fp16 = tile(reps = tile_66_reps_0, x = transpose_132_cast_fp16)[name = string("tile_66_cast_fp16")]; + tensor concat_165 = const()[name = string("concat_165"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_132_cast_fp16 = reshape(shape = concat_165, x = tile_66_cast_fp16)[name = string("reshape_132_cast_fp16")]; + tensor transpose_133_perm_0 = const()[name = string("transpose_133_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_166 = const()[name = string("concat_166"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_133_cast_fp16 = transpose(perm = transpose_133_perm_0, x = reshape_132_cast_fp16)[name = string("transpose_280")]; + tensor reshape_133_cast_fp16 = reshape(shape = concat_166, x = transpose_133_cast_fp16)[name = string("reshape_133_cast_fp16")]; + tensor transpose_134_perm_0 = const()[name = string("transpose_134_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_67_reps_0 = const()[name = string("tile_67_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_134_cast_fp16 = transpose(perm = transpose_134_perm_0, x = mh_v_133_cast_fp16)[name = string("transpose_279")]; + tensor tile_67_cast_fp16 = tile(reps = tile_67_reps_0, x = transpose_134_cast_fp16)[name = string("tile_67_cast_fp16")]; + tensor concat_167 = const()[name = string("concat_167"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_134_cast_fp16 = reshape(shape = concat_167, x = tile_67_cast_fp16)[name = string("reshape_134_cast_fp16")]; + tensor transpose_135_perm_0 = const()[name = string("transpose_135_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_168 = const()[name = string("concat_168"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_135_cast_fp16 = transpose(perm = transpose_135_perm_0, x = reshape_134_cast_fp16)[name = string("transpose_278")]; + tensor reshape_135_cast_fp16 = reshape(shape = concat_168, x = transpose_135_cast_fp16)[name = string("reshape_135_cast_fp16")]; + tensor transpose_449_perm_0 = const()[name = string("transpose_449_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_199_transpose_x_1 = const()[name = string("mh_w_199_transpose_x_1"), val = bool(true)]; + bool mh_w_199_transpose_y_1 = const()[name = string("mh_w_199_transpose_y_1"), val = bool(false)]; + tensor transpose_449_cast_fp16 = transpose(perm = transpose_449_perm_0, x = reshape_133_cast_fp16)[name = string("transpose_277")]; + tensor mh_w_199_cast_fp16 = matmul(transpose_x = mh_w_199_transpose_x_1, transpose_y = mh_w_199_transpose_y_1, x = mh_q_271_cast_fp16, y = transpose_449_cast_fp16)[name = string("mh_w_199_cast_fp16")]; + tensor var_10276_to_fp16 = const()[name = string("op_10276_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195584)))]; + tensor mh_w_201_cast_fp16 = add(x = mh_w_199_cast_fp16, y = var_10276_to_fp16)[name = string("mh_w_201_cast_fp16")]; + tensor mh_w_203_cast_fp16 = softmax(axis = var_10096, x = mh_w_201_cast_fp16)[name = string("mh_w_203_cast_fp16")]; + tensor transpose_450_perm_0 = const()[name = string("transpose_450_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_67_transpose_x_1 = const()[name = string("attn_67_transpose_x_1"), val = bool(false)]; + bool attn_67_transpose_y_1 = const()[name = string("attn_67_transpose_y_1"), val = bool(true)]; + tensor transpose_450_cast_fp16 = transpose(perm = transpose_450_perm_0, x = reshape_135_cast_fp16)[name = string("transpose_276")]; + tensor attn_67_cast_fp16 = matmul(transpose_x = attn_67_transpose_x_1, transpose_y = attn_67_transpose_y_1, x = transpose_450_cast_fp16, y = mh_w_203_cast_fp16)[name = string("attn_67_cast_fp16")]; + tensor var_10282 = const()[name = string("op_10282"), val = tensor([1, 2048, 1, 1])]; + tensor input_285_cast_fp16 = reshape(shape = var_10282, x = attn_67_cast_fp16)[name = string("input_285_cast_fp16")]; + string obj_299_pad_type_0 = const()[name = string("obj_299_pad_type_0"), val = string("valid")]; + tensor obj_299_strides_0 = const()[name = string("obj_299_strides_0"), val = tensor([1, 1])]; + tensor obj_299_pad_0 = const()[name = string("obj_299_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_299_dilations_0 = const()[name = string("obj_299_dilations_0"), val = tensor([1, 1])]; + int32 obj_299_groups_0 = const()[name = string("obj_299_groups_0"), val = int32(1)]; + tensor obj_299_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_299_dilations_0, groups = obj_299_groups_0, pad = obj_299_pad_0, pad_type = obj_299_pad_type_0, strides = obj_299_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_285_cast_fp16)[name = string("obj_299_cast_fp16")]; + tensor inputs_281_cast_fp16 = add(x = inputs_275_cast_fp16, y = obj_299_cast_fp16)[name = string("inputs_281_cast_fp16")]; + tensor inputs_sq_281_cast_fp16 = mul(x = inputs_281_cast_fp16, y = inputs_281_cast_fp16)[name = string("inputs_sq_281_cast_fp16")]; + tensor variance_281_axes_0 = const()[name = string("variance_281_axes_0"), val = tensor([1])]; + bool variance_281_keep_dims_0 = const()[name = string("variance_281_keep_dims_0"), val = bool(true)]; + tensor variance_281_cast_fp16 = reduce_mean(axes = variance_281_axes_0, keep_dims = variance_281_keep_dims_0, x = inputs_sq_281_cast_fp16)[name = string("variance_281_cast_fp16")]; + fp16 var_10300_to_fp16 = const()[name = string("op_10300_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10301_cast_fp16 = add(x = variance_281_cast_fp16, y = var_10300_to_fp16)[name = string("op_10301_cast_fp16")]; + fp32 var_10302_epsilon_0 = const()[name = string("op_10302_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10302_cast_fp16 = rsqrt(epsilon = var_10302_epsilon_0, x = var_10301_cast_fp16)[name = string("op_10302_cast_fp16")]; + tensor hidden_states_347_cast_fp16 = mul(x = inputs_281_cast_fp16, y = var_10302_cast_fp16)[name = string("hidden_states_347_cast_fp16")]; + tensor input_287_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_347_cast_fp16)[name = string("input_287_cast_fp16")]; + string input_289_pad_type_0 = const()[name = string("input_289_pad_type_0"), val = string("valid")]; + tensor input_289_strides_0 = const()[name = string("input_289_strides_0"), val = tensor([1, 1])]; + tensor input_289_pad_0 = const()[name = string("input_289_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_289_dilations_0 = const()[name = string("input_289_dilations_0"), val = tensor([1, 1])]; + int32 input_289_groups_0 = const()[name = string("input_289_groups_0"), val = int32(1)]; + tensor input_289_cast_fp16 = conv(dilations = input_289_dilations_0, groups = input_289_groups_0, pad = input_289_pad_0, pad_type = input_289_pad_type_0, strides = input_289_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_287_cast_fp16)[name = string("input_289_cast_fp16")]; + tensor var_10316_cast_fp16 = silu(x = input_289_cast_fp16)[name = string("op_10316_cast_fp16")]; + string var_10322_pad_type_0 = const()[name = string("op_10322_pad_type_0"), val = string("valid")]; + tensor var_10322_strides_0 = const()[name = string("op_10322_strides_0"), val = tensor([1, 1])]; + tensor var_10322_pad_0 = const()[name = string("op_10322_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10322_dilations_0 = const()[name = string("op_10322_dilations_0"), val = tensor([1, 1])]; + int32 var_10322_groups_0 = const()[name = string("op_10322_groups_0"), val = int32(1)]; + tensor var_10322_cast_fp16 = conv(dilations = var_10322_dilations_0, groups = var_10322_groups_0, pad = var_10322_pad_0, pad_type = var_10322_pad_type_0, strides = var_10322_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_287_cast_fp16)[name = string("op_10322_cast_fp16")]; + tensor input_291_cast_fp16 = mul(x = var_10316_cast_fp16, y = var_10322_cast_fp16)[name = string("input_291_cast_fp16")]; + string hidden_states_349_pad_type_0 = const()[name = string("hidden_states_349_pad_type_0"), val = string("valid")]; + tensor hidden_states_349_strides_0 = const()[name = string("hidden_states_349_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_349_pad_0 = const()[name = string("hidden_states_349_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_349_dilations_0 = const()[name = string("hidden_states_349_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_349_groups_0 = const()[name = string("hidden_states_349_groups_0"), val = int32(1)]; + tensor hidden_states_349_cast_fp16 = conv(dilations = hidden_states_349_dilations_0, groups = hidden_states_349_groups_0, pad = hidden_states_349_pad_0, pad_type = hidden_states_349_pad_type_0, strides = hidden_states_349_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_291_cast_fp16)[name = string("hidden_states_349_cast_fp16")]; + tensor inputs_283_cast_fp16 = add(x = inputs_281_cast_fp16, y = hidden_states_349_cast_fp16)[name = string("inputs_283_cast_fp16")]; + tensor obj_303_begin_0 = const()[name = string("obj_303_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_303_end_0 = const()[name = string("obj_303_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_303_end_mask_0 = const()[name = string("obj_303_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_303_cast_fp16 = slice_by_index(begin = obj_303_begin_0, end = obj_303_end_0, end_mask = obj_303_end_mask_0, x = key_caches_13_cast_fp16)[name = string("obj_303_cast_fp16")]; + tensor obj_305_begin_0 = const()[name = string("obj_305_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_305_end_0 = const()[name = string("obj_305_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_305_end_mask_0 = const()[name = string("obj_305_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_305_cast_fp16 = slice_by_index(begin = obj_305_begin_0, end = obj_305_end_0, end_mask = obj_305_end_mask_0, x = value_caches_13_cast_fp16)[name = string("obj_305_cast_fp16")]; + int32 var_10370 = const()[name = string("op_10370"), val = int32(3)]; + int32 var_10380 = const()[name = string("op_10380"), val = int32(-2)]; + tensor inputs_sq_283_cast_fp16 = mul(x = inputs_283_cast_fp16, y = inputs_283_cast_fp16)[name = string("inputs_sq_283_cast_fp16")]; + tensor variance_283_axes_0 = const()[name = string("variance_283_axes_0"), val = tensor([1])]; + bool variance_283_keep_dims_0 = const()[name = string("variance_283_keep_dims_0"), val = bool(true)]; + tensor variance_283_cast_fp16 = reduce_mean(axes = variance_283_axes_0, keep_dims = variance_283_keep_dims_0, x = inputs_sq_283_cast_fp16)[name = string("variance_283_cast_fp16")]; + fp16 var_10394_to_fp16 = const()[name = string("op_10394_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10395_cast_fp16 = add(x = variance_283_cast_fp16, y = var_10394_to_fp16)[name = string("op_10395_cast_fp16")]; + fp32 var_10396_epsilon_0 = const()[name = string("op_10396_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10396_cast_fp16 = rsqrt(epsilon = var_10396_epsilon_0, x = var_10395_cast_fp16)[name = string("op_10396_cast_fp16")]; + tensor hidden_states_351_cast_fp16 = mul(x = inputs_283_cast_fp16, y = var_10396_cast_fp16)[name = string("hidden_states_351_cast_fp16")]; + tensor obj_301_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_351_cast_fp16)[name = string("obj_301_cast_fp16")]; + string query_205_pad_type_0 = const()[name = string("query_205_pad_type_0"), val = string("valid")]; + tensor query_205_strides_0 = const()[name = string("query_205_strides_0"), val = tensor([1, 1])]; + tensor query_205_pad_0 = const()[name = string("query_205_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_205_dilations_0 = const()[name = string("query_205_dilations_0"), val = tensor([1, 1])]; + int32 query_205_groups_0 = const()[name = string("query_205_groups_0"), val = int32(1)]; + tensor query_205_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_205_dilations_0, groups = query_205_groups_0, pad = query_205_pad_0, pad_type = query_205_pad_type_0, strides = query_205_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_301_cast_fp16)[name = string("query_205_cast_fp16")]; + string current_key_137_pad_type_0 = const()[name = string("current_key_137_pad_type_0"), val = string("valid")]; + tensor current_key_137_strides_0 = const()[name = string("current_key_137_strides_0"), val = tensor([1, 1])]; + tensor current_key_137_pad_0 = const()[name = string("current_key_137_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_137_dilations_0 = const()[name = string("current_key_137_dilations_0"), val = tensor([1, 1])]; + int32 current_key_137_groups_0 = const()[name = string("current_key_137_groups_0"), val = int32(1)]; + tensor current_key_137_cast_fp16 = conv(dilations = current_key_137_dilations_0, groups = current_key_137_groups_0, pad = current_key_137_pad_0, pad_type = current_key_137_pad_type_0, strides = current_key_137_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_301_cast_fp16)[name = string("current_key_137_cast_fp16")]; + string current_value_69_pad_type_0 = const()[name = string("current_value_69_pad_type_0"), val = string("valid")]; + tensor current_value_69_strides_0 = const()[name = string("current_value_69_strides_0"), val = tensor([1, 1])]; + tensor current_value_69_pad_0 = const()[name = string("current_value_69_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_69_dilations_0 = const()[name = string("current_value_69_dilations_0"), val = tensor([1, 1])]; + int32 current_value_69_groups_0 = const()[name = string("current_value_69_groups_0"), val = int32(1)]; + tensor current_value_69_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_69_dilations_0, groups = current_value_69_groups_0, pad = current_value_69_pad_0, pad_type = current_value_69_pad_type_0, strides = current_value_69_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_301_cast_fp16)[name = string("current_value_69_cast_fp16")]; + tensor var_10433 = const()[name = string("op_10433"), val = tensor([16, 128, 1, 1])]; + tensor inputs_285_cast_fp16 = reshape(shape = var_10433, x = query_205_cast_fp16)[name = string("inputs_285_cast_fp16")]; + tensor inputs_sq_285_cast_fp16 = mul(x = inputs_285_cast_fp16, y = inputs_285_cast_fp16)[name = string("inputs_sq_285_cast_fp16")]; + tensor variance_285_axes_0 = const()[name = string("variance_285_axes_0"), val = tensor([1])]; + bool variance_285_keep_dims_0 = const()[name = string("variance_285_keep_dims_0"), val = bool(true)]; + tensor variance_285_cast_fp16 = reduce_mean(axes = variance_285_axes_0, keep_dims = variance_285_keep_dims_0, x = inputs_sq_285_cast_fp16)[name = string("variance_285_cast_fp16")]; + fp16 var_10439_to_fp16 = const()[name = string("op_10439_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10440_cast_fp16 = add(x = variance_285_cast_fp16, y = var_10439_to_fp16)[name = string("op_10440_cast_fp16")]; + fp32 var_10441_epsilon_0 = const()[name = string("op_10441_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10441_cast_fp16 = rsqrt(epsilon = var_10441_epsilon_0, x = var_10440_cast_fp16)[name = string("op_10441_cast_fp16")]; + tensor hidden_states_353_cast_fp16 = mul(x = inputs_285_cast_fp16, y = var_10441_cast_fp16)[name = string("hidden_states_353_cast_fp16")]; + tensor query_normed_69_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_353_cast_fp16)[name = string("query_normed_69_cast_fp16")]; + tensor var_10449 = const()[name = string("op_10449"), val = tensor([8, 128, 1, 1])]; + tensor inputs_287_cast_fp16 = reshape(shape = var_10449, x = current_key_137_cast_fp16)[name = string("inputs_287_cast_fp16")]; + tensor inputs_sq_287_cast_fp16 = mul(x = inputs_287_cast_fp16, y = inputs_287_cast_fp16)[name = string("inputs_sq_287_cast_fp16")]; + tensor variance_287_axes_0 = const()[name = string("variance_287_axes_0"), val = tensor([1])]; + bool variance_287_keep_dims_0 = const()[name = string("variance_287_keep_dims_0"), val = bool(true)]; + tensor variance_287_cast_fp16 = reduce_mean(axes = variance_287_axes_0, keep_dims = variance_287_keep_dims_0, x = inputs_sq_287_cast_fp16)[name = string("variance_287_cast_fp16")]; + fp16 var_10455_to_fp16 = const()[name = string("op_10455_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10456_cast_fp16 = add(x = variance_287_cast_fp16, y = var_10455_to_fp16)[name = string("op_10456_cast_fp16")]; + fp32 var_10457_epsilon_0 = const()[name = string("op_10457_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10457_cast_fp16 = rsqrt(epsilon = var_10457_epsilon_0, x = var_10456_cast_fp16)[name = string("op_10457_cast_fp16")]; + tensor hidden_states_355_cast_fp16 = mul(x = inputs_287_cast_fp16, y = var_10457_cast_fp16)[name = string("hidden_states_355_cast_fp16")]; + tensor current_key_normed_69_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_355_cast_fp16)[name = string("current_key_normed_69_cast_fp16")]; + tensor var_10475 = const()[name = string("op_10475"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_273_cast_fp16 = reshape(shape = var_10475, x = query_normed_69_cast_fp16)[name = string("mh_q_273_cast_fp16")]; + tensor var_10477 = const()[name = string("op_10477"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_273_cast_fp16 = reshape(shape = var_10477, x = current_key_normed_69_cast_fp16)[name = string("mh_k_273_cast_fp16")]; + tensor var_10481_cast_fp16 = mul(x = mh_q_273_cast_fp16, y = cos_61_to_fp16)[name = string("op_10481_cast_fp16")]; + tensor var_10486_begin_0 = const()[name = string("op_10486_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10486_end_0 = const()[name = string("op_10486_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_10486_end_mask_0 = const()[name = string("op_10486_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_10486_cast_fp16 = slice_by_index(begin = var_10486_begin_0, end = var_10486_end_0, end_mask = var_10486_end_mask_0, x = mh_q_273_cast_fp16)[name = string("op_10486_cast_fp16")]; + tensor var_10492_begin_0 = const()[name = string("op_10492_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_10492_end_0 = const()[name = string("op_10492_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_10492_end_mask_0 = const()[name = string("op_10492_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_10492_cast_fp16 = slice_by_index(begin = var_10492_begin_0, end = var_10492_end_0, end_mask = var_10492_end_mask_0, x = mh_q_273_cast_fp16)[name = string("op_10492_cast_fp16")]; + fp16 const_700_promoted_to_fp16 = const()[name = string("const_700_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10494_cast_fp16 = mul(x = var_10492_cast_fp16, y = const_700_promoted_to_fp16)[name = string("op_10494_cast_fp16")]; + bool var_10496_interleave_0 = const()[name = string("op_10496_interleave_0"), val = bool(false)]; + tensor var_10496_cast_fp16 = concat(axis = var_10380, interleave = var_10496_interleave_0, values = (var_10494_cast_fp16, var_10486_cast_fp16))[name = string("op_10496_cast_fp16")]; + tensor var_10497_cast_fp16 = mul(x = var_10496_cast_fp16, y = sin_61_to_fp16)[name = string("op_10497_cast_fp16")]; + tensor mh_q_275_cast_fp16 = add(x = var_10481_cast_fp16, y = var_10497_cast_fp16)[name = string("mh_q_275_cast_fp16")]; + tensor var_10499_cast_fp16 = mul(x = mh_k_273_cast_fp16, y = cos_61_to_fp16)[name = string("op_10499_cast_fp16")]; + tensor var_10504_begin_0 = const()[name = string("op_10504_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10504_end_0 = const()[name = string("op_10504_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_10504_end_mask_0 = const()[name = string("op_10504_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_10504_cast_fp16 = slice_by_index(begin = var_10504_begin_0, end = var_10504_end_0, end_mask = var_10504_end_mask_0, x = mh_k_273_cast_fp16)[name = string("op_10504_cast_fp16")]; + tensor var_10510_begin_0 = const()[name = string("op_10510_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_10510_end_0 = const()[name = string("op_10510_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_10510_end_mask_0 = const()[name = string("op_10510_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_10510_cast_fp16 = slice_by_index(begin = var_10510_begin_0, end = var_10510_end_0, end_mask = var_10510_end_mask_0, x = mh_k_273_cast_fp16)[name = string("op_10510_cast_fp16")]; + fp16 const_703_promoted_to_fp16 = const()[name = string("const_703_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10512_cast_fp16 = mul(x = var_10510_cast_fp16, y = const_703_promoted_to_fp16)[name = string("op_10512_cast_fp16")]; + bool var_10514_interleave_0 = const()[name = string("op_10514_interleave_0"), val = bool(false)]; + tensor var_10514_cast_fp16 = concat(axis = var_10380, interleave = var_10514_interleave_0, values = (var_10512_cast_fp16, var_10504_cast_fp16))[name = string("op_10514_cast_fp16")]; + tensor var_10515_cast_fp16 = mul(x = var_10514_cast_fp16, y = sin_61_to_fp16)[name = string("op_10515_cast_fp16")]; + tensor mh_k_275_cast_fp16 = add(x = var_10499_cast_fp16, y = var_10515_cast_fp16)[name = string("mh_k_275_cast_fp16")]; + tensor var_10519 = const()[name = string("op_10519"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_139_cast_fp16 = reshape(shape = var_10519, x = mh_k_275_cast_fp16)[name = string("current_key_139_cast_fp16")]; + tensor var_10525_to_fp16 = const()[name = string("op_10525_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195328)))]; + tensor var_10526_cast_fp16 = mul(x = obj_303_cast_fp16, y = var_10525_to_fp16)[name = string("op_10526_cast_fp16")]; + tensor var_10523_to_fp16 = const()[name = string("op_10523_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195456)))]; + tensor var_10527_cast_fp16 = mul(x = current_key_139_cast_fp16, y = var_10523_to_fp16)[name = string("op_10527_cast_fp16")]; + tensor key_139_cast_fp16 = add(x = var_10526_cast_fp16, y = var_10527_cast_fp16)[name = string("key_139_cast_fp16")]; + tensor var_10529_to_fp16 = const()[name = string("op_10529_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195328)))]; + tensor var_10530_cast_fp16 = mul(x = obj_305_cast_fp16, y = var_10529_to_fp16)[name = string("op_10530_cast_fp16")]; + tensor var_10531_cast_fp16 = mul(x = current_value_69_cast_fp16, y = var_10523_to_fp16)[name = string("op_10531_cast_fp16")]; + tensor value_69_cast_fp16 = add(x = var_10530_cast_fp16, y = var_10531_cast_fp16)[name = string("value_69_cast_fp16")]; + fp16 var_10538_to_fp16 = const()[name = string("op_10538_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_279_cast_fp16 = mul(x = mh_q_275_cast_fp16, y = var_10538_to_fp16)[name = string("mh_q_279_cast_fp16")]; + tensor var_10540 = const()[name = string("op_10540"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_277_cast_fp16 = reshape(shape = var_10540, x = key_139_cast_fp16)[name = string("mh_k_277_cast_fp16")]; + tensor var_10542 = const()[name = string("op_10542"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_137_cast_fp16 = reshape(shape = var_10542, x = value_69_cast_fp16)[name = string("mh_v_137_cast_fp16")]; + tensor transpose_136_perm_0 = const()[name = string("transpose_136_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_68_reps_0 = const()[name = string("tile_68_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_136_cast_fp16 = transpose(perm = transpose_136_perm_0, x = mh_k_277_cast_fp16)[name = string("transpose_275")]; + tensor tile_68_cast_fp16 = tile(reps = tile_68_reps_0, x = transpose_136_cast_fp16)[name = string("tile_68_cast_fp16")]; + tensor concat_169 = const()[name = string("concat_169"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_136_cast_fp16 = reshape(shape = concat_169, x = tile_68_cast_fp16)[name = string("reshape_136_cast_fp16")]; + tensor transpose_137_perm_0 = const()[name = string("transpose_137_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_170 = const()[name = string("concat_170"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_137_cast_fp16 = transpose(perm = transpose_137_perm_0, x = reshape_136_cast_fp16)[name = string("transpose_274")]; + tensor reshape_137_cast_fp16 = reshape(shape = concat_170, x = transpose_137_cast_fp16)[name = string("reshape_137_cast_fp16")]; + tensor transpose_138_perm_0 = const()[name = string("transpose_138_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_69_reps_0 = const()[name = string("tile_69_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_138_cast_fp16 = transpose(perm = transpose_138_perm_0, x = mh_v_137_cast_fp16)[name = string("transpose_273")]; + tensor tile_69_cast_fp16 = tile(reps = tile_69_reps_0, x = transpose_138_cast_fp16)[name = string("tile_69_cast_fp16")]; + tensor concat_171 = const()[name = string("concat_171"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_138_cast_fp16 = reshape(shape = concat_171, x = tile_69_cast_fp16)[name = string("reshape_138_cast_fp16")]; + tensor transpose_139_perm_0 = const()[name = string("transpose_139_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_172 = const()[name = string("concat_172"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_139_cast_fp16 = transpose(perm = transpose_139_perm_0, x = reshape_138_cast_fp16)[name = string("transpose_272")]; + tensor reshape_139_cast_fp16 = reshape(shape = concat_172, x = transpose_139_cast_fp16)[name = string("reshape_139_cast_fp16")]; + tensor transpose_453_perm_0 = const()[name = string("transpose_453_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_205_transpose_x_1 = const()[name = string("mh_w_205_transpose_x_1"), val = bool(true)]; + bool mh_w_205_transpose_y_1 = const()[name = string("mh_w_205_transpose_y_1"), val = bool(false)]; + tensor transpose_453_cast_fp16 = transpose(perm = transpose_453_perm_0, x = reshape_137_cast_fp16)[name = string("transpose_271")]; + tensor mh_w_205_cast_fp16 = matmul(transpose_x = mh_w_205_transpose_x_1, transpose_y = mh_w_205_transpose_y_1, x = mh_q_279_cast_fp16, y = transpose_453_cast_fp16)[name = string("mh_w_205_cast_fp16")]; + tensor var_10550_to_fp16 = const()[name = string("op_10550_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195584)))]; + tensor mh_w_207_cast_fp16 = add(x = mh_w_205_cast_fp16, y = var_10550_to_fp16)[name = string("mh_w_207_cast_fp16")]; + tensor mh_w_209_cast_fp16 = softmax(axis = var_10370, x = mh_w_207_cast_fp16)[name = string("mh_w_209_cast_fp16")]; + tensor transpose_454_perm_0 = const()[name = string("transpose_454_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_69_transpose_x_1 = const()[name = string("attn_69_transpose_x_1"), val = bool(false)]; + bool attn_69_transpose_y_1 = const()[name = string("attn_69_transpose_y_1"), val = bool(true)]; + tensor transpose_454_cast_fp16 = transpose(perm = transpose_454_perm_0, x = reshape_139_cast_fp16)[name = string("transpose_270")]; + tensor attn_69_cast_fp16 = matmul(transpose_x = attn_69_transpose_x_1, transpose_y = attn_69_transpose_y_1, x = transpose_454_cast_fp16, y = mh_w_209_cast_fp16)[name = string("attn_69_cast_fp16")]; + tensor var_10556 = const()[name = string("op_10556"), val = tensor([1, 2048, 1, 1])]; + tensor input_293_cast_fp16 = reshape(shape = var_10556, x = attn_69_cast_fp16)[name = string("input_293_cast_fp16")]; + string obj_307_pad_type_0 = const()[name = string("obj_307_pad_type_0"), val = string("valid")]; + tensor obj_307_strides_0 = const()[name = string("obj_307_strides_0"), val = tensor([1, 1])]; + tensor obj_307_pad_0 = const()[name = string("obj_307_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_307_dilations_0 = const()[name = string("obj_307_dilations_0"), val = tensor([1, 1])]; + int32 obj_307_groups_0 = const()[name = string("obj_307_groups_0"), val = int32(1)]; + tensor obj_307_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_307_dilations_0, groups = obj_307_groups_0, pad = obj_307_pad_0, pad_type = obj_307_pad_type_0, strides = obj_307_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_293_cast_fp16)[name = string("obj_307_cast_fp16")]; + tensor inputs_289_cast_fp16 = add(x = inputs_283_cast_fp16, y = obj_307_cast_fp16)[name = string("inputs_289_cast_fp16")]; + tensor inputs_sq_289_cast_fp16 = mul(x = inputs_289_cast_fp16, y = inputs_289_cast_fp16)[name = string("inputs_sq_289_cast_fp16")]; + tensor variance_289_axes_0 = const()[name = string("variance_289_axes_0"), val = tensor([1])]; + bool variance_289_keep_dims_0 = const()[name = string("variance_289_keep_dims_0"), val = bool(true)]; + tensor variance_289_cast_fp16 = reduce_mean(axes = variance_289_axes_0, keep_dims = variance_289_keep_dims_0, x = inputs_sq_289_cast_fp16)[name = string("variance_289_cast_fp16")]; + fp16 var_10574_to_fp16 = const()[name = string("op_10574_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10575_cast_fp16 = add(x = variance_289_cast_fp16, y = var_10574_to_fp16)[name = string("op_10575_cast_fp16")]; + fp32 var_10576_epsilon_0 = const()[name = string("op_10576_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10576_cast_fp16 = rsqrt(epsilon = var_10576_epsilon_0, x = var_10575_cast_fp16)[name = string("op_10576_cast_fp16")]; + tensor hidden_states_357_cast_fp16 = mul(x = inputs_289_cast_fp16, y = var_10576_cast_fp16)[name = string("hidden_states_357_cast_fp16")]; + tensor input_295_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_357_cast_fp16)[name = string("input_295_cast_fp16")]; + string input_297_pad_type_0 = const()[name = string("input_297_pad_type_0"), val = string("valid")]; + tensor input_297_strides_0 = const()[name = string("input_297_strides_0"), val = tensor([1, 1])]; + tensor input_297_pad_0 = const()[name = string("input_297_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_297_dilations_0 = const()[name = string("input_297_dilations_0"), val = tensor([1, 1])]; + int32 input_297_groups_0 = const()[name = string("input_297_groups_0"), val = int32(1)]; + tensor input_297_cast_fp16 = conv(dilations = input_297_dilations_0, groups = input_297_groups_0, pad = input_297_pad_0, pad_type = input_297_pad_type_0, strides = input_297_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_295_cast_fp16)[name = string("input_297_cast_fp16")]; + tensor var_10590_cast_fp16 = silu(x = input_297_cast_fp16)[name = string("op_10590_cast_fp16")]; + string var_10596_pad_type_0 = const()[name = string("op_10596_pad_type_0"), val = string("valid")]; + tensor var_10596_strides_0 = const()[name = string("op_10596_strides_0"), val = tensor([1, 1])]; + tensor var_10596_pad_0 = const()[name = string("op_10596_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10596_dilations_0 = const()[name = string("op_10596_dilations_0"), val = tensor([1, 1])]; + int32 var_10596_groups_0 = const()[name = string("op_10596_groups_0"), val = int32(1)]; + tensor var_10596_cast_fp16 = conv(dilations = var_10596_dilations_0, groups = var_10596_groups_0, pad = var_10596_pad_0, pad_type = var_10596_pad_type_0, strides = var_10596_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_295_cast_fp16)[name = string("op_10596_cast_fp16")]; + tensor input_299_cast_fp16 = mul(x = var_10590_cast_fp16, y = var_10596_cast_fp16)[name = string("input_299_cast_fp16")]; + string hidden_states_359_pad_type_0 = const()[name = string("hidden_states_359_pad_type_0"), val = string("valid")]; + tensor hidden_states_359_strides_0 = const()[name = string("hidden_states_359_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_359_pad_0 = const()[name = string("hidden_states_359_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_359_dilations_0 = const()[name = string("hidden_states_359_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_359_groups_0 = const()[name = string("hidden_states_359_groups_0"), val = int32(1)]; + tensor hidden_states_359_cast_fp16 = conv(dilations = hidden_states_359_dilations_0, groups = hidden_states_359_groups_0, pad = hidden_states_359_pad_0, pad_type = hidden_states_359_pad_type_0, strides = hidden_states_359_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_299_cast_fp16)[name = string("hidden_states_359_cast_fp16")]; + tensor inputs_291_cast_fp16 = add(x = inputs_289_cast_fp16, y = hidden_states_359_cast_fp16)[name = string("inputs_291_cast_fp16")]; + int32 var_10624 = const()[name = string("op_10624"), val = int32(1)]; + bool key_caches_15_interleave_0 = const()[name = string("key_caches_15_interleave_0"), val = bool(false)]; + tensor key_caches_15_cast_fp16 = concat(axis = var_10624, interleave = key_caches_15_interleave_0, values = (key_123_cast_fp16, key_127_cast_fp16, key_131_cast_fp16, key_135_cast_fp16, key_139_cast_fp16))[name = string("key_caches_15_cast_fp16")]; + int32 var_10627 = const()[name = string("op_10627"), val = int32(1)]; + bool value_caches_15_interleave_0 = const()[name = string("value_caches_15_interleave_0"), val = bool(false)]; + tensor value_caches_15_cast_fp16 = concat(axis = var_10627, interleave = value_caches_15_interleave_0, values = (value_61_cast_fp16, value_63_cast_fp16, value_65_cast_fp16, value_67_cast_fp16, value_69_cast_fp16))[name = string("value_caches_15_cast_fp16")]; + tensor inputs_sq_291_cast_fp16 = mul(x = inputs_291_cast_fp16, y = inputs_291_cast_fp16)[name = string("inputs_sq_291_cast_fp16")]; + tensor variance_291_axes_0 = const()[name = string("variance_291_axes_0"), val = tensor([1])]; + bool variance_291_keep_dims_0 = const()[name = string("variance_291_keep_dims_0"), val = bool(true)]; + tensor variance_291_cast_fp16 = reduce_mean(axes = variance_291_axes_0, keep_dims = variance_291_keep_dims_0, x = inputs_sq_291_cast_fp16)[name = string("variance_291_cast_fp16")]; + fp16 var_10637_to_fp16 = const()[name = string("op_10637_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10638_cast_fp16 = add(x = variance_291_cast_fp16, y = var_10637_to_fp16)[name = string("op_10638_cast_fp16")]; + fp32 var_10639_epsilon_0 = const()[name = string("op_10639_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10639_cast_fp16 = rsqrt(epsilon = var_10639_epsilon_0, x = var_10638_cast_fp16)[name = string("op_10639_cast_fp16")]; + tensor hidden_states_361_cast_fp16 = mul(x = inputs_291_cast_fp16, y = var_10639_cast_fp16)[name = string("hidden_states_361_cast_fp16")]; + tensor input_301_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_361_cast_fp16)[name = string("input_301_cast_fp16")]; + string logits_21_pad_type_0 = const()[name = string("logits_21_pad_type_0"), val = string("valid")]; + tensor logits_21_strides_0 = const()[name = string("logits_21_strides_0"), val = tensor([1, 1])]; + tensor logits_21_pad_0 = const()[name = string("logits_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_21_dilations_0 = const()[name = string("logits_21_dilations_0"), val = tensor([1, 1])]; + int32 logits_21_groups_0 = const()[name = string("logits_21_groups_0"), val = int32(1)]; + tensor lm_heads_5_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91295552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93392768))))[name = string("lm_heads_5_weight_to_fp16_palettized")]; + tensor logits_21_cast_fp16 = conv(dilations = logits_21_dilations_0, groups = logits_21_groups_0, pad = logits_21_pad_0, pad_type = logits_21_pad_type_0, strides = logits_21_strides_0, weight = lm_heads_5_weight_to_fp16_palettized, x = input_301_cast_fp16)[name = string("logits_21_cast_fp16")]; + tensor var_10657 = const()[name = string("op_10657"), val = tensor([1, 2048])]; + tensor logits_23_cast_fp16 = reshape(shape = var_10657, x = logits_21_cast_fp16)[name = string("logits_23_cast_fp16")]; + tensor scaled_logits_11_cast_fp16 = real_div(x = logits_23_cast_fp16, y = temperature)[name = string("scaled_logits_11_cast_fp16")]; + int32 var_10667 = const()[name = string("op_10667"), val = int32(100)]; + int32 top_values_11_axis_0 = const()[name = string("top_values_11_axis_0"), val = int32(1)]; + bool top_values_11_ascending_0 = const()[name = string("top_values_11_ascending_0"), val = bool(false)]; + bool top_values_11_sort_0 = const()[name = string("top_values_11_sort_0"), val = bool(true)]; + bool top_values_11_return_indices_0 = const()[name = string("top_values_11_return_indices_0"), val = bool(true)]; + string top_values_11_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_11_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_11_cast_fp16_cast_uint16_0, tensor top_values_11_cast_fp16_cast_uint16_1 = topk(ascending = top_values_11_ascending_0, axis = top_values_11_axis_0, k = var_10667, output_indices_dtype = top_values_11_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_11_return_indices_0, sort = top_values_11_sort_0, x = scaled_logits_11_cast_fp16)[name = string("top_values_11_cast_fp16_cast_uint16")]; + tensor var_10673_cast_fp16 = mul(x = top_values_11_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_10673_cast_fp16")]; + tensor var_10677_cast_fp16 = add(x = var_10673_cast_fp16, y = var_3001_cast_fp16)[name = string("op_10677_cast_fp16")]; + tensor reduce_min_5_axes_0 = const()[name = string("reduce_min_5_axes_0"), val = tensor([1])]; + bool reduce_min_5_keep_dims_0 = const()[name = string("reduce_min_5_keep_dims_0"), val = bool(true)]; + tensor reduce_min_5_cast_fp16 = reduce_min(axes = reduce_min_5_axes_0, keep_dims = reduce_min_5_keep_dims_0, x = var_10677_cast_fp16)[name = string("reduce_min_5_cast_fp16")]; + tensor var_10680_cast_fp16 = greater_equal(x = scaled_logits_11_cast_fp16, y = reduce_min_5_cast_fp16)[name = string("op_10680_cast_fp16")]; + fp16 var_10681_value_0_to_fp16 = const()[name = string("op_10681_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_10681_cast_fp16 = fill_like(ref_tensor = scaled_logits_11_cast_fp16, value = var_10681_value_0_to_fp16)[name = string("op_10681_cast_fp16")]; + tensor masked_logits_11_cast_fp16 = select(a = scaled_logits_11_cast_fp16, b = var_10681_cast_fp16, cond = var_10680_cast_fp16)[name = string("masked_logits_11_cast_fp16")]; + tensor var_10685_begin_0 = const()[name = string("op_10685_begin_0"), val = tensor([5, 0])]; + tensor var_10685_end_0 = const()[name = string("op_10685_end_0"), val = tensor([6, 2048])]; + tensor var_10685_end_mask_0 = const()[name = string("op_10685_end_mask_0"), val = tensor([false, true])]; + tensor var_10685_squeeze_mask_0 = const()[name = string("op_10685_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_10685_cast_fp16 = slice_by_index(begin = var_10685_begin_0, end = var_10685_end_0, end_mask = var_10685_end_mask_0, squeeze_mask = var_10685_squeeze_mask_0, x = gumbel)[name = string("op_10685_cast_fp16")]; + tensor var_10688 = const()[name = string("op_10688"), val = tensor([1, 2048])]; + tensor var_10689_cast_fp16 = reshape(shape = var_10688, x = var_10685_cast_fp16)[name = string("op_10689_cast_fp16")]; + tensor noisy_logits_11_cast_fp16 = add(x = masked_logits_11_cast_fp16, y = var_10689_cast_fp16)[name = string("noisy_logits_11_cast_fp16")]; + int32 code_11_axis_0 = const()[name = string("code_11_axis_0"), val = int32(1)]; + bool code_11_keep_dims_0 = const()[name = string("code_11_keep_dims_0"), val = bool(false)]; + string code_11_output_dtype_0 = const()[name = string("code_11_output_dtype_0"), val = string("int32")]; + tensor code_11_cast_fp16 = reduce_argmax(axis = code_11_axis_0, keep_dims = code_11_keep_dims_0, output_dtype = code_11_output_dtype_0, x = noisy_logits_11_cast_fp16)[name = string("code_11_cast_fp16")]; + int32 var_10700 = const()[name = string("op_10700"), val = int32(10240)]; + tensor input_303 = add(x = code_11_cast_fp16, y = var_10700)[name = string("input_303")]; + int32 code_embed_21_axis_0 = const()[name = string("code_embed_21_axis_0"), val = int32(0)]; + int32 code_embed_21_batch_dims_0 = const()[name = string("code_embed_21_batch_dims_0"), val = int32(0)]; + bool code_embed_21_validate_indices_0 = const()[name = string("code_embed_21_validate_indices_0"), val = bool(false)]; + string input_303_to_uint16_dtype_0 = const()[name = string("input_303_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_303_to_uint16 = cast(dtype = input_303_to_uint16_dtype_0, x = input_303)[name = string("cast_9")]; + tensor code_embed_21_cast_fp16_cast_uint16 = gather(axis = code_embed_21_axis_0, batch_dims = code_embed_21_batch_dims_0, indices = input_303_to_uint16, validate_indices = code_embed_21_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_21_cast_fp16_cast_uint16")]; + tensor var_10704 = const()[name = string("op_10704"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_23_cast_fp16 = reshape(shape = var_10704, x = code_embed_21_cast_fp16_cast_uint16)[name = string("code_embed_23_cast_fp16")]; + tensor embed_sum_13_cast_fp16 = add(x = embed_sum_11_cast_fp16, y = code_embed_23_cast_fp16)[name = string("embed_sum_13_cast_fp16")]; + string inputs_293_pad_type_0 = const()[name = string("inputs_293_pad_type_0"), val = string("valid")]; + tensor inputs_293_strides_0 = const()[name = string("inputs_293_strides_0"), val = tensor([1, 1])]; + tensor inputs_293_pad_0 = const()[name = string("inputs_293_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_293_dilations_0 = const()[name = string("inputs_293_dilations_0"), val = tensor([1, 1])]; + int32 inputs_293_groups_0 = const()[name = string("inputs_293_groups_0"), val = int32(1)]; + tensor inputs_293_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_293_dilations_0, groups = inputs_293_groups_0, pad = inputs_293_pad_0, pad_type = inputs_293_pad_type_0, strides = inputs_293_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_23_cast_fp16)[name = string("inputs_293_cast_fp16")]; + tensor obj_311_begin_0 = const()[name = string("obj_311_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_311_end_0 = const()[name = string("obj_311_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_311_end_mask_0 = const()[name = string("obj_311_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_311_cast_fp16 = slice_by_index(begin = obj_311_begin_0, end = obj_311_end_0, end_mask = obj_311_end_mask_0, x = key_caches_15_cast_fp16)[name = string("obj_311_cast_fp16")]; + tensor obj_313_begin_0 = const()[name = string("obj_313_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_313_end_0 = const()[name = string("obj_313_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_313_end_mask_0 = const()[name = string("obj_313_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_313_cast_fp16 = slice_by_index(begin = obj_313_begin_0, end = obj_313_end_0, end_mask = obj_313_end_mask_0, x = value_caches_15_cast_fp16)[name = string("obj_313_cast_fp16")]; + int32 var_10809 = const()[name = string("op_10809"), val = int32(3)]; + int32 var_10819 = const()[name = string("op_10819"), val = int32(-2)]; + tensor inputs_sq_293_cast_fp16 = mul(x = inputs_293_cast_fp16, y = inputs_293_cast_fp16)[name = string("inputs_sq_293_cast_fp16")]; + tensor variance_293_axes_0 = const()[name = string("variance_293_axes_0"), val = tensor([1])]; + bool variance_293_keep_dims_0 = const()[name = string("variance_293_keep_dims_0"), val = bool(true)]; + tensor variance_293_cast_fp16 = reduce_mean(axes = variance_293_axes_0, keep_dims = variance_293_keep_dims_0, x = inputs_sq_293_cast_fp16)[name = string("variance_293_cast_fp16")]; + fp16 var_10833_to_fp16 = const()[name = string("op_10833_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10834_cast_fp16 = add(x = variance_293_cast_fp16, y = var_10833_to_fp16)[name = string("op_10834_cast_fp16")]; + fp32 var_10835_epsilon_0 = const()[name = string("op_10835_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10835_cast_fp16 = rsqrt(epsilon = var_10835_epsilon_0, x = var_10834_cast_fp16)[name = string("op_10835_cast_fp16")]; + tensor hidden_states_363_cast_fp16 = mul(x = inputs_293_cast_fp16, y = var_10835_cast_fp16)[name = string("hidden_states_363_cast_fp16")]; + tensor obj_309_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_363_cast_fp16)[name = string("obj_309_cast_fp16")]; + string query_211_pad_type_0 = const()[name = string("query_211_pad_type_0"), val = string("valid")]; + tensor query_211_strides_0 = const()[name = string("query_211_strides_0"), val = tensor([1, 1])]; + tensor query_211_pad_0 = const()[name = string("query_211_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_211_dilations_0 = const()[name = string("query_211_dilations_0"), val = tensor([1, 1])]; + int32 query_211_groups_0 = const()[name = string("query_211_groups_0"), val = int32(1)]; + tensor query_211_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_211_dilations_0, groups = query_211_groups_0, pad = query_211_pad_0, pad_type = query_211_pad_type_0, strides = query_211_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_309_cast_fp16)[name = string("query_211_cast_fp16")]; + string current_key_141_pad_type_0 = const()[name = string("current_key_141_pad_type_0"), val = string("valid")]; + tensor current_key_141_strides_0 = const()[name = string("current_key_141_strides_0"), val = tensor([1, 1])]; + tensor current_key_141_pad_0 = const()[name = string("current_key_141_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_141_dilations_0 = const()[name = string("current_key_141_dilations_0"), val = tensor([1, 1])]; + int32 current_key_141_groups_0 = const()[name = string("current_key_141_groups_0"), val = int32(1)]; + tensor current_key_141_cast_fp16 = conv(dilations = current_key_141_dilations_0, groups = current_key_141_groups_0, pad = current_key_141_pad_0, pad_type = current_key_141_pad_type_0, strides = current_key_141_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_309_cast_fp16)[name = string("current_key_141_cast_fp16")]; + string current_value_71_pad_type_0 = const()[name = string("current_value_71_pad_type_0"), val = string("valid")]; + tensor current_value_71_strides_0 = const()[name = string("current_value_71_strides_0"), val = tensor([1, 1])]; + tensor current_value_71_pad_0 = const()[name = string("current_value_71_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_71_dilations_0 = const()[name = string("current_value_71_dilations_0"), val = tensor([1, 1])]; + int32 current_value_71_groups_0 = const()[name = string("current_value_71_groups_0"), val = int32(1)]; + tensor current_value_71_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_71_dilations_0, groups = current_value_71_groups_0, pad = current_value_71_pad_0, pad_type = current_value_71_pad_type_0, strides = current_value_71_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_309_cast_fp16)[name = string("current_value_71_cast_fp16")]; + tensor var_10872 = const()[name = string("op_10872"), val = tensor([16, 128, 1, 1])]; + tensor inputs_295_cast_fp16 = reshape(shape = var_10872, x = query_211_cast_fp16)[name = string("inputs_295_cast_fp16")]; + tensor inputs_sq_295_cast_fp16 = mul(x = inputs_295_cast_fp16, y = inputs_295_cast_fp16)[name = string("inputs_sq_295_cast_fp16")]; + tensor variance_295_axes_0 = const()[name = string("variance_295_axes_0"), val = tensor([1])]; + bool variance_295_keep_dims_0 = const()[name = string("variance_295_keep_dims_0"), val = bool(true)]; + tensor variance_295_cast_fp16 = reduce_mean(axes = variance_295_axes_0, keep_dims = variance_295_keep_dims_0, x = inputs_sq_295_cast_fp16)[name = string("variance_295_cast_fp16")]; + fp16 var_10878_to_fp16 = const()[name = string("op_10878_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10879_cast_fp16 = add(x = variance_295_cast_fp16, y = var_10878_to_fp16)[name = string("op_10879_cast_fp16")]; + fp32 var_10880_epsilon_0 = const()[name = string("op_10880_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10880_cast_fp16 = rsqrt(epsilon = var_10880_epsilon_0, x = var_10879_cast_fp16)[name = string("op_10880_cast_fp16")]; + tensor hidden_states_365_cast_fp16 = mul(x = inputs_295_cast_fp16, y = var_10880_cast_fp16)[name = string("hidden_states_365_cast_fp16")]; + tensor query_normed_71_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_365_cast_fp16)[name = string("query_normed_71_cast_fp16")]; + tensor var_10888 = const()[name = string("op_10888"), val = tensor([8, 128, 1, 1])]; + tensor inputs_297_cast_fp16 = reshape(shape = var_10888, x = current_key_141_cast_fp16)[name = string("inputs_297_cast_fp16")]; + tensor inputs_sq_297_cast_fp16 = mul(x = inputs_297_cast_fp16, y = inputs_297_cast_fp16)[name = string("inputs_sq_297_cast_fp16")]; + tensor variance_297_axes_0 = const()[name = string("variance_297_axes_0"), val = tensor([1])]; + bool variance_297_keep_dims_0 = const()[name = string("variance_297_keep_dims_0"), val = bool(true)]; + tensor variance_297_cast_fp16 = reduce_mean(axes = variance_297_axes_0, keep_dims = variance_297_keep_dims_0, x = inputs_sq_297_cast_fp16)[name = string("variance_297_cast_fp16")]; + fp16 var_10894_to_fp16 = const()[name = string("op_10894_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_10895_cast_fp16 = add(x = variance_297_cast_fp16, y = var_10894_to_fp16)[name = string("op_10895_cast_fp16")]; + fp32 var_10896_epsilon_0 = const()[name = string("op_10896_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_10896_cast_fp16 = rsqrt(epsilon = var_10896_epsilon_0, x = var_10895_cast_fp16)[name = string("op_10896_cast_fp16")]; + tensor hidden_states_367_cast_fp16 = mul(x = inputs_297_cast_fp16, y = var_10896_cast_fp16)[name = string("hidden_states_367_cast_fp16")]; + tensor current_key_normed_71_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_367_cast_fp16)[name = string("current_key_normed_71_cast_fp16")]; + tensor var_10914 = const()[name = string("op_10914"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_281_cast_fp16 = reshape(shape = var_10914, x = query_normed_71_cast_fp16)[name = string("mh_q_281_cast_fp16")]; + tensor var_10916 = const()[name = string("op_10916"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_281_cast_fp16 = reshape(shape = var_10916, x = current_key_normed_71_cast_fp16)[name = string("mh_k_281_cast_fp16")]; + tensor cos_71_to_fp16 = const()[name = string("cos_71_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175195712)))]; + tensor var_10920_cast_fp16 = mul(x = mh_q_281_cast_fp16, y = cos_71_to_fp16)[name = string("op_10920_cast_fp16")]; + tensor var_10925_begin_0 = const()[name = string("op_10925_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10925_end_0 = const()[name = string("op_10925_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_10925_end_mask_0 = const()[name = string("op_10925_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_10925_cast_fp16 = slice_by_index(begin = var_10925_begin_0, end = var_10925_end_0, end_mask = var_10925_end_mask_0, x = mh_q_281_cast_fp16)[name = string("op_10925_cast_fp16")]; + tensor var_10931_begin_0 = const()[name = string("op_10931_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_10931_end_0 = const()[name = string("op_10931_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_10931_end_mask_0 = const()[name = string("op_10931_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_10931_cast_fp16 = slice_by_index(begin = var_10931_begin_0, end = var_10931_end_0, end_mask = var_10931_end_mask_0, x = mh_q_281_cast_fp16)[name = string("op_10931_cast_fp16")]; + fp16 const_721_promoted_to_fp16 = const()[name = string("const_721_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10933_cast_fp16 = mul(x = var_10931_cast_fp16, y = const_721_promoted_to_fp16)[name = string("op_10933_cast_fp16")]; + bool var_10935_interleave_0 = const()[name = string("op_10935_interleave_0"), val = bool(false)]; + tensor var_10935_cast_fp16 = concat(axis = var_10819, interleave = var_10935_interleave_0, values = (var_10933_cast_fp16, var_10925_cast_fp16))[name = string("op_10935_cast_fp16")]; + tensor sin_71_to_fp16 = const()[name = string("sin_71_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196032)))]; + tensor var_10936_cast_fp16 = mul(x = var_10935_cast_fp16, y = sin_71_to_fp16)[name = string("op_10936_cast_fp16")]; + tensor mh_q_283_cast_fp16 = add(x = var_10920_cast_fp16, y = var_10936_cast_fp16)[name = string("mh_q_283_cast_fp16")]; + tensor var_10938_cast_fp16 = mul(x = mh_k_281_cast_fp16, y = cos_71_to_fp16)[name = string("op_10938_cast_fp16")]; + tensor var_10943_begin_0 = const()[name = string("op_10943_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_10943_end_0 = const()[name = string("op_10943_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_10943_end_mask_0 = const()[name = string("op_10943_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_10943_cast_fp16 = slice_by_index(begin = var_10943_begin_0, end = var_10943_end_0, end_mask = var_10943_end_mask_0, x = mh_k_281_cast_fp16)[name = string("op_10943_cast_fp16")]; + tensor var_10949_begin_0 = const()[name = string("op_10949_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_10949_end_0 = const()[name = string("op_10949_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_10949_end_mask_0 = const()[name = string("op_10949_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_10949_cast_fp16 = slice_by_index(begin = var_10949_begin_0, end = var_10949_end_0, end_mask = var_10949_end_mask_0, x = mh_k_281_cast_fp16)[name = string("op_10949_cast_fp16")]; + fp16 const_724_promoted_to_fp16 = const()[name = string("const_724_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_10951_cast_fp16 = mul(x = var_10949_cast_fp16, y = const_724_promoted_to_fp16)[name = string("op_10951_cast_fp16")]; + bool var_10953_interleave_0 = const()[name = string("op_10953_interleave_0"), val = bool(false)]; + tensor var_10953_cast_fp16 = concat(axis = var_10819, interleave = var_10953_interleave_0, values = (var_10951_cast_fp16, var_10943_cast_fp16))[name = string("op_10953_cast_fp16")]; + tensor var_10954_cast_fp16 = mul(x = var_10953_cast_fp16, y = sin_71_to_fp16)[name = string("op_10954_cast_fp16")]; + tensor mh_k_283_cast_fp16 = add(x = var_10938_cast_fp16, y = var_10954_cast_fp16)[name = string("mh_k_283_cast_fp16")]; + tensor var_10958 = const()[name = string("op_10958"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_143_cast_fp16 = reshape(shape = var_10958, x = mh_k_283_cast_fp16)[name = string("current_key_143_cast_fp16")]; + tensor var_10964_to_fp16 = const()[name = string("op_10964_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196352)))]; + tensor var_10965_cast_fp16 = mul(x = obj_311_cast_fp16, y = var_10964_to_fp16)[name = string("op_10965_cast_fp16")]; + tensor var_10962_to_fp16 = const()[name = string("op_10962_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196480)))]; + tensor var_10966_cast_fp16 = mul(x = current_key_143_cast_fp16, y = var_10962_to_fp16)[name = string("op_10966_cast_fp16")]; + tensor key_143_cast_fp16 = add(x = var_10965_cast_fp16, y = var_10966_cast_fp16)[name = string("key_143_cast_fp16")]; + tensor var_10968_to_fp16 = const()[name = string("op_10968_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196352)))]; + tensor var_10969_cast_fp16 = mul(x = obj_313_cast_fp16, y = var_10968_to_fp16)[name = string("op_10969_cast_fp16")]; + tensor var_10970_cast_fp16 = mul(x = current_value_71_cast_fp16, y = var_10962_to_fp16)[name = string("op_10970_cast_fp16")]; + tensor value_71_cast_fp16 = add(x = var_10969_cast_fp16, y = var_10970_cast_fp16)[name = string("value_71_cast_fp16")]; + fp16 var_10977_to_fp16 = const()[name = string("op_10977_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_287_cast_fp16 = mul(x = mh_q_283_cast_fp16, y = var_10977_to_fp16)[name = string("mh_q_287_cast_fp16")]; + tensor var_10979 = const()[name = string("op_10979"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_285_cast_fp16 = reshape(shape = var_10979, x = key_143_cast_fp16)[name = string("mh_k_285_cast_fp16")]; + tensor var_10981 = const()[name = string("op_10981"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_141_cast_fp16 = reshape(shape = var_10981, x = value_71_cast_fp16)[name = string("mh_v_141_cast_fp16")]; + tensor transpose_140_perm_0 = const()[name = string("transpose_140_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_70_reps_0 = const()[name = string("tile_70_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_140_cast_fp16 = transpose(perm = transpose_140_perm_0, x = mh_k_285_cast_fp16)[name = string("transpose_269")]; + tensor tile_70_cast_fp16 = tile(reps = tile_70_reps_0, x = transpose_140_cast_fp16)[name = string("tile_70_cast_fp16")]; + tensor concat_178 = const()[name = string("concat_178"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_140_cast_fp16 = reshape(shape = concat_178, x = tile_70_cast_fp16)[name = string("reshape_140_cast_fp16")]; + tensor transpose_141_perm_0 = const()[name = string("transpose_141_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_179 = const()[name = string("concat_179"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_141_cast_fp16 = transpose(perm = transpose_141_perm_0, x = reshape_140_cast_fp16)[name = string("transpose_268")]; + tensor reshape_141_cast_fp16 = reshape(shape = concat_179, x = transpose_141_cast_fp16)[name = string("reshape_141_cast_fp16")]; + tensor transpose_142_perm_0 = const()[name = string("transpose_142_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_71_reps_0 = const()[name = string("tile_71_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_142_cast_fp16 = transpose(perm = transpose_142_perm_0, x = mh_v_141_cast_fp16)[name = string("transpose_267")]; + tensor tile_71_cast_fp16 = tile(reps = tile_71_reps_0, x = transpose_142_cast_fp16)[name = string("tile_71_cast_fp16")]; + tensor concat_180 = const()[name = string("concat_180"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_142_cast_fp16 = reshape(shape = concat_180, x = tile_71_cast_fp16)[name = string("reshape_142_cast_fp16")]; + tensor transpose_143_perm_0 = const()[name = string("transpose_143_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_181 = const()[name = string("concat_181"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_143_cast_fp16 = transpose(perm = transpose_143_perm_0, x = reshape_142_cast_fp16)[name = string("transpose_266")]; + tensor reshape_143_cast_fp16 = reshape(shape = concat_181, x = transpose_143_cast_fp16)[name = string("reshape_143_cast_fp16")]; + tensor transpose_457_perm_0 = const()[name = string("transpose_457_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_211_transpose_x_1 = const()[name = string("mh_w_211_transpose_x_1"), val = bool(true)]; + bool mh_w_211_transpose_y_1 = const()[name = string("mh_w_211_transpose_y_1"), val = bool(false)]; + tensor transpose_457_cast_fp16 = transpose(perm = transpose_457_perm_0, x = reshape_141_cast_fp16)[name = string("transpose_265")]; + tensor mh_w_211_cast_fp16 = matmul(transpose_x = mh_w_211_transpose_x_1, transpose_y = mh_w_211_transpose_y_1, x = mh_q_287_cast_fp16, y = transpose_457_cast_fp16)[name = string("mh_w_211_cast_fp16")]; + tensor var_10989_to_fp16 = const()[name = string("op_10989_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196608)))]; + tensor mh_w_213_cast_fp16 = add(x = mh_w_211_cast_fp16, y = var_10989_to_fp16)[name = string("mh_w_213_cast_fp16")]; + tensor mh_w_215_cast_fp16 = softmax(axis = var_10809, x = mh_w_213_cast_fp16)[name = string("mh_w_215_cast_fp16")]; + tensor transpose_458_perm_0 = const()[name = string("transpose_458_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_71_transpose_x_1 = const()[name = string("attn_71_transpose_x_1"), val = bool(false)]; + bool attn_71_transpose_y_1 = const()[name = string("attn_71_transpose_y_1"), val = bool(true)]; + tensor transpose_458_cast_fp16 = transpose(perm = transpose_458_perm_0, x = reshape_143_cast_fp16)[name = string("transpose_264")]; + tensor attn_71_cast_fp16 = matmul(transpose_x = attn_71_transpose_x_1, transpose_y = attn_71_transpose_y_1, x = transpose_458_cast_fp16, y = mh_w_215_cast_fp16)[name = string("attn_71_cast_fp16")]; + tensor var_10995 = const()[name = string("op_10995"), val = tensor([1, 2048, 1, 1])]; + tensor input_305_cast_fp16 = reshape(shape = var_10995, x = attn_71_cast_fp16)[name = string("input_305_cast_fp16")]; + string obj_319_pad_type_0 = const()[name = string("obj_319_pad_type_0"), val = string("valid")]; + tensor obj_319_strides_0 = const()[name = string("obj_319_strides_0"), val = tensor([1, 1])]; + tensor obj_319_pad_0 = const()[name = string("obj_319_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_319_dilations_0 = const()[name = string("obj_319_dilations_0"), val = tensor([1, 1])]; + int32 obj_319_groups_0 = const()[name = string("obj_319_groups_0"), val = int32(1)]; + tensor obj_319_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_319_dilations_0, groups = obj_319_groups_0, pad = obj_319_pad_0, pad_type = obj_319_pad_type_0, strides = obj_319_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_305_cast_fp16)[name = string("obj_319_cast_fp16")]; + tensor inputs_299_cast_fp16 = add(x = inputs_293_cast_fp16, y = obj_319_cast_fp16)[name = string("inputs_299_cast_fp16")]; + tensor inputs_sq_299_cast_fp16 = mul(x = inputs_299_cast_fp16, y = inputs_299_cast_fp16)[name = string("inputs_sq_299_cast_fp16")]; + tensor variance_299_axes_0 = const()[name = string("variance_299_axes_0"), val = tensor([1])]; + bool variance_299_keep_dims_0 = const()[name = string("variance_299_keep_dims_0"), val = bool(true)]; + tensor variance_299_cast_fp16 = reduce_mean(axes = variance_299_axes_0, keep_dims = variance_299_keep_dims_0, x = inputs_sq_299_cast_fp16)[name = string("variance_299_cast_fp16")]; + fp16 var_11013_to_fp16 = const()[name = string("op_11013_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11014_cast_fp16 = add(x = variance_299_cast_fp16, y = var_11013_to_fp16)[name = string("op_11014_cast_fp16")]; + fp32 var_11015_epsilon_0 = const()[name = string("op_11015_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11015_cast_fp16 = rsqrt(epsilon = var_11015_epsilon_0, x = var_11014_cast_fp16)[name = string("op_11015_cast_fp16")]; + tensor hidden_states_369_cast_fp16 = mul(x = inputs_299_cast_fp16, y = var_11015_cast_fp16)[name = string("hidden_states_369_cast_fp16")]; + tensor input_307_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_369_cast_fp16)[name = string("input_307_cast_fp16")]; + string input_309_pad_type_0 = const()[name = string("input_309_pad_type_0"), val = string("valid")]; + tensor input_309_strides_0 = const()[name = string("input_309_strides_0"), val = tensor([1, 1])]; + tensor input_309_pad_0 = const()[name = string("input_309_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_309_dilations_0 = const()[name = string("input_309_dilations_0"), val = tensor([1, 1])]; + int32 input_309_groups_0 = const()[name = string("input_309_groups_0"), val = int32(1)]; + tensor input_309_cast_fp16 = conv(dilations = input_309_dilations_0, groups = input_309_groups_0, pad = input_309_pad_0, pad_type = input_309_pad_type_0, strides = input_309_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_307_cast_fp16)[name = string("input_309_cast_fp16")]; + tensor var_11029_cast_fp16 = silu(x = input_309_cast_fp16)[name = string("op_11029_cast_fp16")]; + string var_11035_pad_type_0 = const()[name = string("op_11035_pad_type_0"), val = string("valid")]; + tensor var_11035_strides_0 = const()[name = string("op_11035_strides_0"), val = tensor([1, 1])]; + tensor var_11035_pad_0 = const()[name = string("op_11035_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11035_dilations_0 = const()[name = string("op_11035_dilations_0"), val = tensor([1, 1])]; + int32 var_11035_groups_0 = const()[name = string("op_11035_groups_0"), val = int32(1)]; + tensor var_11035_cast_fp16 = conv(dilations = var_11035_dilations_0, groups = var_11035_groups_0, pad = var_11035_pad_0, pad_type = var_11035_pad_type_0, strides = var_11035_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_307_cast_fp16)[name = string("op_11035_cast_fp16")]; + tensor input_311_cast_fp16 = mul(x = var_11029_cast_fp16, y = var_11035_cast_fp16)[name = string("input_311_cast_fp16")]; + string hidden_states_371_pad_type_0 = const()[name = string("hidden_states_371_pad_type_0"), val = string("valid")]; + tensor hidden_states_371_strides_0 = const()[name = string("hidden_states_371_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_371_pad_0 = const()[name = string("hidden_states_371_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_371_dilations_0 = const()[name = string("hidden_states_371_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_371_groups_0 = const()[name = string("hidden_states_371_groups_0"), val = int32(1)]; + tensor hidden_states_371_cast_fp16 = conv(dilations = hidden_states_371_dilations_0, groups = hidden_states_371_groups_0, pad = hidden_states_371_pad_0, pad_type = hidden_states_371_pad_type_0, strides = hidden_states_371_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_311_cast_fp16)[name = string("hidden_states_371_cast_fp16")]; + tensor inputs_301_cast_fp16 = add(x = inputs_299_cast_fp16, y = hidden_states_371_cast_fp16)[name = string("inputs_301_cast_fp16")]; + tensor obj_323_begin_0 = const()[name = string("obj_323_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_323_end_0 = const()[name = string("obj_323_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_323_end_mask_0 = const()[name = string("obj_323_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_323_cast_fp16 = slice_by_index(begin = obj_323_begin_0, end = obj_323_end_0, end_mask = obj_323_end_mask_0, x = key_caches_15_cast_fp16)[name = string("obj_323_cast_fp16")]; + tensor obj_325_begin_0 = const()[name = string("obj_325_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_325_end_0 = const()[name = string("obj_325_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_325_end_mask_0 = const()[name = string("obj_325_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_325_cast_fp16 = slice_by_index(begin = obj_325_begin_0, end = obj_325_end_0, end_mask = obj_325_end_mask_0, x = value_caches_15_cast_fp16)[name = string("obj_325_cast_fp16")]; + int32 var_11083 = const()[name = string("op_11083"), val = int32(3)]; + int32 var_11093 = const()[name = string("op_11093"), val = int32(-2)]; + tensor inputs_sq_301_cast_fp16 = mul(x = inputs_301_cast_fp16, y = inputs_301_cast_fp16)[name = string("inputs_sq_301_cast_fp16")]; + tensor variance_301_axes_0 = const()[name = string("variance_301_axes_0"), val = tensor([1])]; + bool variance_301_keep_dims_0 = const()[name = string("variance_301_keep_dims_0"), val = bool(true)]; + tensor variance_301_cast_fp16 = reduce_mean(axes = variance_301_axes_0, keep_dims = variance_301_keep_dims_0, x = inputs_sq_301_cast_fp16)[name = string("variance_301_cast_fp16")]; + fp16 var_11107_to_fp16 = const()[name = string("op_11107_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11108_cast_fp16 = add(x = variance_301_cast_fp16, y = var_11107_to_fp16)[name = string("op_11108_cast_fp16")]; + fp32 var_11109_epsilon_0 = const()[name = string("op_11109_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11109_cast_fp16 = rsqrt(epsilon = var_11109_epsilon_0, x = var_11108_cast_fp16)[name = string("op_11109_cast_fp16")]; + tensor hidden_states_373_cast_fp16 = mul(x = inputs_301_cast_fp16, y = var_11109_cast_fp16)[name = string("hidden_states_373_cast_fp16")]; + tensor obj_321_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_373_cast_fp16)[name = string("obj_321_cast_fp16")]; + string query_217_pad_type_0 = const()[name = string("query_217_pad_type_0"), val = string("valid")]; + tensor query_217_strides_0 = const()[name = string("query_217_strides_0"), val = tensor([1, 1])]; + tensor query_217_pad_0 = const()[name = string("query_217_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_217_dilations_0 = const()[name = string("query_217_dilations_0"), val = tensor([1, 1])]; + int32 query_217_groups_0 = const()[name = string("query_217_groups_0"), val = int32(1)]; + tensor query_217_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_217_dilations_0, groups = query_217_groups_0, pad = query_217_pad_0, pad_type = query_217_pad_type_0, strides = query_217_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_321_cast_fp16)[name = string("query_217_cast_fp16")]; + string current_key_145_pad_type_0 = const()[name = string("current_key_145_pad_type_0"), val = string("valid")]; + tensor current_key_145_strides_0 = const()[name = string("current_key_145_strides_0"), val = tensor([1, 1])]; + tensor current_key_145_pad_0 = const()[name = string("current_key_145_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_145_dilations_0 = const()[name = string("current_key_145_dilations_0"), val = tensor([1, 1])]; + int32 current_key_145_groups_0 = const()[name = string("current_key_145_groups_0"), val = int32(1)]; + tensor current_key_145_cast_fp16 = conv(dilations = current_key_145_dilations_0, groups = current_key_145_groups_0, pad = current_key_145_pad_0, pad_type = current_key_145_pad_type_0, strides = current_key_145_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_321_cast_fp16)[name = string("current_key_145_cast_fp16")]; + string current_value_73_pad_type_0 = const()[name = string("current_value_73_pad_type_0"), val = string("valid")]; + tensor current_value_73_strides_0 = const()[name = string("current_value_73_strides_0"), val = tensor([1, 1])]; + tensor current_value_73_pad_0 = const()[name = string("current_value_73_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_73_dilations_0 = const()[name = string("current_value_73_dilations_0"), val = tensor([1, 1])]; + int32 current_value_73_groups_0 = const()[name = string("current_value_73_groups_0"), val = int32(1)]; + tensor current_value_73_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_73_dilations_0, groups = current_value_73_groups_0, pad = current_value_73_pad_0, pad_type = current_value_73_pad_type_0, strides = current_value_73_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_321_cast_fp16)[name = string("current_value_73_cast_fp16")]; + tensor var_11146 = const()[name = string("op_11146"), val = tensor([16, 128, 1, 1])]; + tensor inputs_303_cast_fp16 = reshape(shape = var_11146, x = query_217_cast_fp16)[name = string("inputs_303_cast_fp16")]; + tensor inputs_sq_303_cast_fp16 = mul(x = inputs_303_cast_fp16, y = inputs_303_cast_fp16)[name = string("inputs_sq_303_cast_fp16")]; + tensor variance_303_axes_0 = const()[name = string("variance_303_axes_0"), val = tensor([1])]; + bool variance_303_keep_dims_0 = const()[name = string("variance_303_keep_dims_0"), val = bool(true)]; + tensor variance_303_cast_fp16 = reduce_mean(axes = variance_303_axes_0, keep_dims = variance_303_keep_dims_0, x = inputs_sq_303_cast_fp16)[name = string("variance_303_cast_fp16")]; + fp16 var_11152_to_fp16 = const()[name = string("op_11152_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11153_cast_fp16 = add(x = variance_303_cast_fp16, y = var_11152_to_fp16)[name = string("op_11153_cast_fp16")]; + fp32 var_11154_epsilon_0 = const()[name = string("op_11154_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11154_cast_fp16 = rsqrt(epsilon = var_11154_epsilon_0, x = var_11153_cast_fp16)[name = string("op_11154_cast_fp16")]; + tensor hidden_states_375_cast_fp16 = mul(x = inputs_303_cast_fp16, y = var_11154_cast_fp16)[name = string("hidden_states_375_cast_fp16")]; + tensor query_normed_73_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_375_cast_fp16)[name = string("query_normed_73_cast_fp16")]; + tensor var_11162 = const()[name = string("op_11162"), val = tensor([8, 128, 1, 1])]; + tensor inputs_305_cast_fp16 = reshape(shape = var_11162, x = current_key_145_cast_fp16)[name = string("inputs_305_cast_fp16")]; + tensor inputs_sq_305_cast_fp16 = mul(x = inputs_305_cast_fp16, y = inputs_305_cast_fp16)[name = string("inputs_sq_305_cast_fp16")]; + tensor variance_305_axes_0 = const()[name = string("variance_305_axes_0"), val = tensor([1])]; + bool variance_305_keep_dims_0 = const()[name = string("variance_305_keep_dims_0"), val = bool(true)]; + tensor variance_305_cast_fp16 = reduce_mean(axes = variance_305_axes_0, keep_dims = variance_305_keep_dims_0, x = inputs_sq_305_cast_fp16)[name = string("variance_305_cast_fp16")]; + fp16 var_11168_to_fp16 = const()[name = string("op_11168_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11169_cast_fp16 = add(x = variance_305_cast_fp16, y = var_11168_to_fp16)[name = string("op_11169_cast_fp16")]; + fp32 var_11170_epsilon_0 = const()[name = string("op_11170_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11170_cast_fp16 = rsqrt(epsilon = var_11170_epsilon_0, x = var_11169_cast_fp16)[name = string("op_11170_cast_fp16")]; + tensor hidden_states_377_cast_fp16 = mul(x = inputs_305_cast_fp16, y = var_11170_cast_fp16)[name = string("hidden_states_377_cast_fp16")]; + tensor current_key_normed_73_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_377_cast_fp16)[name = string("current_key_normed_73_cast_fp16")]; + tensor var_11188 = const()[name = string("op_11188"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_289_cast_fp16 = reshape(shape = var_11188, x = query_normed_73_cast_fp16)[name = string("mh_q_289_cast_fp16")]; + tensor var_11190 = const()[name = string("op_11190"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_289_cast_fp16 = reshape(shape = var_11190, x = current_key_normed_73_cast_fp16)[name = string("mh_k_289_cast_fp16")]; + tensor var_11194_cast_fp16 = mul(x = mh_q_289_cast_fp16, y = cos_71_to_fp16)[name = string("op_11194_cast_fp16")]; + tensor var_11199_begin_0 = const()[name = string("op_11199_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11199_end_0 = const()[name = string("op_11199_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_11199_end_mask_0 = const()[name = string("op_11199_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_11199_cast_fp16 = slice_by_index(begin = var_11199_begin_0, end = var_11199_end_0, end_mask = var_11199_end_mask_0, x = mh_q_289_cast_fp16)[name = string("op_11199_cast_fp16")]; + tensor var_11205_begin_0 = const()[name = string("op_11205_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_11205_end_0 = const()[name = string("op_11205_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_11205_end_mask_0 = const()[name = string("op_11205_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_11205_cast_fp16 = slice_by_index(begin = var_11205_begin_0, end = var_11205_end_0, end_mask = var_11205_end_mask_0, x = mh_q_289_cast_fp16)[name = string("op_11205_cast_fp16")]; + fp16 const_741_promoted_to_fp16 = const()[name = string("const_741_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11207_cast_fp16 = mul(x = var_11205_cast_fp16, y = const_741_promoted_to_fp16)[name = string("op_11207_cast_fp16")]; + bool var_11209_interleave_0 = const()[name = string("op_11209_interleave_0"), val = bool(false)]; + tensor var_11209_cast_fp16 = concat(axis = var_11093, interleave = var_11209_interleave_0, values = (var_11207_cast_fp16, var_11199_cast_fp16))[name = string("op_11209_cast_fp16")]; + tensor var_11210_cast_fp16 = mul(x = var_11209_cast_fp16, y = sin_71_to_fp16)[name = string("op_11210_cast_fp16")]; + tensor mh_q_291_cast_fp16 = add(x = var_11194_cast_fp16, y = var_11210_cast_fp16)[name = string("mh_q_291_cast_fp16")]; + tensor var_11212_cast_fp16 = mul(x = mh_k_289_cast_fp16, y = cos_71_to_fp16)[name = string("op_11212_cast_fp16")]; + tensor var_11217_begin_0 = const()[name = string("op_11217_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11217_end_0 = const()[name = string("op_11217_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_11217_end_mask_0 = const()[name = string("op_11217_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_11217_cast_fp16 = slice_by_index(begin = var_11217_begin_0, end = var_11217_end_0, end_mask = var_11217_end_mask_0, x = mh_k_289_cast_fp16)[name = string("op_11217_cast_fp16")]; + tensor var_11223_begin_0 = const()[name = string("op_11223_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_11223_end_0 = const()[name = string("op_11223_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_11223_end_mask_0 = const()[name = string("op_11223_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_11223_cast_fp16 = slice_by_index(begin = var_11223_begin_0, end = var_11223_end_0, end_mask = var_11223_end_mask_0, x = mh_k_289_cast_fp16)[name = string("op_11223_cast_fp16")]; + fp16 const_744_promoted_to_fp16 = const()[name = string("const_744_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11225_cast_fp16 = mul(x = var_11223_cast_fp16, y = const_744_promoted_to_fp16)[name = string("op_11225_cast_fp16")]; + bool var_11227_interleave_0 = const()[name = string("op_11227_interleave_0"), val = bool(false)]; + tensor var_11227_cast_fp16 = concat(axis = var_11093, interleave = var_11227_interleave_0, values = (var_11225_cast_fp16, var_11217_cast_fp16))[name = string("op_11227_cast_fp16")]; + tensor var_11228_cast_fp16 = mul(x = var_11227_cast_fp16, y = sin_71_to_fp16)[name = string("op_11228_cast_fp16")]; + tensor mh_k_291_cast_fp16 = add(x = var_11212_cast_fp16, y = var_11228_cast_fp16)[name = string("mh_k_291_cast_fp16")]; + tensor var_11232 = const()[name = string("op_11232"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_147_cast_fp16 = reshape(shape = var_11232, x = mh_k_291_cast_fp16)[name = string("current_key_147_cast_fp16")]; + tensor var_11238_to_fp16 = const()[name = string("op_11238_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196352)))]; + tensor var_11239_cast_fp16 = mul(x = obj_323_cast_fp16, y = var_11238_to_fp16)[name = string("op_11239_cast_fp16")]; + tensor var_11236_to_fp16 = const()[name = string("op_11236_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196480)))]; + tensor var_11240_cast_fp16 = mul(x = current_key_147_cast_fp16, y = var_11236_to_fp16)[name = string("op_11240_cast_fp16")]; + tensor key_147_cast_fp16 = add(x = var_11239_cast_fp16, y = var_11240_cast_fp16)[name = string("key_147_cast_fp16")]; + tensor var_11242_to_fp16 = const()[name = string("op_11242_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196352)))]; + tensor var_11243_cast_fp16 = mul(x = obj_325_cast_fp16, y = var_11242_to_fp16)[name = string("op_11243_cast_fp16")]; + tensor var_11244_cast_fp16 = mul(x = current_value_73_cast_fp16, y = var_11236_to_fp16)[name = string("op_11244_cast_fp16")]; + tensor value_73_cast_fp16 = add(x = var_11243_cast_fp16, y = var_11244_cast_fp16)[name = string("value_73_cast_fp16")]; + fp16 var_11251_to_fp16 = const()[name = string("op_11251_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_295_cast_fp16 = mul(x = mh_q_291_cast_fp16, y = var_11251_to_fp16)[name = string("mh_q_295_cast_fp16")]; + tensor var_11253 = const()[name = string("op_11253"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_293_cast_fp16 = reshape(shape = var_11253, x = key_147_cast_fp16)[name = string("mh_k_293_cast_fp16")]; + tensor var_11255 = const()[name = string("op_11255"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_145_cast_fp16 = reshape(shape = var_11255, x = value_73_cast_fp16)[name = string("mh_v_145_cast_fp16")]; + tensor transpose_144_perm_0 = const()[name = string("transpose_144_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_72_reps_0 = const()[name = string("tile_72_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_144_cast_fp16 = transpose(perm = transpose_144_perm_0, x = mh_k_293_cast_fp16)[name = string("transpose_263")]; + tensor tile_72_cast_fp16 = tile(reps = tile_72_reps_0, x = transpose_144_cast_fp16)[name = string("tile_72_cast_fp16")]; + tensor concat_182 = const()[name = string("concat_182"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_144_cast_fp16 = reshape(shape = concat_182, x = tile_72_cast_fp16)[name = string("reshape_144_cast_fp16")]; + tensor transpose_145_perm_0 = const()[name = string("transpose_145_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_183 = const()[name = string("concat_183"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_145_cast_fp16 = transpose(perm = transpose_145_perm_0, x = reshape_144_cast_fp16)[name = string("transpose_262")]; + tensor reshape_145_cast_fp16 = reshape(shape = concat_183, x = transpose_145_cast_fp16)[name = string("reshape_145_cast_fp16")]; + tensor transpose_146_perm_0 = const()[name = string("transpose_146_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_73_reps_0 = const()[name = string("tile_73_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_146_cast_fp16 = transpose(perm = transpose_146_perm_0, x = mh_v_145_cast_fp16)[name = string("transpose_261")]; + tensor tile_73_cast_fp16 = tile(reps = tile_73_reps_0, x = transpose_146_cast_fp16)[name = string("tile_73_cast_fp16")]; + tensor concat_184 = const()[name = string("concat_184"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_146_cast_fp16 = reshape(shape = concat_184, x = tile_73_cast_fp16)[name = string("reshape_146_cast_fp16")]; + tensor transpose_147_perm_0 = const()[name = string("transpose_147_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_185 = const()[name = string("concat_185"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_147_cast_fp16 = transpose(perm = transpose_147_perm_0, x = reshape_146_cast_fp16)[name = string("transpose_260")]; + tensor reshape_147_cast_fp16 = reshape(shape = concat_185, x = transpose_147_cast_fp16)[name = string("reshape_147_cast_fp16")]; + tensor transpose_461_perm_0 = const()[name = string("transpose_461_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_217_transpose_x_1 = const()[name = string("mh_w_217_transpose_x_1"), val = bool(true)]; + bool mh_w_217_transpose_y_1 = const()[name = string("mh_w_217_transpose_y_1"), val = bool(false)]; + tensor transpose_461_cast_fp16 = transpose(perm = transpose_461_perm_0, x = reshape_145_cast_fp16)[name = string("transpose_259")]; + tensor mh_w_217_cast_fp16 = matmul(transpose_x = mh_w_217_transpose_x_1, transpose_y = mh_w_217_transpose_y_1, x = mh_q_295_cast_fp16, y = transpose_461_cast_fp16)[name = string("mh_w_217_cast_fp16")]; + tensor var_11263_to_fp16 = const()[name = string("op_11263_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196608)))]; + tensor mh_w_219_cast_fp16 = add(x = mh_w_217_cast_fp16, y = var_11263_to_fp16)[name = string("mh_w_219_cast_fp16")]; + tensor mh_w_221_cast_fp16 = softmax(axis = var_11083, x = mh_w_219_cast_fp16)[name = string("mh_w_221_cast_fp16")]; + tensor transpose_462_perm_0 = const()[name = string("transpose_462_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_73_transpose_x_1 = const()[name = string("attn_73_transpose_x_1"), val = bool(false)]; + bool attn_73_transpose_y_1 = const()[name = string("attn_73_transpose_y_1"), val = bool(true)]; + tensor transpose_462_cast_fp16 = transpose(perm = transpose_462_perm_0, x = reshape_147_cast_fp16)[name = string("transpose_258")]; + tensor attn_73_cast_fp16 = matmul(transpose_x = attn_73_transpose_x_1, transpose_y = attn_73_transpose_y_1, x = transpose_462_cast_fp16, y = mh_w_221_cast_fp16)[name = string("attn_73_cast_fp16")]; + tensor var_11269 = const()[name = string("op_11269"), val = tensor([1, 2048, 1, 1])]; + tensor input_313_cast_fp16 = reshape(shape = var_11269, x = attn_73_cast_fp16)[name = string("input_313_cast_fp16")]; + string obj_327_pad_type_0 = const()[name = string("obj_327_pad_type_0"), val = string("valid")]; + tensor obj_327_strides_0 = const()[name = string("obj_327_strides_0"), val = tensor([1, 1])]; + tensor obj_327_pad_0 = const()[name = string("obj_327_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_327_dilations_0 = const()[name = string("obj_327_dilations_0"), val = tensor([1, 1])]; + int32 obj_327_groups_0 = const()[name = string("obj_327_groups_0"), val = int32(1)]; + tensor obj_327_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_327_dilations_0, groups = obj_327_groups_0, pad = obj_327_pad_0, pad_type = obj_327_pad_type_0, strides = obj_327_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_313_cast_fp16)[name = string("obj_327_cast_fp16")]; + tensor inputs_307_cast_fp16 = add(x = inputs_301_cast_fp16, y = obj_327_cast_fp16)[name = string("inputs_307_cast_fp16")]; + tensor inputs_sq_307_cast_fp16 = mul(x = inputs_307_cast_fp16, y = inputs_307_cast_fp16)[name = string("inputs_sq_307_cast_fp16")]; + tensor variance_307_axes_0 = const()[name = string("variance_307_axes_0"), val = tensor([1])]; + bool variance_307_keep_dims_0 = const()[name = string("variance_307_keep_dims_0"), val = bool(true)]; + tensor variance_307_cast_fp16 = reduce_mean(axes = variance_307_axes_0, keep_dims = variance_307_keep_dims_0, x = inputs_sq_307_cast_fp16)[name = string("variance_307_cast_fp16")]; + fp16 var_11287_to_fp16 = const()[name = string("op_11287_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11288_cast_fp16 = add(x = variance_307_cast_fp16, y = var_11287_to_fp16)[name = string("op_11288_cast_fp16")]; + fp32 var_11289_epsilon_0 = const()[name = string("op_11289_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11289_cast_fp16 = rsqrt(epsilon = var_11289_epsilon_0, x = var_11288_cast_fp16)[name = string("op_11289_cast_fp16")]; + tensor hidden_states_379_cast_fp16 = mul(x = inputs_307_cast_fp16, y = var_11289_cast_fp16)[name = string("hidden_states_379_cast_fp16")]; + tensor input_315_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_379_cast_fp16)[name = string("input_315_cast_fp16")]; + string input_317_pad_type_0 = const()[name = string("input_317_pad_type_0"), val = string("valid")]; + tensor input_317_strides_0 = const()[name = string("input_317_strides_0"), val = tensor([1, 1])]; + tensor input_317_pad_0 = const()[name = string("input_317_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_317_dilations_0 = const()[name = string("input_317_dilations_0"), val = tensor([1, 1])]; + int32 input_317_groups_0 = const()[name = string("input_317_groups_0"), val = int32(1)]; + tensor input_317_cast_fp16 = conv(dilations = input_317_dilations_0, groups = input_317_groups_0, pad = input_317_pad_0, pad_type = input_317_pad_type_0, strides = input_317_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_315_cast_fp16)[name = string("input_317_cast_fp16")]; + tensor var_11303_cast_fp16 = silu(x = input_317_cast_fp16)[name = string("op_11303_cast_fp16")]; + string var_11309_pad_type_0 = const()[name = string("op_11309_pad_type_0"), val = string("valid")]; + tensor var_11309_strides_0 = const()[name = string("op_11309_strides_0"), val = tensor([1, 1])]; + tensor var_11309_pad_0 = const()[name = string("op_11309_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11309_dilations_0 = const()[name = string("op_11309_dilations_0"), val = tensor([1, 1])]; + int32 var_11309_groups_0 = const()[name = string("op_11309_groups_0"), val = int32(1)]; + tensor var_11309_cast_fp16 = conv(dilations = var_11309_dilations_0, groups = var_11309_groups_0, pad = var_11309_pad_0, pad_type = var_11309_pad_type_0, strides = var_11309_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_315_cast_fp16)[name = string("op_11309_cast_fp16")]; + tensor input_319_cast_fp16 = mul(x = var_11303_cast_fp16, y = var_11309_cast_fp16)[name = string("input_319_cast_fp16")]; + string hidden_states_381_pad_type_0 = const()[name = string("hidden_states_381_pad_type_0"), val = string("valid")]; + tensor hidden_states_381_strides_0 = const()[name = string("hidden_states_381_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_381_pad_0 = const()[name = string("hidden_states_381_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_381_dilations_0 = const()[name = string("hidden_states_381_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_381_groups_0 = const()[name = string("hidden_states_381_groups_0"), val = int32(1)]; + tensor hidden_states_381_cast_fp16 = conv(dilations = hidden_states_381_dilations_0, groups = hidden_states_381_groups_0, pad = hidden_states_381_pad_0, pad_type = hidden_states_381_pad_type_0, strides = hidden_states_381_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_319_cast_fp16)[name = string("hidden_states_381_cast_fp16")]; + tensor inputs_309_cast_fp16 = add(x = inputs_307_cast_fp16, y = hidden_states_381_cast_fp16)[name = string("inputs_309_cast_fp16")]; + tensor obj_331_begin_0 = const()[name = string("obj_331_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_331_end_0 = const()[name = string("obj_331_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_331_end_mask_0 = const()[name = string("obj_331_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_331_cast_fp16 = slice_by_index(begin = obj_331_begin_0, end = obj_331_end_0, end_mask = obj_331_end_mask_0, x = key_caches_15_cast_fp16)[name = string("obj_331_cast_fp16")]; + tensor obj_333_begin_0 = const()[name = string("obj_333_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_333_end_0 = const()[name = string("obj_333_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_333_end_mask_0 = const()[name = string("obj_333_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_333_cast_fp16 = slice_by_index(begin = obj_333_begin_0, end = obj_333_end_0, end_mask = obj_333_end_mask_0, x = value_caches_15_cast_fp16)[name = string("obj_333_cast_fp16")]; + int32 var_11357 = const()[name = string("op_11357"), val = int32(3)]; + int32 var_11367 = const()[name = string("op_11367"), val = int32(-2)]; + tensor inputs_sq_309_cast_fp16 = mul(x = inputs_309_cast_fp16, y = inputs_309_cast_fp16)[name = string("inputs_sq_309_cast_fp16")]; + tensor variance_309_axes_0 = const()[name = string("variance_309_axes_0"), val = tensor([1])]; + bool variance_309_keep_dims_0 = const()[name = string("variance_309_keep_dims_0"), val = bool(true)]; + tensor variance_309_cast_fp16 = reduce_mean(axes = variance_309_axes_0, keep_dims = variance_309_keep_dims_0, x = inputs_sq_309_cast_fp16)[name = string("variance_309_cast_fp16")]; + fp16 var_11381_to_fp16 = const()[name = string("op_11381_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11382_cast_fp16 = add(x = variance_309_cast_fp16, y = var_11381_to_fp16)[name = string("op_11382_cast_fp16")]; + fp32 var_11383_epsilon_0 = const()[name = string("op_11383_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11383_cast_fp16 = rsqrt(epsilon = var_11383_epsilon_0, x = var_11382_cast_fp16)[name = string("op_11383_cast_fp16")]; + tensor hidden_states_383_cast_fp16 = mul(x = inputs_309_cast_fp16, y = var_11383_cast_fp16)[name = string("hidden_states_383_cast_fp16")]; + tensor obj_329_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_383_cast_fp16)[name = string("obj_329_cast_fp16")]; + string query_223_pad_type_0 = const()[name = string("query_223_pad_type_0"), val = string("valid")]; + tensor query_223_strides_0 = const()[name = string("query_223_strides_0"), val = tensor([1, 1])]; + tensor query_223_pad_0 = const()[name = string("query_223_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_223_dilations_0 = const()[name = string("query_223_dilations_0"), val = tensor([1, 1])]; + int32 query_223_groups_0 = const()[name = string("query_223_groups_0"), val = int32(1)]; + tensor query_223_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_223_dilations_0, groups = query_223_groups_0, pad = query_223_pad_0, pad_type = query_223_pad_type_0, strides = query_223_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_329_cast_fp16)[name = string("query_223_cast_fp16")]; + string current_key_149_pad_type_0 = const()[name = string("current_key_149_pad_type_0"), val = string("valid")]; + tensor current_key_149_strides_0 = const()[name = string("current_key_149_strides_0"), val = tensor([1, 1])]; + tensor current_key_149_pad_0 = const()[name = string("current_key_149_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_149_dilations_0 = const()[name = string("current_key_149_dilations_0"), val = tensor([1, 1])]; + int32 current_key_149_groups_0 = const()[name = string("current_key_149_groups_0"), val = int32(1)]; + tensor current_key_149_cast_fp16 = conv(dilations = current_key_149_dilations_0, groups = current_key_149_groups_0, pad = current_key_149_pad_0, pad_type = current_key_149_pad_type_0, strides = current_key_149_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_329_cast_fp16)[name = string("current_key_149_cast_fp16")]; + string current_value_75_pad_type_0 = const()[name = string("current_value_75_pad_type_0"), val = string("valid")]; + tensor current_value_75_strides_0 = const()[name = string("current_value_75_strides_0"), val = tensor([1, 1])]; + tensor current_value_75_pad_0 = const()[name = string("current_value_75_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_75_dilations_0 = const()[name = string("current_value_75_dilations_0"), val = tensor([1, 1])]; + int32 current_value_75_groups_0 = const()[name = string("current_value_75_groups_0"), val = int32(1)]; + tensor current_value_75_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_75_dilations_0, groups = current_value_75_groups_0, pad = current_value_75_pad_0, pad_type = current_value_75_pad_type_0, strides = current_value_75_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_329_cast_fp16)[name = string("current_value_75_cast_fp16")]; + tensor var_11420 = const()[name = string("op_11420"), val = tensor([16, 128, 1, 1])]; + tensor inputs_311_cast_fp16 = reshape(shape = var_11420, x = query_223_cast_fp16)[name = string("inputs_311_cast_fp16")]; + tensor inputs_sq_311_cast_fp16 = mul(x = inputs_311_cast_fp16, y = inputs_311_cast_fp16)[name = string("inputs_sq_311_cast_fp16")]; + tensor variance_311_axes_0 = const()[name = string("variance_311_axes_0"), val = tensor([1])]; + bool variance_311_keep_dims_0 = const()[name = string("variance_311_keep_dims_0"), val = bool(true)]; + tensor variance_311_cast_fp16 = reduce_mean(axes = variance_311_axes_0, keep_dims = variance_311_keep_dims_0, x = inputs_sq_311_cast_fp16)[name = string("variance_311_cast_fp16")]; + fp16 var_11426_to_fp16 = const()[name = string("op_11426_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11427_cast_fp16 = add(x = variance_311_cast_fp16, y = var_11426_to_fp16)[name = string("op_11427_cast_fp16")]; + fp32 var_11428_epsilon_0 = const()[name = string("op_11428_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11428_cast_fp16 = rsqrt(epsilon = var_11428_epsilon_0, x = var_11427_cast_fp16)[name = string("op_11428_cast_fp16")]; + tensor hidden_states_385_cast_fp16 = mul(x = inputs_311_cast_fp16, y = var_11428_cast_fp16)[name = string("hidden_states_385_cast_fp16")]; + tensor query_normed_75_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_385_cast_fp16)[name = string("query_normed_75_cast_fp16")]; + tensor var_11436 = const()[name = string("op_11436"), val = tensor([8, 128, 1, 1])]; + tensor inputs_313_cast_fp16 = reshape(shape = var_11436, x = current_key_149_cast_fp16)[name = string("inputs_313_cast_fp16")]; + tensor inputs_sq_313_cast_fp16 = mul(x = inputs_313_cast_fp16, y = inputs_313_cast_fp16)[name = string("inputs_sq_313_cast_fp16")]; + tensor variance_313_axes_0 = const()[name = string("variance_313_axes_0"), val = tensor([1])]; + bool variance_313_keep_dims_0 = const()[name = string("variance_313_keep_dims_0"), val = bool(true)]; + tensor variance_313_cast_fp16 = reduce_mean(axes = variance_313_axes_0, keep_dims = variance_313_keep_dims_0, x = inputs_sq_313_cast_fp16)[name = string("variance_313_cast_fp16")]; + fp16 var_11442_to_fp16 = const()[name = string("op_11442_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11443_cast_fp16 = add(x = variance_313_cast_fp16, y = var_11442_to_fp16)[name = string("op_11443_cast_fp16")]; + fp32 var_11444_epsilon_0 = const()[name = string("op_11444_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11444_cast_fp16 = rsqrt(epsilon = var_11444_epsilon_0, x = var_11443_cast_fp16)[name = string("op_11444_cast_fp16")]; + tensor hidden_states_387_cast_fp16 = mul(x = inputs_313_cast_fp16, y = var_11444_cast_fp16)[name = string("hidden_states_387_cast_fp16")]; + tensor current_key_normed_75_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_387_cast_fp16)[name = string("current_key_normed_75_cast_fp16")]; + tensor var_11462 = const()[name = string("op_11462"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_297_cast_fp16 = reshape(shape = var_11462, x = query_normed_75_cast_fp16)[name = string("mh_q_297_cast_fp16")]; + tensor var_11464 = const()[name = string("op_11464"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_297_cast_fp16 = reshape(shape = var_11464, x = current_key_normed_75_cast_fp16)[name = string("mh_k_297_cast_fp16")]; + tensor var_11468_cast_fp16 = mul(x = mh_q_297_cast_fp16, y = cos_71_to_fp16)[name = string("op_11468_cast_fp16")]; + tensor var_11473_begin_0 = const()[name = string("op_11473_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11473_end_0 = const()[name = string("op_11473_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_11473_end_mask_0 = const()[name = string("op_11473_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_11473_cast_fp16 = slice_by_index(begin = var_11473_begin_0, end = var_11473_end_0, end_mask = var_11473_end_mask_0, x = mh_q_297_cast_fp16)[name = string("op_11473_cast_fp16")]; + tensor var_11479_begin_0 = const()[name = string("op_11479_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_11479_end_0 = const()[name = string("op_11479_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_11479_end_mask_0 = const()[name = string("op_11479_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_11479_cast_fp16 = slice_by_index(begin = var_11479_begin_0, end = var_11479_end_0, end_mask = var_11479_end_mask_0, x = mh_q_297_cast_fp16)[name = string("op_11479_cast_fp16")]; + fp16 const_761_promoted_to_fp16 = const()[name = string("const_761_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11481_cast_fp16 = mul(x = var_11479_cast_fp16, y = const_761_promoted_to_fp16)[name = string("op_11481_cast_fp16")]; + bool var_11483_interleave_0 = const()[name = string("op_11483_interleave_0"), val = bool(false)]; + tensor var_11483_cast_fp16 = concat(axis = var_11367, interleave = var_11483_interleave_0, values = (var_11481_cast_fp16, var_11473_cast_fp16))[name = string("op_11483_cast_fp16")]; + tensor var_11484_cast_fp16 = mul(x = var_11483_cast_fp16, y = sin_71_to_fp16)[name = string("op_11484_cast_fp16")]; + tensor mh_q_299_cast_fp16 = add(x = var_11468_cast_fp16, y = var_11484_cast_fp16)[name = string("mh_q_299_cast_fp16")]; + tensor var_11486_cast_fp16 = mul(x = mh_k_297_cast_fp16, y = cos_71_to_fp16)[name = string("op_11486_cast_fp16")]; + tensor var_11491_begin_0 = const()[name = string("op_11491_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11491_end_0 = const()[name = string("op_11491_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_11491_end_mask_0 = const()[name = string("op_11491_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_11491_cast_fp16 = slice_by_index(begin = var_11491_begin_0, end = var_11491_end_0, end_mask = var_11491_end_mask_0, x = mh_k_297_cast_fp16)[name = string("op_11491_cast_fp16")]; + tensor var_11497_begin_0 = const()[name = string("op_11497_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_11497_end_0 = const()[name = string("op_11497_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_11497_end_mask_0 = const()[name = string("op_11497_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_11497_cast_fp16 = slice_by_index(begin = var_11497_begin_0, end = var_11497_end_0, end_mask = var_11497_end_mask_0, x = mh_k_297_cast_fp16)[name = string("op_11497_cast_fp16")]; + fp16 const_764_promoted_to_fp16 = const()[name = string("const_764_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11499_cast_fp16 = mul(x = var_11497_cast_fp16, y = const_764_promoted_to_fp16)[name = string("op_11499_cast_fp16")]; + bool var_11501_interleave_0 = const()[name = string("op_11501_interleave_0"), val = bool(false)]; + tensor var_11501_cast_fp16 = concat(axis = var_11367, interleave = var_11501_interleave_0, values = (var_11499_cast_fp16, var_11491_cast_fp16))[name = string("op_11501_cast_fp16")]; + tensor var_11502_cast_fp16 = mul(x = var_11501_cast_fp16, y = sin_71_to_fp16)[name = string("op_11502_cast_fp16")]; + tensor mh_k_299_cast_fp16 = add(x = var_11486_cast_fp16, y = var_11502_cast_fp16)[name = string("mh_k_299_cast_fp16")]; + tensor var_11506 = const()[name = string("op_11506"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_151_cast_fp16 = reshape(shape = var_11506, x = mh_k_299_cast_fp16)[name = string("current_key_151_cast_fp16")]; + tensor var_11512_to_fp16 = const()[name = string("op_11512_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196352)))]; + tensor var_11513_cast_fp16 = mul(x = obj_331_cast_fp16, y = var_11512_to_fp16)[name = string("op_11513_cast_fp16")]; + tensor var_11510_to_fp16 = const()[name = string("op_11510_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196480)))]; + tensor var_11514_cast_fp16 = mul(x = current_key_151_cast_fp16, y = var_11510_to_fp16)[name = string("op_11514_cast_fp16")]; + tensor key_151_cast_fp16 = add(x = var_11513_cast_fp16, y = var_11514_cast_fp16)[name = string("key_151_cast_fp16")]; + tensor var_11516_to_fp16 = const()[name = string("op_11516_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196352)))]; + tensor var_11517_cast_fp16 = mul(x = obj_333_cast_fp16, y = var_11516_to_fp16)[name = string("op_11517_cast_fp16")]; + tensor var_11518_cast_fp16 = mul(x = current_value_75_cast_fp16, y = var_11510_to_fp16)[name = string("op_11518_cast_fp16")]; + tensor value_75_cast_fp16 = add(x = var_11517_cast_fp16, y = var_11518_cast_fp16)[name = string("value_75_cast_fp16")]; + fp16 var_11525_to_fp16 = const()[name = string("op_11525_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_303_cast_fp16 = mul(x = mh_q_299_cast_fp16, y = var_11525_to_fp16)[name = string("mh_q_303_cast_fp16")]; + tensor var_11527 = const()[name = string("op_11527"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_301_cast_fp16 = reshape(shape = var_11527, x = key_151_cast_fp16)[name = string("mh_k_301_cast_fp16")]; + tensor var_11529 = const()[name = string("op_11529"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_149_cast_fp16 = reshape(shape = var_11529, x = value_75_cast_fp16)[name = string("mh_v_149_cast_fp16")]; + tensor transpose_148_perm_0 = const()[name = string("transpose_148_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_74_reps_0 = const()[name = string("tile_74_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_148_cast_fp16 = transpose(perm = transpose_148_perm_0, x = mh_k_301_cast_fp16)[name = string("transpose_257")]; + tensor tile_74_cast_fp16 = tile(reps = tile_74_reps_0, x = transpose_148_cast_fp16)[name = string("tile_74_cast_fp16")]; + tensor concat_186 = const()[name = string("concat_186"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_148_cast_fp16 = reshape(shape = concat_186, x = tile_74_cast_fp16)[name = string("reshape_148_cast_fp16")]; + tensor transpose_149_perm_0 = const()[name = string("transpose_149_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_187 = const()[name = string("concat_187"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_149_cast_fp16 = transpose(perm = transpose_149_perm_0, x = reshape_148_cast_fp16)[name = string("transpose_256")]; + tensor reshape_149_cast_fp16 = reshape(shape = concat_187, x = transpose_149_cast_fp16)[name = string("reshape_149_cast_fp16")]; + tensor transpose_150_perm_0 = const()[name = string("transpose_150_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_75_reps_0 = const()[name = string("tile_75_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_150_cast_fp16 = transpose(perm = transpose_150_perm_0, x = mh_v_149_cast_fp16)[name = string("transpose_255")]; + tensor tile_75_cast_fp16 = tile(reps = tile_75_reps_0, x = transpose_150_cast_fp16)[name = string("tile_75_cast_fp16")]; + tensor concat_188 = const()[name = string("concat_188"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_150_cast_fp16 = reshape(shape = concat_188, x = tile_75_cast_fp16)[name = string("reshape_150_cast_fp16")]; + tensor transpose_151_perm_0 = const()[name = string("transpose_151_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_189 = const()[name = string("concat_189"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_151_cast_fp16 = transpose(perm = transpose_151_perm_0, x = reshape_150_cast_fp16)[name = string("transpose_254")]; + tensor reshape_151_cast_fp16 = reshape(shape = concat_189, x = transpose_151_cast_fp16)[name = string("reshape_151_cast_fp16")]; + tensor transpose_465_perm_0 = const()[name = string("transpose_465_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_223_transpose_x_1 = const()[name = string("mh_w_223_transpose_x_1"), val = bool(true)]; + bool mh_w_223_transpose_y_1 = const()[name = string("mh_w_223_transpose_y_1"), val = bool(false)]; + tensor transpose_465_cast_fp16 = transpose(perm = transpose_465_perm_0, x = reshape_149_cast_fp16)[name = string("transpose_253")]; + tensor mh_w_223_cast_fp16 = matmul(transpose_x = mh_w_223_transpose_x_1, transpose_y = mh_w_223_transpose_y_1, x = mh_q_303_cast_fp16, y = transpose_465_cast_fp16)[name = string("mh_w_223_cast_fp16")]; + tensor var_11537_to_fp16 = const()[name = string("op_11537_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196608)))]; + tensor mh_w_225_cast_fp16 = add(x = mh_w_223_cast_fp16, y = var_11537_to_fp16)[name = string("mh_w_225_cast_fp16")]; + tensor mh_w_227_cast_fp16 = softmax(axis = var_11357, x = mh_w_225_cast_fp16)[name = string("mh_w_227_cast_fp16")]; + tensor transpose_466_perm_0 = const()[name = string("transpose_466_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_75_transpose_x_1 = const()[name = string("attn_75_transpose_x_1"), val = bool(false)]; + bool attn_75_transpose_y_1 = const()[name = string("attn_75_transpose_y_1"), val = bool(true)]; + tensor transpose_466_cast_fp16 = transpose(perm = transpose_466_perm_0, x = reshape_151_cast_fp16)[name = string("transpose_252")]; + tensor attn_75_cast_fp16 = matmul(transpose_x = attn_75_transpose_x_1, transpose_y = attn_75_transpose_y_1, x = transpose_466_cast_fp16, y = mh_w_227_cast_fp16)[name = string("attn_75_cast_fp16")]; + tensor var_11543 = const()[name = string("op_11543"), val = tensor([1, 2048, 1, 1])]; + tensor input_321_cast_fp16 = reshape(shape = var_11543, x = attn_75_cast_fp16)[name = string("input_321_cast_fp16")]; + string obj_335_pad_type_0 = const()[name = string("obj_335_pad_type_0"), val = string("valid")]; + tensor obj_335_strides_0 = const()[name = string("obj_335_strides_0"), val = tensor([1, 1])]; + tensor obj_335_pad_0 = const()[name = string("obj_335_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_335_dilations_0 = const()[name = string("obj_335_dilations_0"), val = tensor([1, 1])]; + int32 obj_335_groups_0 = const()[name = string("obj_335_groups_0"), val = int32(1)]; + tensor obj_335_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_335_dilations_0, groups = obj_335_groups_0, pad = obj_335_pad_0, pad_type = obj_335_pad_type_0, strides = obj_335_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_321_cast_fp16)[name = string("obj_335_cast_fp16")]; + tensor inputs_315_cast_fp16 = add(x = inputs_309_cast_fp16, y = obj_335_cast_fp16)[name = string("inputs_315_cast_fp16")]; + tensor inputs_sq_315_cast_fp16 = mul(x = inputs_315_cast_fp16, y = inputs_315_cast_fp16)[name = string("inputs_sq_315_cast_fp16")]; + tensor variance_315_axes_0 = const()[name = string("variance_315_axes_0"), val = tensor([1])]; + bool variance_315_keep_dims_0 = const()[name = string("variance_315_keep_dims_0"), val = bool(true)]; + tensor variance_315_cast_fp16 = reduce_mean(axes = variance_315_axes_0, keep_dims = variance_315_keep_dims_0, x = inputs_sq_315_cast_fp16)[name = string("variance_315_cast_fp16")]; + fp16 var_11561_to_fp16 = const()[name = string("op_11561_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11562_cast_fp16 = add(x = variance_315_cast_fp16, y = var_11561_to_fp16)[name = string("op_11562_cast_fp16")]; + fp32 var_11563_epsilon_0 = const()[name = string("op_11563_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11563_cast_fp16 = rsqrt(epsilon = var_11563_epsilon_0, x = var_11562_cast_fp16)[name = string("op_11563_cast_fp16")]; + tensor hidden_states_389_cast_fp16 = mul(x = inputs_315_cast_fp16, y = var_11563_cast_fp16)[name = string("hidden_states_389_cast_fp16")]; + tensor input_323_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_389_cast_fp16)[name = string("input_323_cast_fp16")]; + string input_325_pad_type_0 = const()[name = string("input_325_pad_type_0"), val = string("valid")]; + tensor input_325_strides_0 = const()[name = string("input_325_strides_0"), val = tensor([1, 1])]; + tensor input_325_pad_0 = const()[name = string("input_325_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_325_dilations_0 = const()[name = string("input_325_dilations_0"), val = tensor([1, 1])]; + int32 input_325_groups_0 = const()[name = string("input_325_groups_0"), val = int32(1)]; + tensor input_325_cast_fp16 = conv(dilations = input_325_dilations_0, groups = input_325_groups_0, pad = input_325_pad_0, pad_type = input_325_pad_type_0, strides = input_325_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_323_cast_fp16)[name = string("input_325_cast_fp16")]; + tensor var_11577_cast_fp16 = silu(x = input_325_cast_fp16)[name = string("op_11577_cast_fp16")]; + string var_11583_pad_type_0 = const()[name = string("op_11583_pad_type_0"), val = string("valid")]; + tensor var_11583_strides_0 = const()[name = string("op_11583_strides_0"), val = tensor([1, 1])]; + tensor var_11583_pad_0 = const()[name = string("op_11583_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11583_dilations_0 = const()[name = string("op_11583_dilations_0"), val = tensor([1, 1])]; + int32 var_11583_groups_0 = const()[name = string("op_11583_groups_0"), val = int32(1)]; + tensor var_11583_cast_fp16 = conv(dilations = var_11583_dilations_0, groups = var_11583_groups_0, pad = var_11583_pad_0, pad_type = var_11583_pad_type_0, strides = var_11583_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_323_cast_fp16)[name = string("op_11583_cast_fp16")]; + tensor input_327_cast_fp16 = mul(x = var_11577_cast_fp16, y = var_11583_cast_fp16)[name = string("input_327_cast_fp16")]; + string hidden_states_391_pad_type_0 = const()[name = string("hidden_states_391_pad_type_0"), val = string("valid")]; + tensor hidden_states_391_strides_0 = const()[name = string("hidden_states_391_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_391_pad_0 = const()[name = string("hidden_states_391_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_391_dilations_0 = const()[name = string("hidden_states_391_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_391_groups_0 = const()[name = string("hidden_states_391_groups_0"), val = int32(1)]; + tensor hidden_states_391_cast_fp16 = conv(dilations = hidden_states_391_dilations_0, groups = hidden_states_391_groups_0, pad = hidden_states_391_pad_0, pad_type = hidden_states_391_pad_type_0, strides = hidden_states_391_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_327_cast_fp16)[name = string("hidden_states_391_cast_fp16")]; + tensor inputs_317_cast_fp16 = add(x = inputs_315_cast_fp16, y = hidden_states_391_cast_fp16)[name = string("inputs_317_cast_fp16")]; + tensor obj_339_begin_0 = const()[name = string("obj_339_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_339_end_0 = const()[name = string("obj_339_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_339_end_mask_0 = const()[name = string("obj_339_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_339_cast_fp16 = slice_by_index(begin = obj_339_begin_0, end = obj_339_end_0, end_mask = obj_339_end_mask_0, x = key_caches_15_cast_fp16)[name = string("obj_339_cast_fp16")]; + tensor obj_341_begin_0 = const()[name = string("obj_341_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_341_end_0 = const()[name = string("obj_341_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_341_end_mask_0 = const()[name = string("obj_341_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_341_cast_fp16 = slice_by_index(begin = obj_341_begin_0, end = obj_341_end_0, end_mask = obj_341_end_mask_0, x = value_caches_15_cast_fp16)[name = string("obj_341_cast_fp16")]; + int32 var_11631 = const()[name = string("op_11631"), val = int32(3)]; + int32 var_11641 = const()[name = string("op_11641"), val = int32(-2)]; + tensor inputs_sq_317_cast_fp16 = mul(x = inputs_317_cast_fp16, y = inputs_317_cast_fp16)[name = string("inputs_sq_317_cast_fp16")]; + tensor variance_317_axes_0 = const()[name = string("variance_317_axes_0"), val = tensor([1])]; + bool variance_317_keep_dims_0 = const()[name = string("variance_317_keep_dims_0"), val = bool(true)]; + tensor variance_317_cast_fp16 = reduce_mean(axes = variance_317_axes_0, keep_dims = variance_317_keep_dims_0, x = inputs_sq_317_cast_fp16)[name = string("variance_317_cast_fp16")]; + fp16 var_11655_to_fp16 = const()[name = string("op_11655_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11656_cast_fp16 = add(x = variance_317_cast_fp16, y = var_11655_to_fp16)[name = string("op_11656_cast_fp16")]; + fp32 var_11657_epsilon_0 = const()[name = string("op_11657_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11657_cast_fp16 = rsqrt(epsilon = var_11657_epsilon_0, x = var_11656_cast_fp16)[name = string("op_11657_cast_fp16")]; + tensor hidden_states_393_cast_fp16 = mul(x = inputs_317_cast_fp16, y = var_11657_cast_fp16)[name = string("hidden_states_393_cast_fp16")]; + tensor obj_337_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_393_cast_fp16)[name = string("obj_337_cast_fp16")]; + string query_229_pad_type_0 = const()[name = string("query_229_pad_type_0"), val = string("valid")]; + tensor query_229_strides_0 = const()[name = string("query_229_strides_0"), val = tensor([1, 1])]; + tensor query_229_pad_0 = const()[name = string("query_229_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_229_dilations_0 = const()[name = string("query_229_dilations_0"), val = tensor([1, 1])]; + int32 query_229_groups_0 = const()[name = string("query_229_groups_0"), val = int32(1)]; + tensor query_229_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_229_dilations_0, groups = query_229_groups_0, pad = query_229_pad_0, pad_type = query_229_pad_type_0, strides = query_229_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_337_cast_fp16)[name = string("query_229_cast_fp16")]; + string current_key_153_pad_type_0 = const()[name = string("current_key_153_pad_type_0"), val = string("valid")]; + tensor current_key_153_strides_0 = const()[name = string("current_key_153_strides_0"), val = tensor([1, 1])]; + tensor current_key_153_pad_0 = const()[name = string("current_key_153_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_153_dilations_0 = const()[name = string("current_key_153_dilations_0"), val = tensor([1, 1])]; + int32 current_key_153_groups_0 = const()[name = string("current_key_153_groups_0"), val = int32(1)]; + tensor current_key_153_cast_fp16 = conv(dilations = current_key_153_dilations_0, groups = current_key_153_groups_0, pad = current_key_153_pad_0, pad_type = current_key_153_pad_type_0, strides = current_key_153_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_337_cast_fp16)[name = string("current_key_153_cast_fp16")]; + string current_value_77_pad_type_0 = const()[name = string("current_value_77_pad_type_0"), val = string("valid")]; + tensor current_value_77_strides_0 = const()[name = string("current_value_77_strides_0"), val = tensor([1, 1])]; + tensor current_value_77_pad_0 = const()[name = string("current_value_77_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_77_dilations_0 = const()[name = string("current_value_77_dilations_0"), val = tensor([1, 1])]; + int32 current_value_77_groups_0 = const()[name = string("current_value_77_groups_0"), val = int32(1)]; + tensor current_value_77_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_77_dilations_0, groups = current_value_77_groups_0, pad = current_value_77_pad_0, pad_type = current_value_77_pad_type_0, strides = current_value_77_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_337_cast_fp16)[name = string("current_value_77_cast_fp16")]; + tensor var_11694 = const()[name = string("op_11694"), val = tensor([16, 128, 1, 1])]; + tensor inputs_319_cast_fp16 = reshape(shape = var_11694, x = query_229_cast_fp16)[name = string("inputs_319_cast_fp16")]; + tensor inputs_sq_319_cast_fp16 = mul(x = inputs_319_cast_fp16, y = inputs_319_cast_fp16)[name = string("inputs_sq_319_cast_fp16")]; + tensor variance_319_axes_0 = const()[name = string("variance_319_axes_0"), val = tensor([1])]; + bool variance_319_keep_dims_0 = const()[name = string("variance_319_keep_dims_0"), val = bool(true)]; + tensor variance_319_cast_fp16 = reduce_mean(axes = variance_319_axes_0, keep_dims = variance_319_keep_dims_0, x = inputs_sq_319_cast_fp16)[name = string("variance_319_cast_fp16")]; + fp16 var_11700_to_fp16 = const()[name = string("op_11700_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11701_cast_fp16 = add(x = variance_319_cast_fp16, y = var_11700_to_fp16)[name = string("op_11701_cast_fp16")]; + fp32 var_11702_epsilon_0 = const()[name = string("op_11702_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11702_cast_fp16 = rsqrt(epsilon = var_11702_epsilon_0, x = var_11701_cast_fp16)[name = string("op_11702_cast_fp16")]; + tensor hidden_states_395_cast_fp16 = mul(x = inputs_319_cast_fp16, y = var_11702_cast_fp16)[name = string("hidden_states_395_cast_fp16")]; + tensor query_normed_77_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_395_cast_fp16)[name = string("query_normed_77_cast_fp16")]; + tensor var_11710 = const()[name = string("op_11710"), val = tensor([8, 128, 1, 1])]; + tensor inputs_321_cast_fp16 = reshape(shape = var_11710, x = current_key_153_cast_fp16)[name = string("inputs_321_cast_fp16")]; + tensor inputs_sq_321_cast_fp16 = mul(x = inputs_321_cast_fp16, y = inputs_321_cast_fp16)[name = string("inputs_sq_321_cast_fp16")]; + tensor variance_321_axes_0 = const()[name = string("variance_321_axes_0"), val = tensor([1])]; + bool variance_321_keep_dims_0 = const()[name = string("variance_321_keep_dims_0"), val = bool(true)]; + tensor variance_321_cast_fp16 = reduce_mean(axes = variance_321_axes_0, keep_dims = variance_321_keep_dims_0, x = inputs_sq_321_cast_fp16)[name = string("variance_321_cast_fp16")]; + fp16 var_11716_to_fp16 = const()[name = string("op_11716_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11717_cast_fp16 = add(x = variance_321_cast_fp16, y = var_11716_to_fp16)[name = string("op_11717_cast_fp16")]; + fp32 var_11718_epsilon_0 = const()[name = string("op_11718_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11718_cast_fp16 = rsqrt(epsilon = var_11718_epsilon_0, x = var_11717_cast_fp16)[name = string("op_11718_cast_fp16")]; + tensor hidden_states_397_cast_fp16 = mul(x = inputs_321_cast_fp16, y = var_11718_cast_fp16)[name = string("hidden_states_397_cast_fp16")]; + tensor current_key_normed_77_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_397_cast_fp16)[name = string("current_key_normed_77_cast_fp16")]; + tensor var_11736 = const()[name = string("op_11736"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_305_cast_fp16 = reshape(shape = var_11736, x = query_normed_77_cast_fp16)[name = string("mh_q_305_cast_fp16")]; + tensor var_11738 = const()[name = string("op_11738"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_305_cast_fp16 = reshape(shape = var_11738, x = current_key_normed_77_cast_fp16)[name = string("mh_k_305_cast_fp16")]; + tensor var_11742_cast_fp16 = mul(x = mh_q_305_cast_fp16, y = cos_71_to_fp16)[name = string("op_11742_cast_fp16")]; + tensor var_11747_begin_0 = const()[name = string("op_11747_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11747_end_0 = const()[name = string("op_11747_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_11747_end_mask_0 = const()[name = string("op_11747_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_11747_cast_fp16 = slice_by_index(begin = var_11747_begin_0, end = var_11747_end_0, end_mask = var_11747_end_mask_0, x = mh_q_305_cast_fp16)[name = string("op_11747_cast_fp16")]; + tensor var_11753_begin_0 = const()[name = string("op_11753_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_11753_end_0 = const()[name = string("op_11753_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_11753_end_mask_0 = const()[name = string("op_11753_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_11753_cast_fp16 = slice_by_index(begin = var_11753_begin_0, end = var_11753_end_0, end_mask = var_11753_end_mask_0, x = mh_q_305_cast_fp16)[name = string("op_11753_cast_fp16")]; + fp16 const_781_promoted_to_fp16 = const()[name = string("const_781_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11755_cast_fp16 = mul(x = var_11753_cast_fp16, y = const_781_promoted_to_fp16)[name = string("op_11755_cast_fp16")]; + bool var_11757_interleave_0 = const()[name = string("op_11757_interleave_0"), val = bool(false)]; + tensor var_11757_cast_fp16 = concat(axis = var_11641, interleave = var_11757_interleave_0, values = (var_11755_cast_fp16, var_11747_cast_fp16))[name = string("op_11757_cast_fp16")]; + tensor var_11758_cast_fp16 = mul(x = var_11757_cast_fp16, y = sin_71_to_fp16)[name = string("op_11758_cast_fp16")]; + tensor mh_q_307_cast_fp16 = add(x = var_11742_cast_fp16, y = var_11758_cast_fp16)[name = string("mh_q_307_cast_fp16")]; + tensor var_11760_cast_fp16 = mul(x = mh_k_305_cast_fp16, y = cos_71_to_fp16)[name = string("op_11760_cast_fp16")]; + tensor var_11765_begin_0 = const()[name = string("op_11765_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11765_end_0 = const()[name = string("op_11765_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_11765_end_mask_0 = const()[name = string("op_11765_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_11765_cast_fp16 = slice_by_index(begin = var_11765_begin_0, end = var_11765_end_0, end_mask = var_11765_end_mask_0, x = mh_k_305_cast_fp16)[name = string("op_11765_cast_fp16")]; + tensor var_11771_begin_0 = const()[name = string("op_11771_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_11771_end_0 = const()[name = string("op_11771_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_11771_end_mask_0 = const()[name = string("op_11771_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_11771_cast_fp16 = slice_by_index(begin = var_11771_begin_0, end = var_11771_end_0, end_mask = var_11771_end_mask_0, x = mh_k_305_cast_fp16)[name = string("op_11771_cast_fp16")]; + fp16 const_784_promoted_to_fp16 = const()[name = string("const_784_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_11773_cast_fp16 = mul(x = var_11771_cast_fp16, y = const_784_promoted_to_fp16)[name = string("op_11773_cast_fp16")]; + bool var_11775_interleave_0 = const()[name = string("op_11775_interleave_0"), val = bool(false)]; + tensor var_11775_cast_fp16 = concat(axis = var_11641, interleave = var_11775_interleave_0, values = (var_11773_cast_fp16, var_11765_cast_fp16))[name = string("op_11775_cast_fp16")]; + tensor var_11776_cast_fp16 = mul(x = var_11775_cast_fp16, y = sin_71_to_fp16)[name = string("op_11776_cast_fp16")]; + tensor mh_k_307_cast_fp16 = add(x = var_11760_cast_fp16, y = var_11776_cast_fp16)[name = string("mh_k_307_cast_fp16")]; + tensor var_11780 = const()[name = string("op_11780"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_155_cast_fp16 = reshape(shape = var_11780, x = mh_k_307_cast_fp16)[name = string("current_key_155_cast_fp16")]; + tensor var_11786_to_fp16 = const()[name = string("op_11786_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196352)))]; + tensor var_11787_cast_fp16 = mul(x = obj_339_cast_fp16, y = var_11786_to_fp16)[name = string("op_11787_cast_fp16")]; + tensor var_11784_to_fp16 = const()[name = string("op_11784_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196480)))]; + tensor var_11788_cast_fp16 = mul(x = current_key_155_cast_fp16, y = var_11784_to_fp16)[name = string("op_11788_cast_fp16")]; + tensor key_155_cast_fp16 = add(x = var_11787_cast_fp16, y = var_11788_cast_fp16)[name = string("key_155_cast_fp16")]; + tensor var_11790_to_fp16 = const()[name = string("op_11790_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196352)))]; + tensor var_11791_cast_fp16 = mul(x = obj_341_cast_fp16, y = var_11790_to_fp16)[name = string("op_11791_cast_fp16")]; + tensor var_11792_cast_fp16 = mul(x = current_value_77_cast_fp16, y = var_11784_to_fp16)[name = string("op_11792_cast_fp16")]; + tensor value_77_cast_fp16 = add(x = var_11791_cast_fp16, y = var_11792_cast_fp16)[name = string("value_77_cast_fp16")]; + fp16 var_11799_to_fp16 = const()[name = string("op_11799_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_311_cast_fp16 = mul(x = mh_q_307_cast_fp16, y = var_11799_to_fp16)[name = string("mh_q_311_cast_fp16")]; + tensor var_11801 = const()[name = string("op_11801"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_309_cast_fp16 = reshape(shape = var_11801, x = key_155_cast_fp16)[name = string("mh_k_309_cast_fp16")]; + tensor var_11803 = const()[name = string("op_11803"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_153_cast_fp16 = reshape(shape = var_11803, x = value_77_cast_fp16)[name = string("mh_v_153_cast_fp16")]; + tensor transpose_152_perm_0 = const()[name = string("transpose_152_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_76_reps_0 = const()[name = string("tile_76_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_152_cast_fp16 = transpose(perm = transpose_152_perm_0, x = mh_k_309_cast_fp16)[name = string("transpose_251")]; + tensor tile_76_cast_fp16 = tile(reps = tile_76_reps_0, x = transpose_152_cast_fp16)[name = string("tile_76_cast_fp16")]; + tensor concat_190 = const()[name = string("concat_190"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_152_cast_fp16 = reshape(shape = concat_190, x = tile_76_cast_fp16)[name = string("reshape_152_cast_fp16")]; + tensor transpose_153_perm_0 = const()[name = string("transpose_153_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_191 = const()[name = string("concat_191"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_153_cast_fp16 = transpose(perm = transpose_153_perm_0, x = reshape_152_cast_fp16)[name = string("transpose_250")]; + tensor reshape_153_cast_fp16 = reshape(shape = concat_191, x = transpose_153_cast_fp16)[name = string("reshape_153_cast_fp16")]; + tensor transpose_154_perm_0 = const()[name = string("transpose_154_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_77_reps_0 = const()[name = string("tile_77_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_154_cast_fp16 = transpose(perm = transpose_154_perm_0, x = mh_v_153_cast_fp16)[name = string("transpose_249")]; + tensor tile_77_cast_fp16 = tile(reps = tile_77_reps_0, x = transpose_154_cast_fp16)[name = string("tile_77_cast_fp16")]; + tensor concat_192 = const()[name = string("concat_192"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_154_cast_fp16 = reshape(shape = concat_192, x = tile_77_cast_fp16)[name = string("reshape_154_cast_fp16")]; + tensor transpose_155_perm_0 = const()[name = string("transpose_155_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_193 = const()[name = string("concat_193"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_155_cast_fp16 = transpose(perm = transpose_155_perm_0, x = reshape_154_cast_fp16)[name = string("transpose_248")]; + tensor reshape_155_cast_fp16 = reshape(shape = concat_193, x = transpose_155_cast_fp16)[name = string("reshape_155_cast_fp16")]; + tensor transpose_469_perm_0 = const()[name = string("transpose_469_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_229_transpose_x_1 = const()[name = string("mh_w_229_transpose_x_1"), val = bool(true)]; + bool mh_w_229_transpose_y_1 = const()[name = string("mh_w_229_transpose_y_1"), val = bool(false)]; + tensor transpose_469_cast_fp16 = transpose(perm = transpose_469_perm_0, x = reshape_153_cast_fp16)[name = string("transpose_247")]; + tensor mh_w_229_cast_fp16 = matmul(transpose_x = mh_w_229_transpose_x_1, transpose_y = mh_w_229_transpose_y_1, x = mh_q_311_cast_fp16, y = transpose_469_cast_fp16)[name = string("mh_w_229_cast_fp16")]; + tensor var_11811_to_fp16 = const()[name = string("op_11811_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196608)))]; + tensor mh_w_231_cast_fp16 = add(x = mh_w_229_cast_fp16, y = var_11811_to_fp16)[name = string("mh_w_231_cast_fp16")]; + tensor mh_w_233_cast_fp16 = softmax(axis = var_11631, x = mh_w_231_cast_fp16)[name = string("mh_w_233_cast_fp16")]; + tensor transpose_470_perm_0 = const()[name = string("transpose_470_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_77_transpose_x_1 = const()[name = string("attn_77_transpose_x_1"), val = bool(false)]; + bool attn_77_transpose_y_1 = const()[name = string("attn_77_transpose_y_1"), val = bool(true)]; + tensor transpose_470_cast_fp16 = transpose(perm = transpose_470_perm_0, x = reshape_155_cast_fp16)[name = string("transpose_246")]; + tensor attn_77_cast_fp16 = matmul(transpose_x = attn_77_transpose_x_1, transpose_y = attn_77_transpose_y_1, x = transpose_470_cast_fp16, y = mh_w_233_cast_fp16)[name = string("attn_77_cast_fp16")]; + tensor var_11817 = const()[name = string("op_11817"), val = tensor([1, 2048, 1, 1])]; + tensor input_329_cast_fp16 = reshape(shape = var_11817, x = attn_77_cast_fp16)[name = string("input_329_cast_fp16")]; + string obj_343_pad_type_0 = const()[name = string("obj_343_pad_type_0"), val = string("valid")]; + tensor obj_343_strides_0 = const()[name = string("obj_343_strides_0"), val = tensor([1, 1])]; + tensor obj_343_pad_0 = const()[name = string("obj_343_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_343_dilations_0 = const()[name = string("obj_343_dilations_0"), val = tensor([1, 1])]; + int32 obj_343_groups_0 = const()[name = string("obj_343_groups_0"), val = int32(1)]; + tensor obj_343_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_343_dilations_0, groups = obj_343_groups_0, pad = obj_343_pad_0, pad_type = obj_343_pad_type_0, strides = obj_343_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_329_cast_fp16)[name = string("obj_343_cast_fp16")]; + tensor inputs_323_cast_fp16 = add(x = inputs_317_cast_fp16, y = obj_343_cast_fp16)[name = string("inputs_323_cast_fp16")]; + tensor inputs_sq_323_cast_fp16 = mul(x = inputs_323_cast_fp16, y = inputs_323_cast_fp16)[name = string("inputs_sq_323_cast_fp16")]; + tensor variance_323_axes_0 = const()[name = string("variance_323_axes_0"), val = tensor([1])]; + bool variance_323_keep_dims_0 = const()[name = string("variance_323_keep_dims_0"), val = bool(true)]; + tensor variance_323_cast_fp16 = reduce_mean(axes = variance_323_axes_0, keep_dims = variance_323_keep_dims_0, x = inputs_sq_323_cast_fp16)[name = string("variance_323_cast_fp16")]; + fp16 var_11835_to_fp16 = const()[name = string("op_11835_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11836_cast_fp16 = add(x = variance_323_cast_fp16, y = var_11835_to_fp16)[name = string("op_11836_cast_fp16")]; + fp32 var_11837_epsilon_0 = const()[name = string("op_11837_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11837_cast_fp16 = rsqrt(epsilon = var_11837_epsilon_0, x = var_11836_cast_fp16)[name = string("op_11837_cast_fp16")]; + tensor hidden_states_399_cast_fp16 = mul(x = inputs_323_cast_fp16, y = var_11837_cast_fp16)[name = string("hidden_states_399_cast_fp16")]; + tensor input_331_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_399_cast_fp16)[name = string("input_331_cast_fp16")]; + string input_333_pad_type_0 = const()[name = string("input_333_pad_type_0"), val = string("valid")]; + tensor input_333_strides_0 = const()[name = string("input_333_strides_0"), val = tensor([1, 1])]; + tensor input_333_pad_0 = const()[name = string("input_333_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_333_dilations_0 = const()[name = string("input_333_dilations_0"), val = tensor([1, 1])]; + int32 input_333_groups_0 = const()[name = string("input_333_groups_0"), val = int32(1)]; + tensor input_333_cast_fp16 = conv(dilations = input_333_dilations_0, groups = input_333_groups_0, pad = input_333_pad_0, pad_type = input_333_pad_type_0, strides = input_333_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_331_cast_fp16)[name = string("input_333_cast_fp16")]; + tensor var_11851_cast_fp16 = silu(x = input_333_cast_fp16)[name = string("op_11851_cast_fp16")]; + string var_11857_pad_type_0 = const()[name = string("op_11857_pad_type_0"), val = string("valid")]; + tensor var_11857_strides_0 = const()[name = string("op_11857_strides_0"), val = tensor([1, 1])]; + tensor var_11857_pad_0 = const()[name = string("op_11857_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_11857_dilations_0 = const()[name = string("op_11857_dilations_0"), val = tensor([1, 1])]; + int32 var_11857_groups_0 = const()[name = string("op_11857_groups_0"), val = int32(1)]; + tensor var_11857_cast_fp16 = conv(dilations = var_11857_dilations_0, groups = var_11857_groups_0, pad = var_11857_pad_0, pad_type = var_11857_pad_type_0, strides = var_11857_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_331_cast_fp16)[name = string("op_11857_cast_fp16")]; + tensor input_335_cast_fp16 = mul(x = var_11851_cast_fp16, y = var_11857_cast_fp16)[name = string("input_335_cast_fp16")]; + string hidden_states_401_pad_type_0 = const()[name = string("hidden_states_401_pad_type_0"), val = string("valid")]; + tensor hidden_states_401_strides_0 = const()[name = string("hidden_states_401_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_401_pad_0 = const()[name = string("hidden_states_401_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_401_dilations_0 = const()[name = string("hidden_states_401_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_401_groups_0 = const()[name = string("hidden_states_401_groups_0"), val = int32(1)]; + tensor hidden_states_401_cast_fp16 = conv(dilations = hidden_states_401_dilations_0, groups = hidden_states_401_groups_0, pad = hidden_states_401_pad_0, pad_type = hidden_states_401_pad_type_0, strides = hidden_states_401_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_335_cast_fp16)[name = string("hidden_states_401_cast_fp16")]; + tensor inputs_325_cast_fp16 = add(x = inputs_323_cast_fp16, y = hidden_states_401_cast_fp16)[name = string("inputs_325_cast_fp16")]; + tensor obj_347_begin_0 = const()[name = string("obj_347_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_347_end_0 = const()[name = string("obj_347_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_347_end_mask_0 = const()[name = string("obj_347_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_347_cast_fp16 = slice_by_index(begin = obj_347_begin_0, end = obj_347_end_0, end_mask = obj_347_end_mask_0, x = key_caches_15_cast_fp16)[name = string("obj_347_cast_fp16")]; + tensor obj_349_begin_0 = const()[name = string("obj_349_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_349_end_0 = const()[name = string("obj_349_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_349_end_mask_0 = const()[name = string("obj_349_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_349_cast_fp16 = slice_by_index(begin = obj_349_begin_0, end = obj_349_end_0, end_mask = obj_349_end_mask_0, x = value_caches_15_cast_fp16)[name = string("obj_349_cast_fp16")]; + int32 var_11905 = const()[name = string("op_11905"), val = int32(3)]; + int32 var_11915 = const()[name = string("op_11915"), val = int32(-2)]; + tensor inputs_sq_325_cast_fp16 = mul(x = inputs_325_cast_fp16, y = inputs_325_cast_fp16)[name = string("inputs_sq_325_cast_fp16")]; + tensor variance_325_axes_0 = const()[name = string("variance_325_axes_0"), val = tensor([1])]; + bool variance_325_keep_dims_0 = const()[name = string("variance_325_keep_dims_0"), val = bool(true)]; + tensor variance_325_cast_fp16 = reduce_mean(axes = variance_325_axes_0, keep_dims = variance_325_keep_dims_0, x = inputs_sq_325_cast_fp16)[name = string("variance_325_cast_fp16")]; + fp16 var_11929_to_fp16 = const()[name = string("op_11929_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11930_cast_fp16 = add(x = variance_325_cast_fp16, y = var_11929_to_fp16)[name = string("op_11930_cast_fp16")]; + fp32 var_11931_epsilon_0 = const()[name = string("op_11931_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11931_cast_fp16 = rsqrt(epsilon = var_11931_epsilon_0, x = var_11930_cast_fp16)[name = string("op_11931_cast_fp16")]; + tensor hidden_states_403_cast_fp16 = mul(x = inputs_325_cast_fp16, y = var_11931_cast_fp16)[name = string("hidden_states_403_cast_fp16")]; + tensor obj_345_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_403_cast_fp16)[name = string("obj_345_cast_fp16")]; + string query_235_pad_type_0 = const()[name = string("query_235_pad_type_0"), val = string("valid")]; + tensor query_235_strides_0 = const()[name = string("query_235_strides_0"), val = tensor([1, 1])]; + tensor query_235_pad_0 = const()[name = string("query_235_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_235_dilations_0 = const()[name = string("query_235_dilations_0"), val = tensor([1, 1])]; + int32 query_235_groups_0 = const()[name = string("query_235_groups_0"), val = int32(1)]; + tensor query_235_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_235_dilations_0, groups = query_235_groups_0, pad = query_235_pad_0, pad_type = query_235_pad_type_0, strides = query_235_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_345_cast_fp16)[name = string("query_235_cast_fp16")]; + string current_key_157_pad_type_0 = const()[name = string("current_key_157_pad_type_0"), val = string("valid")]; + tensor current_key_157_strides_0 = const()[name = string("current_key_157_strides_0"), val = tensor([1, 1])]; + tensor current_key_157_pad_0 = const()[name = string("current_key_157_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_157_dilations_0 = const()[name = string("current_key_157_dilations_0"), val = tensor([1, 1])]; + int32 current_key_157_groups_0 = const()[name = string("current_key_157_groups_0"), val = int32(1)]; + tensor current_key_157_cast_fp16 = conv(dilations = current_key_157_dilations_0, groups = current_key_157_groups_0, pad = current_key_157_pad_0, pad_type = current_key_157_pad_type_0, strides = current_key_157_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_345_cast_fp16)[name = string("current_key_157_cast_fp16")]; + string current_value_79_pad_type_0 = const()[name = string("current_value_79_pad_type_0"), val = string("valid")]; + tensor current_value_79_strides_0 = const()[name = string("current_value_79_strides_0"), val = tensor([1, 1])]; + tensor current_value_79_pad_0 = const()[name = string("current_value_79_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_79_dilations_0 = const()[name = string("current_value_79_dilations_0"), val = tensor([1, 1])]; + int32 current_value_79_groups_0 = const()[name = string("current_value_79_groups_0"), val = int32(1)]; + tensor current_value_79_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_79_dilations_0, groups = current_value_79_groups_0, pad = current_value_79_pad_0, pad_type = current_value_79_pad_type_0, strides = current_value_79_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_345_cast_fp16)[name = string("current_value_79_cast_fp16")]; + tensor var_11968 = const()[name = string("op_11968"), val = tensor([16, 128, 1, 1])]; + tensor inputs_327_cast_fp16 = reshape(shape = var_11968, x = query_235_cast_fp16)[name = string("inputs_327_cast_fp16")]; + tensor inputs_sq_327_cast_fp16 = mul(x = inputs_327_cast_fp16, y = inputs_327_cast_fp16)[name = string("inputs_sq_327_cast_fp16")]; + tensor variance_327_axes_0 = const()[name = string("variance_327_axes_0"), val = tensor([1])]; + bool variance_327_keep_dims_0 = const()[name = string("variance_327_keep_dims_0"), val = bool(true)]; + tensor variance_327_cast_fp16 = reduce_mean(axes = variance_327_axes_0, keep_dims = variance_327_keep_dims_0, x = inputs_sq_327_cast_fp16)[name = string("variance_327_cast_fp16")]; + fp16 var_11974_to_fp16 = const()[name = string("op_11974_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11975_cast_fp16 = add(x = variance_327_cast_fp16, y = var_11974_to_fp16)[name = string("op_11975_cast_fp16")]; + fp32 var_11976_epsilon_0 = const()[name = string("op_11976_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11976_cast_fp16 = rsqrt(epsilon = var_11976_epsilon_0, x = var_11975_cast_fp16)[name = string("op_11976_cast_fp16")]; + tensor hidden_states_405_cast_fp16 = mul(x = inputs_327_cast_fp16, y = var_11976_cast_fp16)[name = string("hidden_states_405_cast_fp16")]; + tensor query_normed_79_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_405_cast_fp16)[name = string("query_normed_79_cast_fp16")]; + tensor var_11984 = const()[name = string("op_11984"), val = tensor([8, 128, 1, 1])]; + tensor inputs_329_cast_fp16 = reshape(shape = var_11984, x = current_key_157_cast_fp16)[name = string("inputs_329_cast_fp16")]; + tensor inputs_sq_329_cast_fp16 = mul(x = inputs_329_cast_fp16, y = inputs_329_cast_fp16)[name = string("inputs_sq_329_cast_fp16")]; + tensor variance_329_axes_0 = const()[name = string("variance_329_axes_0"), val = tensor([1])]; + bool variance_329_keep_dims_0 = const()[name = string("variance_329_keep_dims_0"), val = bool(true)]; + tensor variance_329_cast_fp16 = reduce_mean(axes = variance_329_axes_0, keep_dims = variance_329_keep_dims_0, x = inputs_sq_329_cast_fp16)[name = string("variance_329_cast_fp16")]; + fp16 var_11990_to_fp16 = const()[name = string("op_11990_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_11991_cast_fp16 = add(x = variance_329_cast_fp16, y = var_11990_to_fp16)[name = string("op_11991_cast_fp16")]; + fp32 var_11992_epsilon_0 = const()[name = string("op_11992_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_11992_cast_fp16 = rsqrt(epsilon = var_11992_epsilon_0, x = var_11991_cast_fp16)[name = string("op_11992_cast_fp16")]; + tensor hidden_states_407_cast_fp16 = mul(x = inputs_329_cast_fp16, y = var_11992_cast_fp16)[name = string("hidden_states_407_cast_fp16")]; + tensor current_key_normed_79_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_407_cast_fp16)[name = string("current_key_normed_79_cast_fp16")]; + tensor var_12010 = const()[name = string("op_12010"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_313_cast_fp16 = reshape(shape = var_12010, x = query_normed_79_cast_fp16)[name = string("mh_q_313_cast_fp16")]; + tensor var_12012 = const()[name = string("op_12012"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_313_cast_fp16 = reshape(shape = var_12012, x = current_key_normed_79_cast_fp16)[name = string("mh_k_313_cast_fp16")]; + tensor var_12016_cast_fp16 = mul(x = mh_q_313_cast_fp16, y = cos_71_to_fp16)[name = string("op_12016_cast_fp16")]; + tensor var_12021_begin_0 = const()[name = string("op_12021_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12021_end_0 = const()[name = string("op_12021_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_12021_end_mask_0 = const()[name = string("op_12021_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_12021_cast_fp16 = slice_by_index(begin = var_12021_begin_0, end = var_12021_end_0, end_mask = var_12021_end_mask_0, x = mh_q_313_cast_fp16)[name = string("op_12021_cast_fp16")]; + tensor var_12027_begin_0 = const()[name = string("op_12027_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_12027_end_0 = const()[name = string("op_12027_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_12027_end_mask_0 = const()[name = string("op_12027_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_12027_cast_fp16 = slice_by_index(begin = var_12027_begin_0, end = var_12027_end_0, end_mask = var_12027_end_mask_0, x = mh_q_313_cast_fp16)[name = string("op_12027_cast_fp16")]; + fp16 const_801_promoted_to_fp16 = const()[name = string("const_801_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12029_cast_fp16 = mul(x = var_12027_cast_fp16, y = const_801_promoted_to_fp16)[name = string("op_12029_cast_fp16")]; + bool var_12031_interleave_0 = const()[name = string("op_12031_interleave_0"), val = bool(false)]; + tensor var_12031_cast_fp16 = concat(axis = var_11915, interleave = var_12031_interleave_0, values = (var_12029_cast_fp16, var_12021_cast_fp16))[name = string("op_12031_cast_fp16")]; + tensor var_12032_cast_fp16 = mul(x = var_12031_cast_fp16, y = sin_71_to_fp16)[name = string("op_12032_cast_fp16")]; + tensor mh_q_315_cast_fp16 = add(x = var_12016_cast_fp16, y = var_12032_cast_fp16)[name = string("mh_q_315_cast_fp16")]; + tensor var_12034_cast_fp16 = mul(x = mh_k_313_cast_fp16, y = cos_71_to_fp16)[name = string("op_12034_cast_fp16")]; + tensor var_12039_begin_0 = const()[name = string("op_12039_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12039_end_0 = const()[name = string("op_12039_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_12039_end_mask_0 = const()[name = string("op_12039_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_12039_cast_fp16 = slice_by_index(begin = var_12039_begin_0, end = var_12039_end_0, end_mask = var_12039_end_mask_0, x = mh_k_313_cast_fp16)[name = string("op_12039_cast_fp16")]; + tensor var_12045_begin_0 = const()[name = string("op_12045_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_12045_end_0 = const()[name = string("op_12045_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_12045_end_mask_0 = const()[name = string("op_12045_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_12045_cast_fp16 = slice_by_index(begin = var_12045_begin_0, end = var_12045_end_0, end_mask = var_12045_end_mask_0, x = mh_k_313_cast_fp16)[name = string("op_12045_cast_fp16")]; + fp16 const_804_promoted_to_fp16 = const()[name = string("const_804_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12047_cast_fp16 = mul(x = var_12045_cast_fp16, y = const_804_promoted_to_fp16)[name = string("op_12047_cast_fp16")]; + bool var_12049_interleave_0 = const()[name = string("op_12049_interleave_0"), val = bool(false)]; + tensor var_12049_cast_fp16 = concat(axis = var_11915, interleave = var_12049_interleave_0, values = (var_12047_cast_fp16, var_12039_cast_fp16))[name = string("op_12049_cast_fp16")]; + tensor var_12050_cast_fp16 = mul(x = var_12049_cast_fp16, y = sin_71_to_fp16)[name = string("op_12050_cast_fp16")]; + tensor mh_k_315_cast_fp16 = add(x = var_12034_cast_fp16, y = var_12050_cast_fp16)[name = string("mh_k_315_cast_fp16")]; + tensor var_12054 = const()[name = string("op_12054"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_159_cast_fp16 = reshape(shape = var_12054, x = mh_k_315_cast_fp16)[name = string("current_key_159_cast_fp16")]; + tensor var_12060_to_fp16 = const()[name = string("op_12060_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196352)))]; + tensor var_12061_cast_fp16 = mul(x = obj_347_cast_fp16, y = var_12060_to_fp16)[name = string("op_12061_cast_fp16")]; + tensor var_12058_to_fp16 = const()[name = string("op_12058_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196480)))]; + tensor var_12062_cast_fp16 = mul(x = current_key_159_cast_fp16, y = var_12058_to_fp16)[name = string("op_12062_cast_fp16")]; + tensor key_159_cast_fp16 = add(x = var_12061_cast_fp16, y = var_12062_cast_fp16)[name = string("key_159_cast_fp16")]; + tensor var_12064_to_fp16 = const()[name = string("op_12064_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196352)))]; + tensor var_12065_cast_fp16 = mul(x = obj_349_cast_fp16, y = var_12064_to_fp16)[name = string("op_12065_cast_fp16")]; + tensor var_12066_cast_fp16 = mul(x = current_value_79_cast_fp16, y = var_12058_to_fp16)[name = string("op_12066_cast_fp16")]; + tensor value_79_cast_fp16 = add(x = var_12065_cast_fp16, y = var_12066_cast_fp16)[name = string("value_79_cast_fp16")]; + fp16 var_12073_to_fp16 = const()[name = string("op_12073_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_319_cast_fp16 = mul(x = mh_q_315_cast_fp16, y = var_12073_to_fp16)[name = string("mh_q_319_cast_fp16")]; + tensor var_12075 = const()[name = string("op_12075"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_317_cast_fp16 = reshape(shape = var_12075, x = key_159_cast_fp16)[name = string("mh_k_317_cast_fp16")]; + tensor var_12077 = const()[name = string("op_12077"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_157_cast_fp16 = reshape(shape = var_12077, x = value_79_cast_fp16)[name = string("mh_v_157_cast_fp16")]; + tensor transpose_156_perm_0 = const()[name = string("transpose_156_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_78_reps_0 = const()[name = string("tile_78_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_156_cast_fp16 = transpose(perm = transpose_156_perm_0, x = mh_k_317_cast_fp16)[name = string("transpose_245")]; + tensor tile_78_cast_fp16 = tile(reps = tile_78_reps_0, x = transpose_156_cast_fp16)[name = string("tile_78_cast_fp16")]; + tensor concat_194 = const()[name = string("concat_194"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_156_cast_fp16 = reshape(shape = concat_194, x = tile_78_cast_fp16)[name = string("reshape_156_cast_fp16")]; + tensor transpose_157_perm_0 = const()[name = string("transpose_157_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_195 = const()[name = string("concat_195"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_157_cast_fp16 = transpose(perm = transpose_157_perm_0, x = reshape_156_cast_fp16)[name = string("transpose_244")]; + tensor reshape_157_cast_fp16 = reshape(shape = concat_195, x = transpose_157_cast_fp16)[name = string("reshape_157_cast_fp16")]; + tensor transpose_158_perm_0 = const()[name = string("transpose_158_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_79_reps_0 = const()[name = string("tile_79_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_158_cast_fp16 = transpose(perm = transpose_158_perm_0, x = mh_v_157_cast_fp16)[name = string("transpose_243")]; + tensor tile_79_cast_fp16 = tile(reps = tile_79_reps_0, x = transpose_158_cast_fp16)[name = string("tile_79_cast_fp16")]; + tensor concat_196 = const()[name = string("concat_196"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_158_cast_fp16 = reshape(shape = concat_196, x = tile_79_cast_fp16)[name = string("reshape_158_cast_fp16")]; + tensor transpose_159_perm_0 = const()[name = string("transpose_159_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_197 = const()[name = string("concat_197"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_159_cast_fp16 = transpose(perm = transpose_159_perm_0, x = reshape_158_cast_fp16)[name = string("transpose_242")]; + tensor reshape_159_cast_fp16 = reshape(shape = concat_197, x = transpose_159_cast_fp16)[name = string("reshape_159_cast_fp16")]; + tensor transpose_473_perm_0 = const()[name = string("transpose_473_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_235_transpose_x_1 = const()[name = string("mh_w_235_transpose_x_1"), val = bool(true)]; + bool mh_w_235_transpose_y_1 = const()[name = string("mh_w_235_transpose_y_1"), val = bool(false)]; + tensor transpose_473_cast_fp16 = transpose(perm = transpose_473_perm_0, x = reshape_157_cast_fp16)[name = string("transpose_241")]; + tensor mh_w_235_cast_fp16 = matmul(transpose_x = mh_w_235_transpose_x_1, transpose_y = mh_w_235_transpose_y_1, x = mh_q_319_cast_fp16, y = transpose_473_cast_fp16)[name = string("mh_w_235_cast_fp16")]; + tensor var_12085_to_fp16 = const()[name = string("op_12085_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196608)))]; + tensor mh_w_237_cast_fp16 = add(x = mh_w_235_cast_fp16, y = var_12085_to_fp16)[name = string("mh_w_237_cast_fp16")]; + tensor mh_w_239_cast_fp16 = softmax(axis = var_11905, x = mh_w_237_cast_fp16)[name = string("mh_w_239_cast_fp16")]; + tensor transpose_474_perm_0 = const()[name = string("transpose_474_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_79_transpose_x_1 = const()[name = string("attn_79_transpose_x_1"), val = bool(false)]; + bool attn_79_transpose_y_1 = const()[name = string("attn_79_transpose_y_1"), val = bool(true)]; + tensor transpose_474_cast_fp16 = transpose(perm = transpose_474_perm_0, x = reshape_159_cast_fp16)[name = string("transpose_240")]; + tensor attn_79_cast_fp16 = matmul(transpose_x = attn_79_transpose_x_1, transpose_y = attn_79_transpose_y_1, x = transpose_474_cast_fp16, y = mh_w_239_cast_fp16)[name = string("attn_79_cast_fp16")]; + tensor var_12091 = const()[name = string("op_12091"), val = tensor([1, 2048, 1, 1])]; + tensor input_337_cast_fp16 = reshape(shape = var_12091, x = attn_79_cast_fp16)[name = string("input_337_cast_fp16")]; + string obj_351_pad_type_0 = const()[name = string("obj_351_pad_type_0"), val = string("valid")]; + tensor obj_351_strides_0 = const()[name = string("obj_351_strides_0"), val = tensor([1, 1])]; + tensor obj_351_pad_0 = const()[name = string("obj_351_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_351_dilations_0 = const()[name = string("obj_351_dilations_0"), val = tensor([1, 1])]; + int32 obj_351_groups_0 = const()[name = string("obj_351_groups_0"), val = int32(1)]; + tensor obj_351_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_351_dilations_0, groups = obj_351_groups_0, pad = obj_351_pad_0, pad_type = obj_351_pad_type_0, strides = obj_351_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_337_cast_fp16)[name = string("obj_351_cast_fp16")]; + tensor inputs_331_cast_fp16 = add(x = inputs_325_cast_fp16, y = obj_351_cast_fp16)[name = string("inputs_331_cast_fp16")]; + tensor inputs_sq_331_cast_fp16 = mul(x = inputs_331_cast_fp16, y = inputs_331_cast_fp16)[name = string("inputs_sq_331_cast_fp16")]; + tensor variance_331_axes_0 = const()[name = string("variance_331_axes_0"), val = tensor([1])]; + bool variance_331_keep_dims_0 = const()[name = string("variance_331_keep_dims_0"), val = bool(true)]; + tensor variance_331_cast_fp16 = reduce_mean(axes = variance_331_axes_0, keep_dims = variance_331_keep_dims_0, x = inputs_sq_331_cast_fp16)[name = string("variance_331_cast_fp16")]; + fp16 var_12109_to_fp16 = const()[name = string("op_12109_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12110_cast_fp16 = add(x = variance_331_cast_fp16, y = var_12109_to_fp16)[name = string("op_12110_cast_fp16")]; + fp32 var_12111_epsilon_0 = const()[name = string("op_12111_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12111_cast_fp16 = rsqrt(epsilon = var_12111_epsilon_0, x = var_12110_cast_fp16)[name = string("op_12111_cast_fp16")]; + tensor hidden_states_409_cast_fp16 = mul(x = inputs_331_cast_fp16, y = var_12111_cast_fp16)[name = string("hidden_states_409_cast_fp16")]; + tensor input_339_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_409_cast_fp16)[name = string("input_339_cast_fp16")]; + string input_341_pad_type_0 = const()[name = string("input_341_pad_type_0"), val = string("valid")]; + tensor input_341_strides_0 = const()[name = string("input_341_strides_0"), val = tensor([1, 1])]; + tensor input_341_pad_0 = const()[name = string("input_341_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_341_dilations_0 = const()[name = string("input_341_dilations_0"), val = tensor([1, 1])]; + int32 input_341_groups_0 = const()[name = string("input_341_groups_0"), val = int32(1)]; + tensor input_341_cast_fp16 = conv(dilations = input_341_dilations_0, groups = input_341_groups_0, pad = input_341_pad_0, pad_type = input_341_pad_type_0, strides = input_341_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_339_cast_fp16)[name = string("input_341_cast_fp16")]; + tensor var_12125_cast_fp16 = silu(x = input_341_cast_fp16)[name = string("op_12125_cast_fp16")]; + string var_12131_pad_type_0 = const()[name = string("op_12131_pad_type_0"), val = string("valid")]; + tensor var_12131_strides_0 = const()[name = string("op_12131_strides_0"), val = tensor([1, 1])]; + tensor var_12131_pad_0 = const()[name = string("op_12131_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12131_dilations_0 = const()[name = string("op_12131_dilations_0"), val = tensor([1, 1])]; + int32 var_12131_groups_0 = const()[name = string("op_12131_groups_0"), val = int32(1)]; + tensor var_12131_cast_fp16 = conv(dilations = var_12131_dilations_0, groups = var_12131_groups_0, pad = var_12131_pad_0, pad_type = var_12131_pad_type_0, strides = var_12131_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_339_cast_fp16)[name = string("op_12131_cast_fp16")]; + tensor input_343_cast_fp16 = mul(x = var_12125_cast_fp16, y = var_12131_cast_fp16)[name = string("input_343_cast_fp16")]; + string hidden_states_411_pad_type_0 = const()[name = string("hidden_states_411_pad_type_0"), val = string("valid")]; + tensor hidden_states_411_strides_0 = const()[name = string("hidden_states_411_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_411_pad_0 = const()[name = string("hidden_states_411_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_411_dilations_0 = const()[name = string("hidden_states_411_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_411_groups_0 = const()[name = string("hidden_states_411_groups_0"), val = int32(1)]; + tensor hidden_states_411_cast_fp16 = conv(dilations = hidden_states_411_dilations_0, groups = hidden_states_411_groups_0, pad = hidden_states_411_pad_0, pad_type = hidden_states_411_pad_type_0, strides = hidden_states_411_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_343_cast_fp16)[name = string("hidden_states_411_cast_fp16")]; + tensor inputs_333_cast_fp16 = add(x = inputs_331_cast_fp16, y = hidden_states_411_cast_fp16)[name = string("inputs_333_cast_fp16")]; + int32 var_12159 = const()[name = string("op_12159"), val = int32(1)]; + bool key_caches_17_interleave_0 = const()[name = string("key_caches_17_interleave_0"), val = bool(false)]; + tensor key_caches_17_cast_fp16 = concat(axis = var_12159, interleave = key_caches_17_interleave_0, values = (key_143_cast_fp16, key_147_cast_fp16, key_151_cast_fp16, key_155_cast_fp16, key_159_cast_fp16))[name = string("key_caches_17_cast_fp16")]; + int32 var_12162 = const()[name = string("op_12162"), val = int32(1)]; + bool value_caches_17_interleave_0 = const()[name = string("value_caches_17_interleave_0"), val = bool(false)]; + tensor value_caches_17_cast_fp16 = concat(axis = var_12162, interleave = value_caches_17_interleave_0, values = (value_71_cast_fp16, value_73_cast_fp16, value_75_cast_fp16, value_77_cast_fp16, value_79_cast_fp16))[name = string("value_caches_17_cast_fp16")]; + tensor inputs_sq_333_cast_fp16 = mul(x = inputs_333_cast_fp16, y = inputs_333_cast_fp16)[name = string("inputs_sq_333_cast_fp16")]; + tensor variance_333_axes_0 = const()[name = string("variance_333_axes_0"), val = tensor([1])]; + bool variance_333_keep_dims_0 = const()[name = string("variance_333_keep_dims_0"), val = bool(true)]; + tensor variance_333_cast_fp16 = reduce_mean(axes = variance_333_axes_0, keep_dims = variance_333_keep_dims_0, x = inputs_sq_333_cast_fp16)[name = string("variance_333_cast_fp16")]; + fp16 var_12172_to_fp16 = const()[name = string("op_12172_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12173_cast_fp16 = add(x = variance_333_cast_fp16, y = var_12172_to_fp16)[name = string("op_12173_cast_fp16")]; + fp32 var_12174_epsilon_0 = const()[name = string("op_12174_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12174_cast_fp16 = rsqrt(epsilon = var_12174_epsilon_0, x = var_12173_cast_fp16)[name = string("op_12174_cast_fp16")]; + tensor hidden_states_413_cast_fp16 = mul(x = inputs_333_cast_fp16, y = var_12174_cast_fp16)[name = string("hidden_states_413_cast_fp16")]; + tensor input_345_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_413_cast_fp16)[name = string("input_345_cast_fp16")]; + string logits_25_pad_type_0 = const()[name = string("logits_25_pad_type_0"), val = string("valid")]; + tensor logits_25_strides_0 = const()[name = string("logits_25_strides_0"), val = tensor([1, 1])]; + tensor logits_25_pad_0 = const()[name = string("logits_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_25_dilations_0 = const()[name = string("logits_25_dilations_0"), val = tensor([1, 1])]; + int32 logits_25_groups_0 = const()[name = string("logits_25_groups_0"), val = int32(1)]; + tensor lm_heads_6_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93393344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95490560))))[name = string("lm_heads_6_weight_to_fp16_palettized")]; + tensor logits_25_cast_fp16 = conv(dilations = logits_25_dilations_0, groups = logits_25_groups_0, pad = logits_25_pad_0, pad_type = logits_25_pad_type_0, strides = logits_25_strides_0, weight = lm_heads_6_weight_to_fp16_palettized, x = input_345_cast_fp16)[name = string("logits_25_cast_fp16")]; + tensor var_12192 = const()[name = string("op_12192"), val = tensor([1, 2048])]; + tensor logits_27_cast_fp16 = reshape(shape = var_12192, x = logits_25_cast_fp16)[name = string("logits_27_cast_fp16")]; + tensor scaled_logits_13_cast_fp16 = real_div(x = logits_27_cast_fp16, y = temperature)[name = string("scaled_logits_13_cast_fp16")]; + int32 var_12202 = const()[name = string("op_12202"), val = int32(100)]; + int32 top_values_13_axis_0 = const()[name = string("top_values_13_axis_0"), val = int32(1)]; + bool top_values_13_ascending_0 = const()[name = string("top_values_13_ascending_0"), val = bool(false)]; + bool top_values_13_sort_0 = const()[name = string("top_values_13_sort_0"), val = bool(true)]; + bool top_values_13_return_indices_0 = const()[name = string("top_values_13_return_indices_0"), val = bool(true)]; + string top_values_13_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_13_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_13_cast_fp16_cast_uint16_0, tensor top_values_13_cast_fp16_cast_uint16_1 = topk(ascending = top_values_13_ascending_0, axis = top_values_13_axis_0, k = var_12202, output_indices_dtype = top_values_13_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_13_return_indices_0, sort = top_values_13_sort_0, x = scaled_logits_13_cast_fp16)[name = string("top_values_13_cast_fp16_cast_uint16")]; + tensor var_12208_cast_fp16 = mul(x = top_values_13_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_12208_cast_fp16")]; + tensor var_12212_cast_fp16 = add(x = var_12208_cast_fp16, y = var_3001_cast_fp16)[name = string("op_12212_cast_fp16")]; + tensor reduce_min_6_axes_0 = const()[name = string("reduce_min_6_axes_0"), val = tensor([1])]; + bool reduce_min_6_keep_dims_0 = const()[name = string("reduce_min_6_keep_dims_0"), val = bool(true)]; + tensor reduce_min_6_cast_fp16 = reduce_min(axes = reduce_min_6_axes_0, keep_dims = reduce_min_6_keep_dims_0, x = var_12212_cast_fp16)[name = string("reduce_min_6_cast_fp16")]; + tensor var_12215_cast_fp16 = greater_equal(x = scaled_logits_13_cast_fp16, y = reduce_min_6_cast_fp16)[name = string("op_12215_cast_fp16")]; + fp16 var_12216_value_0_to_fp16 = const()[name = string("op_12216_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_12216_cast_fp16 = fill_like(ref_tensor = scaled_logits_13_cast_fp16, value = var_12216_value_0_to_fp16)[name = string("op_12216_cast_fp16")]; + tensor masked_logits_13_cast_fp16 = select(a = scaled_logits_13_cast_fp16, b = var_12216_cast_fp16, cond = var_12215_cast_fp16)[name = string("masked_logits_13_cast_fp16")]; + tensor var_12220_begin_0 = const()[name = string("op_12220_begin_0"), val = tensor([6, 0])]; + tensor var_12220_end_0 = const()[name = string("op_12220_end_0"), val = tensor([7, 2048])]; + tensor var_12220_end_mask_0 = const()[name = string("op_12220_end_mask_0"), val = tensor([false, true])]; + tensor var_12220_squeeze_mask_0 = const()[name = string("op_12220_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_12220_cast_fp16 = slice_by_index(begin = var_12220_begin_0, end = var_12220_end_0, end_mask = var_12220_end_mask_0, squeeze_mask = var_12220_squeeze_mask_0, x = gumbel)[name = string("op_12220_cast_fp16")]; + tensor var_12223 = const()[name = string("op_12223"), val = tensor([1, 2048])]; + tensor var_12224_cast_fp16 = reshape(shape = var_12223, x = var_12220_cast_fp16)[name = string("op_12224_cast_fp16")]; + tensor noisy_logits_13_cast_fp16 = add(x = masked_logits_13_cast_fp16, y = var_12224_cast_fp16)[name = string("noisy_logits_13_cast_fp16")]; + int32 code_13_axis_0 = const()[name = string("code_13_axis_0"), val = int32(1)]; + bool code_13_keep_dims_0 = const()[name = string("code_13_keep_dims_0"), val = bool(false)]; + string code_13_output_dtype_0 = const()[name = string("code_13_output_dtype_0"), val = string("int32")]; + tensor code_13_cast_fp16 = reduce_argmax(axis = code_13_axis_0, keep_dims = code_13_keep_dims_0, output_dtype = code_13_output_dtype_0, x = noisy_logits_13_cast_fp16)[name = string("code_13_cast_fp16")]; + int32 var_12235 = const()[name = string("op_12235"), val = int32(12288)]; + tensor input_347 = add(x = code_13_cast_fp16, y = var_12235)[name = string("input_347")]; + int32 code_embed_25_axis_0 = const()[name = string("code_embed_25_axis_0"), val = int32(0)]; + int32 code_embed_25_batch_dims_0 = const()[name = string("code_embed_25_batch_dims_0"), val = int32(0)]; + bool code_embed_25_validate_indices_0 = const()[name = string("code_embed_25_validate_indices_0"), val = bool(false)]; + string input_347_to_uint16_dtype_0 = const()[name = string("input_347_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_347_to_uint16 = cast(dtype = input_347_to_uint16_dtype_0, x = input_347)[name = string("cast_8")]; + tensor code_embed_25_cast_fp16_cast_uint16 = gather(axis = code_embed_25_axis_0, batch_dims = code_embed_25_batch_dims_0, indices = input_347_to_uint16, validate_indices = code_embed_25_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_25_cast_fp16_cast_uint16")]; + tensor var_12239 = const()[name = string("op_12239"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_27_cast_fp16 = reshape(shape = var_12239, x = code_embed_25_cast_fp16_cast_uint16)[name = string("code_embed_27_cast_fp16")]; + tensor embed_sum_15_cast_fp16 = add(x = embed_sum_13_cast_fp16, y = code_embed_27_cast_fp16)[name = string("embed_sum_15_cast_fp16")]; + string inputs_335_pad_type_0 = const()[name = string("inputs_335_pad_type_0"), val = string("valid")]; + tensor inputs_335_strides_0 = const()[name = string("inputs_335_strides_0"), val = tensor([1, 1])]; + tensor inputs_335_pad_0 = const()[name = string("inputs_335_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_335_dilations_0 = const()[name = string("inputs_335_dilations_0"), val = tensor([1, 1])]; + int32 inputs_335_groups_0 = const()[name = string("inputs_335_groups_0"), val = int32(1)]; + tensor inputs_335_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_335_dilations_0, groups = inputs_335_groups_0, pad = inputs_335_pad_0, pad_type = inputs_335_pad_type_0, strides = inputs_335_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_27_cast_fp16)[name = string("inputs_335_cast_fp16")]; + tensor obj_355_begin_0 = const()[name = string("obj_355_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_355_end_0 = const()[name = string("obj_355_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_355_end_mask_0 = const()[name = string("obj_355_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_355_cast_fp16 = slice_by_index(begin = obj_355_begin_0, end = obj_355_end_0, end_mask = obj_355_end_mask_0, x = key_caches_17_cast_fp16)[name = string("obj_355_cast_fp16")]; + tensor obj_357_begin_0 = const()[name = string("obj_357_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_357_end_0 = const()[name = string("obj_357_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_357_end_mask_0 = const()[name = string("obj_357_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_357_cast_fp16 = slice_by_index(begin = obj_357_begin_0, end = obj_357_end_0, end_mask = obj_357_end_mask_0, x = value_caches_17_cast_fp16)[name = string("obj_357_cast_fp16")]; + int32 var_12344 = const()[name = string("op_12344"), val = int32(3)]; + int32 var_12354 = const()[name = string("op_12354"), val = int32(-2)]; + tensor inputs_sq_335_cast_fp16 = mul(x = inputs_335_cast_fp16, y = inputs_335_cast_fp16)[name = string("inputs_sq_335_cast_fp16")]; + tensor variance_335_axes_0 = const()[name = string("variance_335_axes_0"), val = tensor([1])]; + bool variance_335_keep_dims_0 = const()[name = string("variance_335_keep_dims_0"), val = bool(true)]; + tensor variance_335_cast_fp16 = reduce_mean(axes = variance_335_axes_0, keep_dims = variance_335_keep_dims_0, x = inputs_sq_335_cast_fp16)[name = string("variance_335_cast_fp16")]; + fp16 var_12368_to_fp16 = const()[name = string("op_12368_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12369_cast_fp16 = add(x = variance_335_cast_fp16, y = var_12368_to_fp16)[name = string("op_12369_cast_fp16")]; + fp32 var_12370_epsilon_0 = const()[name = string("op_12370_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12370_cast_fp16 = rsqrt(epsilon = var_12370_epsilon_0, x = var_12369_cast_fp16)[name = string("op_12370_cast_fp16")]; + tensor hidden_states_415_cast_fp16 = mul(x = inputs_335_cast_fp16, y = var_12370_cast_fp16)[name = string("hidden_states_415_cast_fp16")]; + tensor obj_353_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_415_cast_fp16)[name = string("obj_353_cast_fp16")]; + string query_241_pad_type_0 = const()[name = string("query_241_pad_type_0"), val = string("valid")]; + tensor query_241_strides_0 = const()[name = string("query_241_strides_0"), val = tensor([1, 1])]; + tensor query_241_pad_0 = const()[name = string("query_241_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_241_dilations_0 = const()[name = string("query_241_dilations_0"), val = tensor([1, 1])]; + int32 query_241_groups_0 = const()[name = string("query_241_groups_0"), val = int32(1)]; + tensor query_241_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_241_dilations_0, groups = query_241_groups_0, pad = query_241_pad_0, pad_type = query_241_pad_type_0, strides = query_241_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_353_cast_fp16)[name = string("query_241_cast_fp16")]; + string current_key_161_pad_type_0 = const()[name = string("current_key_161_pad_type_0"), val = string("valid")]; + tensor current_key_161_strides_0 = const()[name = string("current_key_161_strides_0"), val = tensor([1, 1])]; + tensor current_key_161_pad_0 = const()[name = string("current_key_161_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_161_dilations_0 = const()[name = string("current_key_161_dilations_0"), val = tensor([1, 1])]; + int32 current_key_161_groups_0 = const()[name = string("current_key_161_groups_0"), val = int32(1)]; + tensor current_key_161_cast_fp16 = conv(dilations = current_key_161_dilations_0, groups = current_key_161_groups_0, pad = current_key_161_pad_0, pad_type = current_key_161_pad_type_0, strides = current_key_161_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_353_cast_fp16)[name = string("current_key_161_cast_fp16")]; + string current_value_81_pad_type_0 = const()[name = string("current_value_81_pad_type_0"), val = string("valid")]; + tensor current_value_81_strides_0 = const()[name = string("current_value_81_strides_0"), val = tensor([1, 1])]; + tensor current_value_81_pad_0 = const()[name = string("current_value_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_81_dilations_0 = const()[name = string("current_value_81_dilations_0"), val = tensor([1, 1])]; + int32 current_value_81_groups_0 = const()[name = string("current_value_81_groups_0"), val = int32(1)]; + tensor current_value_81_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_81_dilations_0, groups = current_value_81_groups_0, pad = current_value_81_pad_0, pad_type = current_value_81_pad_type_0, strides = current_value_81_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_353_cast_fp16)[name = string("current_value_81_cast_fp16")]; + tensor var_12407 = const()[name = string("op_12407"), val = tensor([16, 128, 1, 1])]; + tensor inputs_337_cast_fp16 = reshape(shape = var_12407, x = query_241_cast_fp16)[name = string("inputs_337_cast_fp16")]; + tensor inputs_sq_337_cast_fp16 = mul(x = inputs_337_cast_fp16, y = inputs_337_cast_fp16)[name = string("inputs_sq_337_cast_fp16")]; + tensor variance_337_axes_0 = const()[name = string("variance_337_axes_0"), val = tensor([1])]; + bool variance_337_keep_dims_0 = const()[name = string("variance_337_keep_dims_0"), val = bool(true)]; + tensor variance_337_cast_fp16 = reduce_mean(axes = variance_337_axes_0, keep_dims = variance_337_keep_dims_0, x = inputs_sq_337_cast_fp16)[name = string("variance_337_cast_fp16")]; + fp16 var_12413_to_fp16 = const()[name = string("op_12413_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12414_cast_fp16 = add(x = variance_337_cast_fp16, y = var_12413_to_fp16)[name = string("op_12414_cast_fp16")]; + fp32 var_12415_epsilon_0 = const()[name = string("op_12415_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12415_cast_fp16 = rsqrt(epsilon = var_12415_epsilon_0, x = var_12414_cast_fp16)[name = string("op_12415_cast_fp16")]; + tensor hidden_states_417_cast_fp16 = mul(x = inputs_337_cast_fp16, y = var_12415_cast_fp16)[name = string("hidden_states_417_cast_fp16")]; + tensor query_normed_81_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_417_cast_fp16)[name = string("query_normed_81_cast_fp16")]; + tensor var_12423 = const()[name = string("op_12423"), val = tensor([8, 128, 1, 1])]; + tensor inputs_339_cast_fp16 = reshape(shape = var_12423, x = current_key_161_cast_fp16)[name = string("inputs_339_cast_fp16")]; + tensor inputs_sq_339_cast_fp16 = mul(x = inputs_339_cast_fp16, y = inputs_339_cast_fp16)[name = string("inputs_sq_339_cast_fp16")]; + tensor variance_339_axes_0 = const()[name = string("variance_339_axes_0"), val = tensor([1])]; + bool variance_339_keep_dims_0 = const()[name = string("variance_339_keep_dims_0"), val = bool(true)]; + tensor variance_339_cast_fp16 = reduce_mean(axes = variance_339_axes_0, keep_dims = variance_339_keep_dims_0, x = inputs_sq_339_cast_fp16)[name = string("variance_339_cast_fp16")]; + fp16 var_12429_to_fp16 = const()[name = string("op_12429_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12430_cast_fp16 = add(x = variance_339_cast_fp16, y = var_12429_to_fp16)[name = string("op_12430_cast_fp16")]; + fp32 var_12431_epsilon_0 = const()[name = string("op_12431_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12431_cast_fp16 = rsqrt(epsilon = var_12431_epsilon_0, x = var_12430_cast_fp16)[name = string("op_12431_cast_fp16")]; + tensor hidden_states_419_cast_fp16 = mul(x = inputs_339_cast_fp16, y = var_12431_cast_fp16)[name = string("hidden_states_419_cast_fp16")]; + tensor current_key_normed_81_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_419_cast_fp16)[name = string("current_key_normed_81_cast_fp16")]; + tensor var_12449 = const()[name = string("op_12449"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_321_cast_fp16 = reshape(shape = var_12449, x = query_normed_81_cast_fp16)[name = string("mh_q_321_cast_fp16")]; + tensor var_12451 = const()[name = string("op_12451"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_321_cast_fp16 = reshape(shape = var_12451, x = current_key_normed_81_cast_fp16)[name = string("mh_k_321_cast_fp16")]; + tensor cos_81_to_fp16 = const()[name = string("cos_81_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175196736)))]; + tensor var_12455_cast_fp16 = mul(x = mh_q_321_cast_fp16, y = cos_81_to_fp16)[name = string("op_12455_cast_fp16")]; + tensor var_12460_begin_0 = const()[name = string("op_12460_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12460_end_0 = const()[name = string("op_12460_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_12460_end_mask_0 = const()[name = string("op_12460_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_12460_cast_fp16 = slice_by_index(begin = var_12460_begin_0, end = var_12460_end_0, end_mask = var_12460_end_mask_0, x = mh_q_321_cast_fp16)[name = string("op_12460_cast_fp16")]; + tensor var_12466_begin_0 = const()[name = string("op_12466_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_12466_end_0 = const()[name = string("op_12466_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_12466_end_mask_0 = const()[name = string("op_12466_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_12466_cast_fp16 = slice_by_index(begin = var_12466_begin_0, end = var_12466_end_0, end_mask = var_12466_end_mask_0, x = mh_q_321_cast_fp16)[name = string("op_12466_cast_fp16")]; + fp16 const_822_promoted_to_fp16 = const()[name = string("const_822_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12468_cast_fp16 = mul(x = var_12466_cast_fp16, y = const_822_promoted_to_fp16)[name = string("op_12468_cast_fp16")]; + bool var_12470_interleave_0 = const()[name = string("op_12470_interleave_0"), val = bool(false)]; + tensor var_12470_cast_fp16 = concat(axis = var_12354, interleave = var_12470_interleave_0, values = (var_12468_cast_fp16, var_12460_cast_fp16))[name = string("op_12470_cast_fp16")]; + tensor sin_81_to_fp16 = const()[name = string("sin_81_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197056)))]; + tensor var_12471_cast_fp16 = mul(x = var_12470_cast_fp16, y = sin_81_to_fp16)[name = string("op_12471_cast_fp16")]; + tensor mh_q_323_cast_fp16 = add(x = var_12455_cast_fp16, y = var_12471_cast_fp16)[name = string("mh_q_323_cast_fp16")]; + tensor var_12473_cast_fp16 = mul(x = mh_k_321_cast_fp16, y = cos_81_to_fp16)[name = string("op_12473_cast_fp16")]; + tensor var_12478_begin_0 = const()[name = string("op_12478_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12478_end_0 = const()[name = string("op_12478_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_12478_end_mask_0 = const()[name = string("op_12478_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_12478_cast_fp16 = slice_by_index(begin = var_12478_begin_0, end = var_12478_end_0, end_mask = var_12478_end_mask_0, x = mh_k_321_cast_fp16)[name = string("op_12478_cast_fp16")]; + tensor var_12484_begin_0 = const()[name = string("op_12484_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_12484_end_0 = const()[name = string("op_12484_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_12484_end_mask_0 = const()[name = string("op_12484_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_12484_cast_fp16 = slice_by_index(begin = var_12484_begin_0, end = var_12484_end_0, end_mask = var_12484_end_mask_0, x = mh_k_321_cast_fp16)[name = string("op_12484_cast_fp16")]; + fp16 const_825_promoted_to_fp16 = const()[name = string("const_825_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12486_cast_fp16 = mul(x = var_12484_cast_fp16, y = const_825_promoted_to_fp16)[name = string("op_12486_cast_fp16")]; + bool var_12488_interleave_0 = const()[name = string("op_12488_interleave_0"), val = bool(false)]; + tensor var_12488_cast_fp16 = concat(axis = var_12354, interleave = var_12488_interleave_0, values = (var_12486_cast_fp16, var_12478_cast_fp16))[name = string("op_12488_cast_fp16")]; + tensor var_12489_cast_fp16 = mul(x = var_12488_cast_fp16, y = sin_81_to_fp16)[name = string("op_12489_cast_fp16")]; + tensor mh_k_323_cast_fp16 = add(x = var_12473_cast_fp16, y = var_12489_cast_fp16)[name = string("mh_k_323_cast_fp16")]; + tensor var_12493 = const()[name = string("op_12493"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_163_cast_fp16 = reshape(shape = var_12493, x = mh_k_323_cast_fp16)[name = string("current_key_163_cast_fp16")]; + tensor var_12499_to_fp16 = const()[name = string("op_12499_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197376)))]; + tensor var_12500_cast_fp16 = mul(x = obj_355_cast_fp16, y = var_12499_to_fp16)[name = string("op_12500_cast_fp16")]; + tensor var_12497_to_fp16 = const()[name = string("op_12497_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197504)))]; + tensor var_12501_cast_fp16 = mul(x = current_key_163_cast_fp16, y = var_12497_to_fp16)[name = string("op_12501_cast_fp16")]; + tensor key_163_cast_fp16 = add(x = var_12500_cast_fp16, y = var_12501_cast_fp16)[name = string("key_163_cast_fp16")]; + tensor var_12503_to_fp16 = const()[name = string("op_12503_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197376)))]; + tensor var_12504_cast_fp16 = mul(x = obj_357_cast_fp16, y = var_12503_to_fp16)[name = string("op_12504_cast_fp16")]; + tensor var_12505_cast_fp16 = mul(x = current_value_81_cast_fp16, y = var_12497_to_fp16)[name = string("op_12505_cast_fp16")]; + tensor value_81_cast_fp16 = add(x = var_12504_cast_fp16, y = var_12505_cast_fp16)[name = string("value_81_cast_fp16")]; + fp16 var_12512_to_fp16 = const()[name = string("op_12512_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_327_cast_fp16 = mul(x = mh_q_323_cast_fp16, y = var_12512_to_fp16)[name = string("mh_q_327_cast_fp16")]; + tensor var_12514 = const()[name = string("op_12514"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_325_cast_fp16 = reshape(shape = var_12514, x = key_163_cast_fp16)[name = string("mh_k_325_cast_fp16")]; + tensor var_12516 = const()[name = string("op_12516"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_161_cast_fp16 = reshape(shape = var_12516, x = value_81_cast_fp16)[name = string("mh_v_161_cast_fp16")]; + tensor transpose_160_perm_0 = const()[name = string("transpose_160_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_80_reps_0 = const()[name = string("tile_80_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_160_cast_fp16 = transpose(perm = transpose_160_perm_0, x = mh_k_325_cast_fp16)[name = string("transpose_239")]; + tensor tile_80_cast_fp16 = tile(reps = tile_80_reps_0, x = transpose_160_cast_fp16)[name = string("tile_80_cast_fp16")]; + tensor concat_203 = const()[name = string("concat_203"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_160_cast_fp16 = reshape(shape = concat_203, x = tile_80_cast_fp16)[name = string("reshape_160_cast_fp16")]; + tensor transpose_161_perm_0 = const()[name = string("transpose_161_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_204 = const()[name = string("concat_204"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_161_cast_fp16 = transpose(perm = transpose_161_perm_0, x = reshape_160_cast_fp16)[name = string("transpose_238")]; + tensor reshape_161_cast_fp16 = reshape(shape = concat_204, x = transpose_161_cast_fp16)[name = string("reshape_161_cast_fp16")]; + tensor transpose_162_perm_0 = const()[name = string("transpose_162_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_81_reps_0 = const()[name = string("tile_81_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_162_cast_fp16 = transpose(perm = transpose_162_perm_0, x = mh_v_161_cast_fp16)[name = string("transpose_237")]; + tensor tile_81_cast_fp16 = tile(reps = tile_81_reps_0, x = transpose_162_cast_fp16)[name = string("tile_81_cast_fp16")]; + tensor concat_205 = const()[name = string("concat_205"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_162_cast_fp16 = reshape(shape = concat_205, x = tile_81_cast_fp16)[name = string("reshape_162_cast_fp16")]; + tensor transpose_163_perm_0 = const()[name = string("transpose_163_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_206 = const()[name = string("concat_206"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_163_cast_fp16 = transpose(perm = transpose_163_perm_0, x = reshape_162_cast_fp16)[name = string("transpose_236")]; + tensor reshape_163_cast_fp16 = reshape(shape = concat_206, x = transpose_163_cast_fp16)[name = string("reshape_163_cast_fp16")]; + tensor transpose_477_perm_0 = const()[name = string("transpose_477_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_241_transpose_x_1 = const()[name = string("mh_w_241_transpose_x_1"), val = bool(true)]; + bool mh_w_241_transpose_y_1 = const()[name = string("mh_w_241_transpose_y_1"), val = bool(false)]; + tensor transpose_477_cast_fp16 = transpose(perm = transpose_477_perm_0, x = reshape_161_cast_fp16)[name = string("transpose_235")]; + tensor mh_w_241_cast_fp16 = matmul(transpose_x = mh_w_241_transpose_x_1, transpose_y = mh_w_241_transpose_y_1, x = mh_q_327_cast_fp16, y = transpose_477_cast_fp16)[name = string("mh_w_241_cast_fp16")]; + tensor var_12524_to_fp16 = const()[name = string("op_12524_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197632)))]; + tensor mh_w_243_cast_fp16 = add(x = mh_w_241_cast_fp16, y = var_12524_to_fp16)[name = string("mh_w_243_cast_fp16")]; + tensor mh_w_245_cast_fp16 = softmax(axis = var_12344, x = mh_w_243_cast_fp16)[name = string("mh_w_245_cast_fp16")]; + tensor transpose_478_perm_0 = const()[name = string("transpose_478_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_81_transpose_x_1 = const()[name = string("attn_81_transpose_x_1"), val = bool(false)]; + bool attn_81_transpose_y_1 = const()[name = string("attn_81_transpose_y_1"), val = bool(true)]; + tensor transpose_478_cast_fp16 = transpose(perm = transpose_478_perm_0, x = reshape_163_cast_fp16)[name = string("transpose_234")]; + tensor attn_81_cast_fp16 = matmul(transpose_x = attn_81_transpose_x_1, transpose_y = attn_81_transpose_y_1, x = transpose_478_cast_fp16, y = mh_w_245_cast_fp16)[name = string("attn_81_cast_fp16")]; + tensor var_12530 = const()[name = string("op_12530"), val = tensor([1, 2048, 1, 1])]; + tensor input_349_cast_fp16 = reshape(shape = var_12530, x = attn_81_cast_fp16)[name = string("input_349_cast_fp16")]; + string obj_363_pad_type_0 = const()[name = string("obj_363_pad_type_0"), val = string("valid")]; + tensor obj_363_strides_0 = const()[name = string("obj_363_strides_0"), val = tensor([1, 1])]; + tensor obj_363_pad_0 = const()[name = string("obj_363_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_363_dilations_0 = const()[name = string("obj_363_dilations_0"), val = tensor([1, 1])]; + int32 obj_363_groups_0 = const()[name = string("obj_363_groups_0"), val = int32(1)]; + tensor obj_363_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_363_dilations_0, groups = obj_363_groups_0, pad = obj_363_pad_0, pad_type = obj_363_pad_type_0, strides = obj_363_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_349_cast_fp16)[name = string("obj_363_cast_fp16")]; + tensor inputs_341_cast_fp16 = add(x = inputs_335_cast_fp16, y = obj_363_cast_fp16)[name = string("inputs_341_cast_fp16")]; + tensor inputs_sq_341_cast_fp16 = mul(x = inputs_341_cast_fp16, y = inputs_341_cast_fp16)[name = string("inputs_sq_341_cast_fp16")]; + tensor variance_341_axes_0 = const()[name = string("variance_341_axes_0"), val = tensor([1])]; + bool variance_341_keep_dims_0 = const()[name = string("variance_341_keep_dims_0"), val = bool(true)]; + tensor variance_341_cast_fp16 = reduce_mean(axes = variance_341_axes_0, keep_dims = variance_341_keep_dims_0, x = inputs_sq_341_cast_fp16)[name = string("variance_341_cast_fp16")]; + fp16 var_12548_to_fp16 = const()[name = string("op_12548_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12549_cast_fp16 = add(x = variance_341_cast_fp16, y = var_12548_to_fp16)[name = string("op_12549_cast_fp16")]; + fp32 var_12550_epsilon_0 = const()[name = string("op_12550_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12550_cast_fp16 = rsqrt(epsilon = var_12550_epsilon_0, x = var_12549_cast_fp16)[name = string("op_12550_cast_fp16")]; + tensor hidden_states_421_cast_fp16 = mul(x = inputs_341_cast_fp16, y = var_12550_cast_fp16)[name = string("hidden_states_421_cast_fp16")]; + tensor input_351_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_421_cast_fp16)[name = string("input_351_cast_fp16")]; + string input_353_pad_type_0 = const()[name = string("input_353_pad_type_0"), val = string("valid")]; + tensor input_353_strides_0 = const()[name = string("input_353_strides_0"), val = tensor([1, 1])]; + tensor input_353_pad_0 = const()[name = string("input_353_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_353_dilations_0 = const()[name = string("input_353_dilations_0"), val = tensor([1, 1])]; + int32 input_353_groups_0 = const()[name = string("input_353_groups_0"), val = int32(1)]; + tensor input_353_cast_fp16 = conv(dilations = input_353_dilations_0, groups = input_353_groups_0, pad = input_353_pad_0, pad_type = input_353_pad_type_0, strides = input_353_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_351_cast_fp16)[name = string("input_353_cast_fp16")]; + tensor var_12564_cast_fp16 = silu(x = input_353_cast_fp16)[name = string("op_12564_cast_fp16")]; + string var_12570_pad_type_0 = const()[name = string("op_12570_pad_type_0"), val = string("valid")]; + tensor var_12570_strides_0 = const()[name = string("op_12570_strides_0"), val = tensor([1, 1])]; + tensor var_12570_pad_0 = const()[name = string("op_12570_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12570_dilations_0 = const()[name = string("op_12570_dilations_0"), val = tensor([1, 1])]; + int32 var_12570_groups_0 = const()[name = string("op_12570_groups_0"), val = int32(1)]; + tensor var_12570_cast_fp16 = conv(dilations = var_12570_dilations_0, groups = var_12570_groups_0, pad = var_12570_pad_0, pad_type = var_12570_pad_type_0, strides = var_12570_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_351_cast_fp16)[name = string("op_12570_cast_fp16")]; + tensor input_355_cast_fp16 = mul(x = var_12564_cast_fp16, y = var_12570_cast_fp16)[name = string("input_355_cast_fp16")]; + string hidden_states_423_pad_type_0 = const()[name = string("hidden_states_423_pad_type_0"), val = string("valid")]; + tensor hidden_states_423_strides_0 = const()[name = string("hidden_states_423_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_423_pad_0 = const()[name = string("hidden_states_423_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_423_dilations_0 = const()[name = string("hidden_states_423_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_423_groups_0 = const()[name = string("hidden_states_423_groups_0"), val = int32(1)]; + tensor hidden_states_423_cast_fp16 = conv(dilations = hidden_states_423_dilations_0, groups = hidden_states_423_groups_0, pad = hidden_states_423_pad_0, pad_type = hidden_states_423_pad_type_0, strides = hidden_states_423_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_355_cast_fp16)[name = string("hidden_states_423_cast_fp16")]; + tensor inputs_343_cast_fp16 = add(x = inputs_341_cast_fp16, y = hidden_states_423_cast_fp16)[name = string("inputs_343_cast_fp16")]; + tensor obj_367_begin_0 = const()[name = string("obj_367_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_367_end_0 = const()[name = string("obj_367_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_367_end_mask_0 = const()[name = string("obj_367_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_367_cast_fp16 = slice_by_index(begin = obj_367_begin_0, end = obj_367_end_0, end_mask = obj_367_end_mask_0, x = key_caches_17_cast_fp16)[name = string("obj_367_cast_fp16")]; + tensor obj_369_begin_0 = const()[name = string("obj_369_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_369_end_0 = const()[name = string("obj_369_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_369_end_mask_0 = const()[name = string("obj_369_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_369_cast_fp16 = slice_by_index(begin = obj_369_begin_0, end = obj_369_end_0, end_mask = obj_369_end_mask_0, x = value_caches_17_cast_fp16)[name = string("obj_369_cast_fp16")]; + int32 var_12618 = const()[name = string("op_12618"), val = int32(3)]; + int32 var_12628 = const()[name = string("op_12628"), val = int32(-2)]; + tensor inputs_sq_343_cast_fp16 = mul(x = inputs_343_cast_fp16, y = inputs_343_cast_fp16)[name = string("inputs_sq_343_cast_fp16")]; + tensor variance_343_axes_0 = const()[name = string("variance_343_axes_0"), val = tensor([1])]; + bool variance_343_keep_dims_0 = const()[name = string("variance_343_keep_dims_0"), val = bool(true)]; + tensor variance_343_cast_fp16 = reduce_mean(axes = variance_343_axes_0, keep_dims = variance_343_keep_dims_0, x = inputs_sq_343_cast_fp16)[name = string("variance_343_cast_fp16")]; + fp16 var_12642_to_fp16 = const()[name = string("op_12642_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12643_cast_fp16 = add(x = variance_343_cast_fp16, y = var_12642_to_fp16)[name = string("op_12643_cast_fp16")]; + fp32 var_12644_epsilon_0 = const()[name = string("op_12644_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12644_cast_fp16 = rsqrt(epsilon = var_12644_epsilon_0, x = var_12643_cast_fp16)[name = string("op_12644_cast_fp16")]; + tensor hidden_states_425_cast_fp16 = mul(x = inputs_343_cast_fp16, y = var_12644_cast_fp16)[name = string("hidden_states_425_cast_fp16")]; + tensor obj_365_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_425_cast_fp16)[name = string("obj_365_cast_fp16")]; + string query_247_pad_type_0 = const()[name = string("query_247_pad_type_0"), val = string("valid")]; + tensor query_247_strides_0 = const()[name = string("query_247_strides_0"), val = tensor([1, 1])]; + tensor query_247_pad_0 = const()[name = string("query_247_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_247_dilations_0 = const()[name = string("query_247_dilations_0"), val = tensor([1, 1])]; + int32 query_247_groups_0 = const()[name = string("query_247_groups_0"), val = int32(1)]; + tensor query_247_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_247_dilations_0, groups = query_247_groups_0, pad = query_247_pad_0, pad_type = query_247_pad_type_0, strides = query_247_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_365_cast_fp16)[name = string("query_247_cast_fp16")]; + string current_key_165_pad_type_0 = const()[name = string("current_key_165_pad_type_0"), val = string("valid")]; + tensor current_key_165_strides_0 = const()[name = string("current_key_165_strides_0"), val = tensor([1, 1])]; + tensor current_key_165_pad_0 = const()[name = string("current_key_165_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_165_dilations_0 = const()[name = string("current_key_165_dilations_0"), val = tensor([1, 1])]; + int32 current_key_165_groups_0 = const()[name = string("current_key_165_groups_0"), val = int32(1)]; + tensor current_key_165_cast_fp16 = conv(dilations = current_key_165_dilations_0, groups = current_key_165_groups_0, pad = current_key_165_pad_0, pad_type = current_key_165_pad_type_0, strides = current_key_165_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_365_cast_fp16)[name = string("current_key_165_cast_fp16")]; + string current_value_83_pad_type_0 = const()[name = string("current_value_83_pad_type_0"), val = string("valid")]; + tensor current_value_83_strides_0 = const()[name = string("current_value_83_strides_0"), val = tensor([1, 1])]; + tensor current_value_83_pad_0 = const()[name = string("current_value_83_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_83_dilations_0 = const()[name = string("current_value_83_dilations_0"), val = tensor([1, 1])]; + int32 current_value_83_groups_0 = const()[name = string("current_value_83_groups_0"), val = int32(1)]; + tensor current_value_83_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_83_dilations_0, groups = current_value_83_groups_0, pad = current_value_83_pad_0, pad_type = current_value_83_pad_type_0, strides = current_value_83_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_365_cast_fp16)[name = string("current_value_83_cast_fp16")]; + tensor var_12681 = const()[name = string("op_12681"), val = tensor([16, 128, 1, 1])]; + tensor inputs_345_cast_fp16 = reshape(shape = var_12681, x = query_247_cast_fp16)[name = string("inputs_345_cast_fp16")]; + tensor inputs_sq_345_cast_fp16 = mul(x = inputs_345_cast_fp16, y = inputs_345_cast_fp16)[name = string("inputs_sq_345_cast_fp16")]; + tensor variance_345_axes_0 = const()[name = string("variance_345_axes_0"), val = tensor([1])]; + bool variance_345_keep_dims_0 = const()[name = string("variance_345_keep_dims_0"), val = bool(true)]; + tensor variance_345_cast_fp16 = reduce_mean(axes = variance_345_axes_0, keep_dims = variance_345_keep_dims_0, x = inputs_sq_345_cast_fp16)[name = string("variance_345_cast_fp16")]; + fp16 var_12687_to_fp16 = const()[name = string("op_12687_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12688_cast_fp16 = add(x = variance_345_cast_fp16, y = var_12687_to_fp16)[name = string("op_12688_cast_fp16")]; + fp32 var_12689_epsilon_0 = const()[name = string("op_12689_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12689_cast_fp16 = rsqrt(epsilon = var_12689_epsilon_0, x = var_12688_cast_fp16)[name = string("op_12689_cast_fp16")]; + tensor hidden_states_427_cast_fp16 = mul(x = inputs_345_cast_fp16, y = var_12689_cast_fp16)[name = string("hidden_states_427_cast_fp16")]; + tensor query_normed_83_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_427_cast_fp16)[name = string("query_normed_83_cast_fp16")]; + tensor var_12697 = const()[name = string("op_12697"), val = tensor([8, 128, 1, 1])]; + tensor inputs_347_cast_fp16 = reshape(shape = var_12697, x = current_key_165_cast_fp16)[name = string("inputs_347_cast_fp16")]; + tensor inputs_sq_347_cast_fp16 = mul(x = inputs_347_cast_fp16, y = inputs_347_cast_fp16)[name = string("inputs_sq_347_cast_fp16")]; + tensor variance_347_axes_0 = const()[name = string("variance_347_axes_0"), val = tensor([1])]; + bool variance_347_keep_dims_0 = const()[name = string("variance_347_keep_dims_0"), val = bool(true)]; + tensor variance_347_cast_fp16 = reduce_mean(axes = variance_347_axes_0, keep_dims = variance_347_keep_dims_0, x = inputs_sq_347_cast_fp16)[name = string("variance_347_cast_fp16")]; + fp16 var_12703_to_fp16 = const()[name = string("op_12703_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12704_cast_fp16 = add(x = variance_347_cast_fp16, y = var_12703_to_fp16)[name = string("op_12704_cast_fp16")]; + fp32 var_12705_epsilon_0 = const()[name = string("op_12705_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12705_cast_fp16 = rsqrt(epsilon = var_12705_epsilon_0, x = var_12704_cast_fp16)[name = string("op_12705_cast_fp16")]; + tensor hidden_states_429_cast_fp16 = mul(x = inputs_347_cast_fp16, y = var_12705_cast_fp16)[name = string("hidden_states_429_cast_fp16")]; + tensor current_key_normed_83_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_429_cast_fp16)[name = string("current_key_normed_83_cast_fp16")]; + tensor var_12723 = const()[name = string("op_12723"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_329_cast_fp16 = reshape(shape = var_12723, x = query_normed_83_cast_fp16)[name = string("mh_q_329_cast_fp16")]; + tensor var_12725 = const()[name = string("op_12725"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_329_cast_fp16 = reshape(shape = var_12725, x = current_key_normed_83_cast_fp16)[name = string("mh_k_329_cast_fp16")]; + tensor var_12729_cast_fp16 = mul(x = mh_q_329_cast_fp16, y = cos_81_to_fp16)[name = string("op_12729_cast_fp16")]; + tensor var_12734_begin_0 = const()[name = string("op_12734_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12734_end_0 = const()[name = string("op_12734_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_12734_end_mask_0 = const()[name = string("op_12734_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_12734_cast_fp16 = slice_by_index(begin = var_12734_begin_0, end = var_12734_end_0, end_mask = var_12734_end_mask_0, x = mh_q_329_cast_fp16)[name = string("op_12734_cast_fp16")]; + tensor var_12740_begin_0 = const()[name = string("op_12740_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_12740_end_0 = const()[name = string("op_12740_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_12740_end_mask_0 = const()[name = string("op_12740_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_12740_cast_fp16 = slice_by_index(begin = var_12740_begin_0, end = var_12740_end_0, end_mask = var_12740_end_mask_0, x = mh_q_329_cast_fp16)[name = string("op_12740_cast_fp16")]; + fp16 const_842_promoted_to_fp16 = const()[name = string("const_842_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12742_cast_fp16 = mul(x = var_12740_cast_fp16, y = const_842_promoted_to_fp16)[name = string("op_12742_cast_fp16")]; + bool var_12744_interleave_0 = const()[name = string("op_12744_interleave_0"), val = bool(false)]; + tensor var_12744_cast_fp16 = concat(axis = var_12628, interleave = var_12744_interleave_0, values = (var_12742_cast_fp16, var_12734_cast_fp16))[name = string("op_12744_cast_fp16")]; + tensor var_12745_cast_fp16 = mul(x = var_12744_cast_fp16, y = sin_81_to_fp16)[name = string("op_12745_cast_fp16")]; + tensor mh_q_331_cast_fp16 = add(x = var_12729_cast_fp16, y = var_12745_cast_fp16)[name = string("mh_q_331_cast_fp16")]; + tensor var_12747_cast_fp16 = mul(x = mh_k_329_cast_fp16, y = cos_81_to_fp16)[name = string("op_12747_cast_fp16")]; + tensor var_12752_begin_0 = const()[name = string("op_12752_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12752_end_0 = const()[name = string("op_12752_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_12752_end_mask_0 = const()[name = string("op_12752_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_12752_cast_fp16 = slice_by_index(begin = var_12752_begin_0, end = var_12752_end_0, end_mask = var_12752_end_mask_0, x = mh_k_329_cast_fp16)[name = string("op_12752_cast_fp16")]; + tensor var_12758_begin_0 = const()[name = string("op_12758_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_12758_end_0 = const()[name = string("op_12758_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_12758_end_mask_0 = const()[name = string("op_12758_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_12758_cast_fp16 = slice_by_index(begin = var_12758_begin_0, end = var_12758_end_0, end_mask = var_12758_end_mask_0, x = mh_k_329_cast_fp16)[name = string("op_12758_cast_fp16")]; + fp16 const_845_promoted_to_fp16 = const()[name = string("const_845_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_12760_cast_fp16 = mul(x = var_12758_cast_fp16, y = const_845_promoted_to_fp16)[name = string("op_12760_cast_fp16")]; + bool var_12762_interleave_0 = const()[name = string("op_12762_interleave_0"), val = bool(false)]; + tensor var_12762_cast_fp16 = concat(axis = var_12628, interleave = var_12762_interleave_0, values = (var_12760_cast_fp16, var_12752_cast_fp16))[name = string("op_12762_cast_fp16")]; + tensor var_12763_cast_fp16 = mul(x = var_12762_cast_fp16, y = sin_81_to_fp16)[name = string("op_12763_cast_fp16")]; + tensor mh_k_331_cast_fp16 = add(x = var_12747_cast_fp16, y = var_12763_cast_fp16)[name = string("mh_k_331_cast_fp16")]; + tensor var_12767 = const()[name = string("op_12767"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_167_cast_fp16 = reshape(shape = var_12767, x = mh_k_331_cast_fp16)[name = string("current_key_167_cast_fp16")]; + tensor var_12773_to_fp16 = const()[name = string("op_12773_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197376)))]; + tensor var_12774_cast_fp16 = mul(x = obj_367_cast_fp16, y = var_12773_to_fp16)[name = string("op_12774_cast_fp16")]; + tensor var_12771_to_fp16 = const()[name = string("op_12771_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197504)))]; + tensor var_12775_cast_fp16 = mul(x = current_key_167_cast_fp16, y = var_12771_to_fp16)[name = string("op_12775_cast_fp16")]; + tensor key_167_cast_fp16 = add(x = var_12774_cast_fp16, y = var_12775_cast_fp16)[name = string("key_167_cast_fp16")]; + tensor var_12777_to_fp16 = const()[name = string("op_12777_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197376)))]; + tensor var_12778_cast_fp16 = mul(x = obj_369_cast_fp16, y = var_12777_to_fp16)[name = string("op_12778_cast_fp16")]; + tensor var_12779_cast_fp16 = mul(x = current_value_83_cast_fp16, y = var_12771_to_fp16)[name = string("op_12779_cast_fp16")]; + tensor value_83_cast_fp16 = add(x = var_12778_cast_fp16, y = var_12779_cast_fp16)[name = string("value_83_cast_fp16")]; + fp16 var_12786_to_fp16 = const()[name = string("op_12786_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_335_cast_fp16 = mul(x = mh_q_331_cast_fp16, y = var_12786_to_fp16)[name = string("mh_q_335_cast_fp16")]; + tensor var_12788 = const()[name = string("op_12788"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_333_cast_fp16 = reshape(shape = var_12788, x = key_167_cast_fp16)[name = string("mh_k_333_cast_fp16")]; + tensor var_12790 = const()[name = string("op_12790"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_165_cast_fp16 = reshape(shape = var_12790, x = value_83_cast_fp16)[name = string("mh_v_165_cast_fp16")]; + tensor transpose_164_perm_0 = const()[name = string("transpose_164_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_82_reps_0 = const()[name = string("tile_82_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_164_cast_fp16 = transpose(perm = transpose_164_perm_0, x = mh_k_333_cast_fp16)[name = string("transpose_233")]; + tensor tile_82_cast_fp16 = tile(reps = tile_82_reps_0, x = transpose_164_cast_fp16)[name = string("tile_82_cast_fp16")]; + tensor concat_207 = const()[name = string("concat_207"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_164_cast_fp16 = reshape(shape = concat_207, x = tile_82_cast_fp16)[name = string("reshape_164_cast_fp16")]; + tensor transpose_165_perm_0 = const()[name = string("transpose_165_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_208 = const()[name = string("concat_208"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_165_cast_fp16 = transpose(perm = transpose_165_perm_0, x = reshape_164_cast_fp16)[name = string("transpose_232")]; + tensor reshape_165_cast_fp16 = reshape(shape = concat_208, x = transpose_165_cast_fp16)[name = string("reshape_165_cast_fp16")]; + tensor transpose_166_perm_0 = const()[name = string("transpose_166_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_83_reps_0 = const()[name = string("tile_83_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_166_cast_fp16 = transpose(perm = transpose_166_perm_0, x = mh_v_165_cast_fp16)[name = string("transpose_231")]; + tensor tile_83_cast_fp16 = tile(reps = tile_83_reps_0, x = transpose_166_cast_fp16)[name = string("tile_83_cast_fp16")]; + tensor concat_209 = const()[name = string("concat_209"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_166_cast_fp16 = reshape(shape = concat_209, x = tile_83_cast_fp16)[name = string("reshape_166_cast_fp16")]; + tensor transpose_167_perm_0 = const()[name = string("transpose_167_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_210 = const()[name = string("concat_210"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_167_cast_fp16 = transpose(perm = transpose_167_perm_0, x = reshape_166_cast_fp16)[name = string("transpose_230")]; + tensor reshape_167_cast_fp16 = reshape(shape = concat_210, x = transpose_167_cast_fp16)[name = string("reshape_167_cast_fp16")]; + tensor transpose_481_perm_0 = const()[name = string("transpose_481_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_247_transpose_x_1 = const()[name = string("mh_w_247_transpose_x_1"), val = bool(true)]; + bool mh_w_247_transpose_y_1 = const()[name = string("mh_w_247_transpose_y_1"), val = bool(false)]; + tensor transpose_481_cast_fp16 = transpose(perm = transpose_481_perm_0, x = reshape_165_cast_fp16)[name = string("transpose_229")]; + tensor mh_w_247_cast_fp16 = matmul(transpose_x = mh_w_247_transpose_x_1, transpose_y = mh_w_247_transpose_y_1, x = mh_q_335_cast_fp16, y = transpose_481_cast_fp16)[name = string("mh_w_247_cast_fp16")]; + tensor var_12798_to_fp16 = const()[name = string("op_12798_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197632)))]; + tensor mh_w_249_cast_fp16 = add(x = mh_w_247_cast_fp16, y = var_12798_to_fp16)[name = string("mh_w_249_cast_fp16")]; + tensor mh_w_251_cast_fp16 = softmax(axis = var_12618, x = mh_w_249_cast_fp16)[name = string("mh_w_251_cast_fp16")]; + tensor transpose_482_perm_0 = const()[name = string("transpose_482_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_83_transpose_x_1 = const()[name = string("attn_83_transpose_x_1"), val = bool(false)]; + bool attn_83_transpose_y_1 = const()[name = string("attn_83_transpose_y_1"), val = bool(true)]; + tensor transpose_482_cast_fp16 = transpose(perm = transpose_482_perm_0, x = reshape_167_cast_fp16)[name = string("transpose_228")]; + tensor attn_83_cast_fp16 = matmul(transpose_x = attn_83_transpose_x_1, transpose_y = attn_83_transpose_y_1, x = transpose_482_cast_fp16, y = mh_w_251_cast_fp16)[name = string("attn_83_cast_fp16")]; + tensor var_12804 = const()[name = string("op_12804"), val = tensor([1, 2048, 1, 1])]; + tensor input_357_cast_fp16 = reshape(shape = var_12804, x = attn_83_cast_fp16)[name = string("input_357_cast_fp16")]; + string obj_371_pad_type_0 = const()[name = string("obj_371_pad_type_0"), val = string("valid")]; + tensor obj_371_strides_0 = const()[name = string("obj_371_strides_0"), val = tensor([1, 1])]; + tensor obj_371_pad_0 = const()[name = string("obj_371_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_371_dilations_0 = const()[name = string("obj_371_dilations_0"), val = tensor([1, 1])]; + int32 obj_371_groups_0 = const()[name = string("obj_371_groups_0"), val = int32(1)]; + tensor obj_371_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_371_dilations_0, groups = obj_371_groups_0, pad = obj_371_pad_0, pad_type = obj_371_pad_type_0, strides = obj_371_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_357_cast_fp16)[name = string("obj_371_cast_fp16")]; + tensor inputs_349_cast_fp16 = add(x = inputs_343_cast_fp16, y = obj_371_cast_fp16)[name = string("inputs_349_cast_fp16")]; + tensor inputs_sq_349_cast_fp16 = mul(x = inputs_349_cast_fp16, y = inputs_349_cast_fp16)[name = string("inputs_sq_349_cast_fp16")]; + tensor variance_349_axes_0 = const()[name = string("variance_349_axes_0"), val = tensor([1])]; + bool variance_349_keep_dims_0 = const()[name = string("variance_349_keep_dims_0"), val = bool(true)]; + tensor variance_349_cast_fp16 = reduce_mean(axes = variance_349_axes_0, keep_dims = variance_349_keep_dims_0, x = inputs_sq_349_cast_fp16)[name = string("variance_349_cast_fp16")]; + fp16 var_12822_to_fp16 = const()[name = string("op_12822_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12823_cast_fp16 = add(x = variance_349_cast_fp16, y = var_12822_to_fp16)[name = string("op_12823_cast_fp16")]; + fp32 var_12824_epsilon_0 = const()[name = string("op_12824_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12824_cast_fp16 = rsqrt(epsilon = var_12824_epsilon_0, x = var_12823_cast_fp16)[name = string("op_12824_cast_fp16")]; + tensor hidden_states_431_cast_fp16 = mul(x = inputs_349_cast_fp16, y = var_12824_cast_fp16)[name = string("hidden_states_431_cast_fp16")]; + tensor input_359_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_431_cast_fp16)[name = string("input_359_cast_fp16")]; + string input_361_pad_type_0 = const()[name = string("input_361_pad_type_0"), val = string("valid")]; + tensor input_361_strides_0 = const()[name = string("input_361_strides_0"), val = tensor([1, 1])]; + tensor input_361_pad_0 = const()[name = string("input_361_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_361_dilations_0 = const()[name = string("input_361_dilations_0"), val = tensor([1, 1])]; + int32 input_361_groups_0 = const()[name = string("input_361_groups_0"), val = int32(1)]; + tensor input_361_cast_fp16 = conv(dilations = input_361_dilations_0, groups = input_361_groups_0, pad = input_361_pad_0, pad_type = input_361_pad_type_0, strides = input_361_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_359_cast_fp16)[name = string("input_361_cast_fp16")]; + tensor var_12838_cast_fp16 = silu(x = input_361_cast_fp16)[name = string("op_12838_cast_fp16")]; + string var_12844_pad_type_0 = const()[name = string("op_12844_pad_type_0"), val = string("valid")]; + tensor var_12844_strides_0 = const()[name = string("op_12844_strides_0"), val = tensor([1, 1])]; + tensor var_12844_pad_0 = const()[name = string("op_12844_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_12844_dilations_0 = const()[name = string("op_12844_dilations_0"), val = tensor([1, 1])]; + int32 var_12844_groups_0 = const()[name = string("op_12844_groups_0"), val = int32(1)]; + tensor var_12844_cast_fp16 = conv(dilations = var_12844_dilations_0, groups = var_12844_groups_0, pad = var_12844_pad_0, pad_type = var_12844_pad_type_0, strides = var_12844_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_359_cast_fp16)[name = string("op_12844_cast_fp16")]; + tensor input_363_cast_fp16 = mul(x = var_12838_cast_fp16, y = var_12844_cast_fp16)[name = string("input_363_cast_fp16")]; + string hidden_states_433_pad_type_0 = const()[name = string("hidden_states_433_pad_type_0"), val = string("valid")]; + tensor hidden_states_433_strides_0 = const()[name = string("hidden_states_433_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_433_pad_0 = const()[name = string("hidden_states_433_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_433_dilations_0 = const()[name = string("hidden_states_433_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_433_groups_0 = const()[name = string("hidden_states_433_groups_0"), val = int32(1)]; + tensor hidden_states_433_cast_fp16 = conv(dilations = hidden_states_433_dilations_0, groups = hidden_states_433_groups_0, pad = hidden_states_433_pad_0, pad_type = hidden_states_433_pad_type_0, strides = hidden_states_433_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_363_cast_fp16)[name = string("hidden_states_433_cast_fp16")]; + tensor inputs_351_cast_fp16 = add(x = inputs_349_cast_fp16, y = hidden_states_433_cast_fp16)[name = string("inputs_351_cast_fp16")]; + tensor obj_375_begin_0 = const()[name = string("obj_375_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_375_end_0 = const()[name = string("obj_375_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_375_end_mask_0 = const()[name = string("obj_375_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_375_cast_fp16 = slice_by_index(begin = obj_375_begin_0, end = obj_375_end_0, end_mask = obj_375_end_mask_0, x = key_caches_17_cast_fp16)[name = string("obj_375_cast_fp16")]; + tensor obj_377_begin_0 = const()[name = string("obj_377_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_377_end_0 = const()[name = string("obj_377_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_377_end_mask_0 = const()[name = string("obj_377_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_377_cast_fp16 = slice_by_index(begin = obj_377_begin_0, end = obj_377_end_0, end_mask = obj_377_end_mask_0, x = value_caches_17_cast_fp16)[name = string("obj_377_cast_fp16")]; + int32 var_12892 = const()[name = string("op_12892"), val = int32(3)]; + int32 var_12902 = const()[name = string("op_12902"), val = int32(-2)]; + tensor inputs_sq_351_cast_fp16 = mul(x = inputs_351_cast_fp16, y = inputs_351_cast_fp16)[name = string("inputs_sq_351_cast_fp16")]; + tensor variance_351_axes_0 = const()[name = string("variance_351_axes_0"), val = tensor([1])]; + bool variance_351_keep_dims_0 = const()[name = string("variance_351_keep_dims_0"), val = bool(true)]; + tensor variance_351_cast_fp16 = reduce_mean(axes = variance_351_axes_0, keep_dims = variance_351_keep_dims_0, x = inputs_sq_351_cast_fp16)[name = string("variance_351_cast_fp16")]; + fp16 var_12916_to_fp16 = const()[name = string("op_12916_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12917_cast_fp16 = add(x = variance_351_cast_fp16, y = var_12916_to_fp16)[name = string("op_12917_cast_fp16")]; + fp32 var_12918_epsilon_0 = const()[name = string("op_12918_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12918_cast_fp16 = rsqrt(epsilon = var_12918_epsilon_0, x = var_12917_cast_fp16)[name = string("op_12918_cast_fp16")]; + tensor hidden_states_435_cast_fp16 = mul(x = inputs_351_cast_fp16, y = var_12918_cast_fp16)[name = string("hidden_states_435_cast_fp16")]; + tensor obj_373_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_435_cast_fp16)[name = string("obj_373_cast_fp16")]; + string query_253_pad_type_0 = const()[name = string("query_253_pad_type_0"), val = string("valid")]; + tensor query_253_strides_0 = const()[name = string("query_253_strides_0"), val = tensor([1, 1])]; + tensor query_253_pad_0 = const()[name = string("query_253_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_253_dilations_0 = const()[name = string("query_253_dilations_0"), val = tensor([1, 1])]; + int32 query_253_groups_0 = const()[name = string("query_253_groups_0"), val = int32(1)]; + tensor query_253_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_253_dilations_0, groups = query_253_groups_0, pad = query_253_pad_0, pad_type = query_253_pad_type_0, strides = query_253_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_373_cast_fp16)[name = string("query_253_cast_fp16")]; + string current_key_169_pad_type_0 = const()[name = string("current_key_169_pad_type_0"), val = string("valid")]; + tensor current_key_169_strides_0 = const()[name = string("current_key_169_strides_0"), val = tensor([1, 1])]; + tensor current_key_169_pad_0 = const()[name = string("current_key_169_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_169_dilations_0 = const()[name = string("current_key_169_dilations_0"), val = tensor([1, 1])]; + int32 current_key_169_groups_0 = const()[name = string("current_key_169_groups_0"), val = int32(1)]; + tensor current_key_169_cast_fp16 = conv(dilations = current_key_169_dilations_0, groups = current_key_169_groups_0, pad = current_key_169_pad_0, pad_type = current_key_169_pad_type_0, strides = current_key_169_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_373_cast_fp16)[name = string("current_key_169_cast_fp16")]; + string current_value_85_pad_type_0 = const()[name = string("current_value_85_pad_type_0"), val = string("valid")]; + tensor current_value_85_strides_0 = const()[name = string("current_value_85_strides_0"), val = tensor([1, 1])]; + tensor current_value_85_pad_0 = const()[name = string("current_value_85_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_85_dilations_0 = const()[name = string("current_value_85_dilations_0"), val = tensor([1, 1])]; + int32 current_value_85_groups_0 = const()[name = string("current_value_85_groups_0"), val = int32(1)]; + tensor current_value_85_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_85_dilations_0, groups = current_value_85_groups_0, pad = current_value_85_pad_0, pad_type = current_value_85_pad_type_0, strides = current_value_85_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_373_cast_fp16)[name = string("current_value_85_cast_fp16")]; + tensor var_12955 = const()[name = string("op_12955"), val = tensor([16, 128, 1, 1])]; + tensor inputs_353_cast_fp16 = reshape(shape = var_12955, x = query_253_cast_fp16)[name = string("inputs_353_cast_fp16")]; + tensor inputs_sq_353_cast_fp16 = mul(x = inputs_353_cast_fp16, y = inputs_353_cast_fp16)[name = string("inputs_sq_353_cast_fp16")]; + tensor variance_353_axes_0 = const()[name = string("variance_353_axes_0"), val = tensor([1])]; + bool variance_353_keep_dims_0 = const()[name = string("variance_353_keep_dims_0"), val = bool(true)]; + tensor variance_353_cast_fp16 = reduce_mean(axes = variance_353_axes_0, keep_dims = variance_353_keep_dims_0, x = inputs_sq_353_cast_fp16)[name = string("variance_353_cast_fp16")]; + fp16 var_12961_to_fp16 = const()[name = string("op_12961_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12962_cast_fp16 = add(x = variance_353_cast_fp16, y = var_12961_to_fp16)[name = string("op_12962_cast_fp16")]; + fp32 var_12963_epsilon_0 = const()[name = string("op_12963_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12963_cast_fp16 = rsqrt(epsilon = var_12963_epsilon_0, x = var_12962_cast_fp16)[name = string("op_12963_cast_fp16")]; + tensor hidden_states_437_cast_fp16 = mul(x = inputs_353_cast_fp16, y = var_12963_cast_fp16)[name = string("hidden_states_437_cast_fp16")]; + tensor query_normed_85_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_437_cast_fp16)[name = string("query_normed_85_cast_fp16")]; + tensor var_12971 = const()[name = string("op_12971"), val = tensor([8, 128, 1, 1])]; + tensor inputs_355_cast_fp16 = reshape(shape = var_12971, x = current_key_169_cast_fp16)[name = string("inputs_355_cast_fp16")]; + tensor inputs_sq_355_cast_fp16 = mul(x = inputs_355_cast_fp16, y = inputs_355_cast_fp16)[name = string("inputs_sq_355_cast_fp16")]; + tensor variance_355_axes_0 = const()[name = string("variance_355_axes_0"), val = tensor([1])]; + bool variance_355_keep_dims_0 = const()[name = string("variance_355_keep_dims_0"), val = bool(true)]; + tensor variance_355_cast_fp16 = reduce_mean(axes = variance_355_axes_0, keep_dims = variance_355_keep_dims_0, x = inputs_sq_355_cast_fp16)[name = string("variance_355_cast_fp16")]; + fp16 var_12977_to_fp16 = const()[name = string("op_12977_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_12978_cast_fp16 = add(x = variance_355_cast_fp16, y = var_12977_to_fp16)[name = string("op_12978_cast_fp16")]; + fp32 var_12979_epsilon_0 = const()[name = string("op_12979_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_12979_cast_fp16 = rsqrt(epsilon = var_12979_epsilon_0, x = var_12978_cast_fp16)[name = string("op_12979_cast_fp16")]; + tensor hidden_states_439_cast_fp16 = mul(x = inputs_355_cast_fp16, y = var_12979_cast_fp16)[name = string("hidden_states_439_cast_fp16")]; + tensor current_key_normed_85_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_439_cast_fp16)[name = string("current_key_normed_85_cast_fp16")]; + tensor var_12997 = const()[name = string("op_12997"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_337_cast_fp16 = reshape(shape = var_12997, x = query_normed_85_cast_fp16)[name = string("mh_q_337_cast_fp16")]; + tensor var_12999 = const()[name = string("op_12999"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_337_cast_fp16 = reshape(shape = var_12999, x = current_key_normed_85_cast_fp16)[name = string("mh_k_337_cast_fp16")]; + tensor var_13003_cast_fp16 = mul(x = mh_q_337_cast_fp16, y = cos_81_to_fp16)[name = string("op_13003_cast_fp16")]; + tensor var_13008_begin_0 = const()[name = string("op_13008_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13008_end_0 = const()[name = string("op_13008_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_13008_end_mask_0 = const()[name = string("op_13008_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_13008_cast_fp16 = slice_by_index(begin = var_13008_begin_0, end = var_13008_end_0, end_mask = var_13008_end_mask_0, x = mh_q_337_cast_fp16)[name = string("op_13008_cast_fp16")]; + tensor var_13014_begin_0 = const()[name = string("op_13014_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_13014_end_0 = const()[name = string("op_13014_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_13014_end_mask_0 = const()[name = string("op_13014_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_13014_cast_fp16 = slice_by_index(begin = var_13014_begin_0, end = var_13014_end_0, end_mask = var_13014_end_mask_0, x = mh_q_337_cast_fp16)[name = string("op_13014_cast_fp16")]; + fp16 const_862_promoted_to_fp16 = const()[name = string("const_862_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13016_cast_fp16 = mul(x = var_13014_cast_fp16, y = const_862_promoted_to_fp16)[name = string("op_13016_cast_fp16")]; + bool var_13018_interleave_0 = const()[name = string("op_13018_interleave_0"), val = bool(false)]; + tensor var_13018_cast_fp16 = concat(axis = var_12902, interleave = var_13018_interleave_0, values = (var_13016_cast_fp16, var_13008_cast_fp16))[name = string("op_13018_cast_fp16")]; + tensor var_13019_cast_fp16 = mul(x = var_13018_cast_fp16, y = sin_81_to_fp16)[name = string("op_13019_cast_fp16")]; + tensor mh_q_339_cast_fp16 = add(x = var_13003_cast_fp16, y = var_13019_cast_fp16)[name = string("mh_q_339_cast_fp16")]; + tensor var_13021_cast_fp16 = mul(x = mh_k_337_cast_fp16, y = cos_81_to_fp16)[name = string("op_13021_cast_fp16")]; + tensor var_13026_begin_0 = const()[name = string("op_13026_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13026_end_0 = const()[name = string("op_13026_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_13026_end_mask_0 = const()[name = string("op_13026_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_13026_cast_fp16 = slice_by_index(begin = var_13026_begin_0, end = var_13026_end_0, end_mask = var_13026_end_mask_0, x = mh_k_337_cast_fp16)[name = string("op_13026_cast_fp16")]; + tensor var_13032_begin_0 = const()[name = string("op_13032_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_13032_end_0 = const()[name = string("op_13032_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_13032_end_mask_0 = const()[name = string("op_13032_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_13032_cast_fp16 = slice_by_index(begin = var_13032_begin_0, end = var_13032_end_0, end_mask = var_13032_end_mask_0, x = mh_k_337_cast_fp16)[name = string("op_13032_cast_fp16")]; + fp16 const_865_promoted_to_fp16 = const()[name = string("const_865_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13034_cast_fp16 = mul(x = var_13032_cast_fp16, y = const_865_promoted_to_fp16)[name = string("op_13034_cast_fp16")]; + bool var_13036_interleave_0 = const()[name = string("op_13036_interleave_0"), val = bool(false)]; + tensor var_13036_cast_fp16 = concat(axis = var_12902, interleave = var_13036_interleave_0, values = (var_13034_cast_fp16, var_13026_cast_fp16))[name = string("op_13036_cast_fp16")]; + tensor var_13037_cast_fp16 = mul(x = var_13036_cast_fp16, y = sin_81_to_fp16)[name = string("op_13037_cast_fp16")]; + tensor mh_k_339_cast_fp16 = add(x = var_13021_cast_fp16, y = var_13037_cast_fp16)[name = string("mh_k_339_cast_fp16")]; + tensor var_13041 = const()[name = string("op_13041"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_171_cast_fp16 = reshape(shape = var_13041, x = mh_k_339_cast_fp16)[name = string("current_key_171_cast_fp16")]; + tensor var_13047_to_fp16 = const()[name = string("op_13047_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197376)))]; + tensor var_13048_cast_fp16 = mul(x = obj_375_cast_fp16, y = var_13047_to_fp16)[name = string("op_13048_cast_fp16")]; + tensor var_13045_to_fp16 = const()[name = string("op_13045_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197504)))]; + tensor var_13049_cast_fp16 = mul(x = current_key_171_cast_fp16, y = var_13045_to_fp16)[name = string("op_13049_cast_fp16")]; + tensor key_171_cast_fp16 = add(x = var_13048_cast_fp16, y = var_13049_cast_fp16)[name = string("key_171_cast_fp16")]; + tensor var_13051_to_fp16 = const()[name = string("op_13051_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197376)))]; + tensor var_13052_cast_fp16 = mul(x = obj_377_cast_fp16, y = var_13051_to_fp16)[name = string("op_13052_cast_fp16")]; + tensor var_13053_cast_fp16 = mul(x = current_value_85_cast_fp16, y = var_13045_to_fp16)[name = string("op_13053_cast_fp16")]; + tensor value_85_cast_fp16 = add(x = var_13052_cast_fp16, y = var_13053_cast_fp16)[name = string("value_85_cast_fp16")]; + fp16 var_13060_to_fp16 = const()[name = string("op_13060_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_343_cast_fp16 = mul(x = mh_q_339_cast_fp16, y = var_13060_to_fp16)[name = string("mh_q_343_cast_fp16")]; + tensor var_13062 = const()[name = string("op_13062"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_341_cast_fp16 = reshape(shape = var_13062, x = key_171_cast_fp16)[name = string("mh_k_341_cast_fp16")]; + tensor var_13064 = const()[name = string("op_13064"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_169_cast_fp16 = reshape(shape = var_13064, x = value_85_cast_fp16)[name = string("mh_v_169_cast_fp16")]; + tensor transpose_168_perm_0 = const()[name = string("transpose_168_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_84_reps_0 = const()[name = string("tile_84_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_168_cast_fp16 = transpose(perm = transpose_168_perm_0, x = mh_k_341_cast_fp16)[name = string("transpose_227")]; + tensor tile_84_cast_fp16 = tile(reps = tile_84_reps_0, x = transpose_168_cast_fp16)[name = string("tile_84_cast_fp16")]; + tensor concat_211 = const()[name = string("concat_211"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_168_cast_fp16 = reshape(shape = concat_211, x = tile_84_cast_fp16)[name = string("reshape_168_cast_fp16")]; + tensor transpose_169_perm_0 = const()[name = string("transpose_169_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_212 = const()[name = string("concat_212"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_169_cast_fp16 = transpose(perm = transpose_169_perm_0, x = reshape_168_cast_fp16)[name = string("transpose_226")]; + tensor reshape_169_cast_fp16 = reshape(shape = concat_212, x = transpose_169_cast_fp16)[name = string("reshape_169_cast_fp16")]; + tensor transpose_170_perm_0 = const()[name = string("transpose_170_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_85_reps_0 = const()[name = string("tile_85_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_170_cast_fp16 = transpose(perm = transpose_170_perm_0, x = mh_v_169_cast_fp16)[name = string("transpose_225")]; + tensor tile_85_cast_fp16 = tile(reps = tile_85_reps_0, x = transpose_170_cast_fp16)[name = string("tile_85_cast_fp16")]; + tensor concat_213 = const()[name = string("concat_213"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_170_cast_fp16 = reshape(shape = concat_213, x = tile_85_cast_fp16)[name = string("reshape_170_cast_fp16")]; + tensor transpose_171_perm_0 = const()[name = string("transpose_171_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_214 = const()[name = string("concat_214"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_171_cast_fp16 = transpose(perm = transpose_171_perm_0, x = reshape_170_cast_fp16)[name = string("transpose_224")]; + tensor reshape_171_cast_fp16 = reshape(shape = concat_214, x = transpose_171_cast_fp16)[name = string("reshape_171_cast_fp16")]; + tensor transpose_485_perm_0 = const()[name = string("transpose_485_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_253_transpose_x_1 = const()[name = string("mh_w_253_transpose_x_1"), val = bool(true)]; + bool mh_w_253_transpose_y_1 = const()[name = string("mh_w_253_transpose_y_1"), val = bool(false)]; + tensor transpose_485_cast_fp16 = transpose(perm = transpose_485_perm_0, x = reshape_169_cast_fp16)[name = string("transpose_223")]; + tensor mh_w_253_cast_fp16 = matmul(transpose_x = mh_w_253_transpose_x_1, transpose_y = mh_w_253_transpose_y_1, x = mh_q_343_cast_fp16, y = transpose_485_cast_fp16)[name = string("mh_w_253_cast_fp16")]; + tensor var_13072_to_fp16 = const()[name = string("op_13072_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197632)))]; + tensor mh_w_255_cast_fp16 = add(x = mh_w_253_cast_fp16, y = var_13072_to_fp16)[name = string("mh_w_255_cast_fp16")]; + tensor mh_w_257_cast_fp16 = softmax(axis = var_12892, x = mh_w_255_cast_fp16)[name = string("mh_w_257_cast_fp16")]; + tensor transpose_486_perm_0 = const()[name = string("transpose_486_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_85_transpose_x_1 = const()[name = string("attn_85_transpose_x_1"), val = bool(false)]; + bool attn_85_transpose_y_1 = const()[name = string("attn_85_transpose_y_1"), val = bool(true)]; + tensor transpose_486_cast_fp16 = transpose(perm = transpose_486_perm_0, x = reshape_171_cast_fp16)[name = string("transpose_222")]; + tensor attn_85_cast_fp16 = matmul(transpose_x = attn_85_transpose_x_1, transpose_y = attn_85_transpose_y_1, x = transpose_486_cast_fp16, y = mh_w_257_cast_fp16)[name = string("attn_85_cast_fp16")]; + tensor var_13078 = const()[name = string("op_13078"), val = tensor([1, 2048, 1, 1])]; + tensor input_365_cast_fp16 = reshape(shape = var_13078, x = attn_85_cast_fp16)[name = string("input_365_cast_fp16")]; + string obj_379_pad_type_0 = const()[name = string("obj_379_pad_type_0"), val = string("valid")]; + tensor obj_379_strides_0 = const()[name = string("obj_379_strides_0"), val = tensor([1, 1])]; + tensor obj_379_pad_0 = const()[name = string("obj_379_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_379_dilations_0 = const()[name = string("obj_379_dilations_0"), val = tensor([1, 1])]; + int32 obj_379_groups_0 = const()[name = string("obj_379_groups_0"), val = int32(1)]; + tensor obj_379_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_379_dilations_0, groups = obj_379_groups_0, pad = obj_379_pad_0, pad_type = obj_379_pad_type_0, strides = obj_379_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_365_cast_fp16)[name = string("obj_379_cast_fp16")]; + tensor inputs_357_cast_fp16 = add(x = inputs_351_cast_fp16, y = obj_379_cast_fp16)[name = string("inputs_357_cast_fp16")]; + tensor inputs_sq_357_cast_fp16 = mul(x = inputs_357_cast_fp16, y = inputs_357_cast_fp16)[name = string("inputs_sq_357_cast_fp16")]; + tensor variance_357_axes_0 = const()[name = string("variance_357_axes_0"), val = tensor([1])]; + bool variance_357_keep_dims_0 = const()[name = string("variance_357_keep_dims_0"), val = bool(true)]; + tensor variance_357_cast_fp16 = reduce_mean(axes = variance_357_axes_0, keep_dims = variance_357_keep_dims_0, x = inputs_sq_357_cast_fp16)[name = string("variance_357_cast_fp16")]; + fp16 var_13096_to_fp16 = const()[name = string("op_13096_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13097_cast_fp16 = add(x = variance_357_cast_fp16, y = var_13096_to_fp16)[name = string("op_13097_cast_fp16")]; + fp32 var_13098_epsilon_0 = const()[name = string("op_13098_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13098_cast_fp16 = rsqrt(epsilon = var_13098_epsilon_0, x = var_13097_cast_fp16)[name = string("op_13098_cast_fp16")]; + tensor hidden_states_441_cast_fp16 = mul(x = inputs_357_cast_fp16, y = var_13098_cast_fp16)[name = string("hidden_states_441_cast_fp16")]; + tensor input_367_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_441_cast_fp16)[name = string("input_367_cast_fp16")]; + string input_369_pad_type_0 = const()[name = string("input_369_pad_type_0"), val = string("valid")]; + tensor input_369_strides_0 = const()[name = string("input_369_strides_0"), val = tensor([1, 1])]; + tensor input_369_pad_0 = const()[name = string("input_369_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_369_dilations_0 = const()[name = string("input_369_dilations_0"), val = tensor([1, 1])]; + int32 input_369_groups_0 = const()[name = string("input_369_groups_0"), val = int32(1)]; + tensor input_369_cast_fp16 = conv(dilations = input_369_dilations_0, groups = input_369_groups_0, pad = input_369_pad_0, pad_type = input_369_pad_type_0, strides = input_369_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_367_cast_fp16)[name = string("input_369_cast_fp16")]; + tensor var_13112_cast_fp16 = silu(x = input_369_cast_fp16)[name = string("op_13112_cast_fp16")]; + string var_13118_pad_type_0 = const()[name = string("op_13118_pad_type_0"), val = string("valid")]; + tensor var_13118_strides_0 = const()[name = string("op_13118_strides_0"), val = tensor([1, 1])]; + tensor var_13118_pad_0 = const()[name = string("op_13118_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13118_dilations_0 = const()[name = string("op_13118_dilations_0"), val = tensor([1, 1])]; + int32 var_13118_groups_0 = const()[name = string("op_13118_groups_0"), val = int32(1)]; + tensor var_13118_cast_fp16 = conv(dilations = var_13118_dilations_0, groups = var_13118_groups_0, pad = var_13118_pad_0, pad_type = var_13118_pad_type_0, strides = var_13118_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_367_cast_fp16)[name = string("op_13118_cast_fp16")]; + tensor input_371_cast_fp16 = mul(x = var_13112_cast_fp16, y = var_13118_cast_fp16)[name = string("input_371_cast_fp16")]; + string hidden_states_443_pad_type_0 = const()[name = string("hidden_states_443_pad_type_0"), val = string("valid")]; + tensor hidden_states_443_strides_0 = const()[name = string("hidden_states_443_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_443_pad_0 = const()[name = string("hidden_states_443_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_443_dilations_0 = const()[name = string("hidden_states_443_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_443_groups_0 = const()[name = string("hidden_states_443_groups_0"), val = int32(1)]; + tensor hidden_states_443_cast_fp16 = conv(dilations = hidden_states_443_dilations_0, groups = hidden_states_443_groups_0, pad = hidden_states_443_pad_0, pad_type = hidden_states_443_pad_type_0, strides = hidden_states_443_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_371_cast_fp16)[name = string("hidden_states_443_cast_fp16")]; + tensor inputs_359_cast_fp16 = add(x = inputs_357_cast_fp16, y = hidden_states_443_cast_fp16)[name = string("inputs_359_cast_fp16")]; + tensor obj_383_begin_0 = const()[name = string("obj_383_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_383_end_0 = const()[name = string("obj_383_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_383_end_mask_0 = const()[name = string("obj_383_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_383_cast_fp16 = slice_by_index(begin = obj_383_begin_0, end = obj_383_end_0, end_mask = obj_383_end_mask_0, x = key_caches_17_cast_fp16)[name = string("obj_383_cast_fp16")]; + tensor obj_385_begin_0 = const()[name = string("obj_385_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_385_end_0 = const()[name = string("obj_385_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_385_end_mask_0 = const()[name = string("obj_385_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_385_cast_fp16 = slice_by_index(begin = obj_385_begin_0, end = obj_385_end_0, end_mask = obj_385_end_mask_0, x = value_caches_17_cast_fp16)[name = string("obj_385_cast_fp16")]; + int32 var_13166 = const()[name = string("op_13166"), val = int32(3)]; + int32 var_13176 = const()[name = string("op_13176"), val = int32(-2)]; + tensor inputs_sq_359_cast_fp16 = mul(x = inputs_359_cast_fp16, y = inputs_359_cast_fp16)[name = string("inputs_sq_359_cast_fp16")]; + tensor variance_359_axes_0 = const()[name = string("variance_359_axes_0"), val = tensor([1])]; + bool variance_359_keep_dims_0 = const()[name = string("variance_359_keep_dims_0"), val = bool(true)]; + tensor variance_359_cast_fp16 = reduce_mean(axes = variance_359_axes_0, keep_dims = variance_359_keep_dims_0, x = inputs_sq_359_cast_fp16)[name = string("variance_359_cast_fp16")]; + fp16 var_13190_to_fp16 = const()[name = string("op_13190_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13191_cast_fp16 = add(x = variance_359_cast_fp16, y = var_13190_to_fp16)[name = string("op_13191_cast_fp16")]; + fp32 var_13192_epsilon_0 = const()[name = string("op_13192_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13192_cast_fp16 = rsqrt(epsilon = var_13192_epsilon_0, x = var_13191_cast_fp16)[name = string("op_13192_cast_fp16")]; + tensor hidden_states_445_cast_fp16 = mul(x = inputs_359_cast_fp16, y = var_13192_cast_fp16)[name = string("hidden_states_445_cast_fp16")]; + tensor obj_381_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_445_cast_fp16)[name = string("obj_381_cast_fp16")]; + string query_259_pad_type_0 = const()[name = string("query_259_pad_type_0"), val = string("valid")]; + tensor query_259_strides_0 = const()[name = string("query_259_strides_0"), val = tensor([1, 1])]; + tensor query_259_pad_0 = const()[name = string("query_259_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_259_dilations_0 = const()[name = string("query_259_dilations_0"), val = tensor([1, 1])]; + int32 query_259_groups_0 = const()[name = string("query_259_groups_0"), val = int32(1)]; + tensor query_259_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_259_dilations_0, groups = query_259_groups_0, pad = query_259_pad_0, pad_type = query_259_pad_type_0, strides = query_259_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_381_cast_fp16)[name = string("query_259_cast_fp16")]; + string current_key_173_pad_type_0 = const()[name = string("current_key_173_pad_type_0"), val = string("valid")]; + tensor current_key_173_strides_0 = const()[name = string("current_key_173_strides_0"), val = tensor([1, 1])]; + tensor current_key_173_pad_0 = const()[name = string("current_key_173_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_173_dilations_0 = const()[name = string("current_key_173_dilations_0"), val = tensor([1, 1])]; + int32 current_key_173_groups_0 = const()[name = string("current_key_173_groups_0"), val = int32(1)]; + tensor current_key_173_cast_fp16 = conv(dilations = current_key_173_dilations_0, groups = current_key_173_groups_0, pad = current_key_173_pad_0, pad_type = current_key_173_pad_type_0, strides = current_key_173_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_381_cast_fp16)[name = string("current_key_173_cast_fp16")]; + string current_value_87_pad_type_0 = const()[name = string("current_value_87_pad_type_0"), val = string("valid")]; + tensor current_value_87_strides_0 = const()[name = string("current_value_87_strides_0"), val = tensor([1, 1])]; + tensor current_value_87_pad_0 = const()[name = string("current_value_87_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_87_dilations_0 = const()[name = string("current_value_87_dilations_0"), val = tensor([1, 1])]; + int32 current_value_87_groups_0 = const()[name = string("current_value_87_groups_0"), val = int32(1)]; + tensor current_value_87_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_87_dilations_0, groups = current_value_87_groups_0, pad = current_value_87_pad_0, pad_type = current_value_87_pad_type_0, strides = current_value_87_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_381_cast_fp16)[name = string("current_value_87_cast_fp16")]; + tensor var_13229 = const()[name = string("op_13229"), val = tensor([16, 128, 1, 1])]; + tensor inputs_361_cast_fp16 = reshape(shape = var_13229, x = query_259_cast_fp16)[name = string("inputs_361_cast_fp16")]; + tensor inputs_sq_361_cast_fp16 = mul(x = inputs_361_cast_fp16, y = inputs_361_cast_fp16)[name = string("inputs_sq_361_cast_fp16")]; + tensor variance_361_axes_0 = const()[name = string("variance_361_axes_0"), val = tensor([1])]; + bool variance_361_keep_dims_0 = const()[name = string("variance_361_keep_dims_0"), val = bool(true)]; + tensor variance_361_cast_fp16 = reduce_mean(axes = variance_361_axes_0, keep_dims = variance_361_keep_dims_0, x = inputs_sq_361_cast_fp16)[name = string("variance_361_cast_fp16")]; + fp16 var_13235_to_fp16 = const()[name = string("op_13235_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13236_cast_fp16 = add(x = variance_361_cast_fp16, y = var_13235_to_fp16)[name = string("op_13236_cast_fp16")]; + fp32 var_13237_epsilon_0 = const()[name = string("op_13237_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13237_cast_fp16 = rsqrt(epsilon = var_13237_epsilon_0, x = var_13236_cast_fp16)[name = string("op_13237_cast_fp16")]; + tensor hidden_states_447_cast_fp16 = mul(x = inputs_361_cast_fp16, y = var_13237_cast_fp16)[name = string("hidden_states_447_cast_fp16")]; + tensor query_normed_87_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_447_cast_fp16)[name = string("query_normed_87_cast_fp16")]; + tensor var_13245 = const()[name = string("op_13245"), val = tensor([8, 128, 1, 1])]; + tensor inputs_363_cast_fp16 = reshape(shape = var_13245, x = current_key_173_cast_fp16)[name = string("inputs_363_cast_fp16")]; + tensor inputs_sq_363_cast_fp16 = mul(x = inputs_363_cast_fp16, y = inputs_363_cast_fp16)[name = string("inputs_sq_363_cast_fp16")]; + tensor variance_363_axes_0 = const()[name = string("variance_363_axes_0"), val = tensor([1])]; + bool variance_363_keep_dims_0 = const()[name = string("variance_363_keep_dims_0"), val = bool(true)]; + tensor variance_363_cast_fp16 = reduce_mean(axes = variance_363_axes_0, keep_dims = variance_363_keep_dims_0, x = inputs_sq_363_cast_fp16)[name = string("variance_363_cast_fp16")]; + fp16 var_13251_to_fp16 = const()[name = string("op_13251_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13252_cast_fp16 = add(x = variance_363_cast_fp16, y = var_13251_to_fp16)[name = string("op_13252_cast_fp16")]; + fp32 var_13253_epsilon_0 = const()[name = string("op_13253_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13253_cast_fp16 = rsqrt(epsilon = var_13253_epsilon_0, x = var_13252_cast_fp16)[name = string("op_13253_cast_fp16")]; + tensor hidden_states_449_cast_fp16 = mul(x = inputs_363_cast_fp16, y = var_13253_cast_fp16)[name = string("hidden_states_449_cast_fp16")]; + tensor current_key_normed_87_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_449_cast_fp16)[name = string("current_key_normed_87_cast_fp16")]; + tensor var_13271 = const()[name = string("op_13271"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_345_cast_fp16 = reshape(shape = var_13271, x = query_normed_87_cast_fp16)[name = string("mh_q_345_cast_fp16")]; + tensor var_13273 = const()[name = string("op_13273"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_345_cast_fp16 = reshape(shape = var_13273, x = current_key_normed_87_cast_fp16)[name = string("mh_k_345_cast_fp16")]; + tensor var_13277_cast_fp16 = mul(x = mh_q_345_cast_fp16, y = cos_81_to_fp16)[name = string("op_13277_cast_fp16")]; + tensor var_13282_begin_0 = const()[name = string("op_13282_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13282_end_0 = const()[name = string("op_13282_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_13282_end_mask_0 = const()[name = string("op_13282_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_13282_cast_fp16 = slice_by_index(begin = var_13282_begin_0, end = var_13282_end_0, end_mask = var_13282_end_mask_0, x = mh_q_345_cast_fp16)[name = string("op_13282_cast_fp16")]; + tensor var_13288_begin_0 = const()[name = string("op_13288_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_13288_end_0 = const()[name = string("op_13288_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_13288_end_mask_0 = const()[name = string("op_13288_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_13288_cast_fp16 = slice_by_index(begin = var_13288_begin_0, end = var_13288_end_0, end_mask = var_13288_end_mask_0, x = mh_q_345_cast_fp16)[name = string("op_13288_cast_fp16")]; + fp16 const_882_promoted_to_fp16 = const()[name = string("const_882_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13290_cast_fp16 = mul(x = var_13288_cast_fp16, y = const_882_promoted_to_fp16)[name = string("op_13290_cast_fp16")]; + bool var_13292_interleave_0 = const()[name = string("op_13292_interleave_0"), val = bool(false)]; + tensor var_13292_cast_fp16 = concat(axis = var_13176, interleave = var_13292_interleave_0, values = (var_13290_cast_fp16, var_13282_cast_fp16))[name = string("op_13292_cast_fp16")]; + tensor var_13293_cast_fp16 = mul(x = var_13292_cast_fp16, y = sin_81_to_fp16)[name = string("op_13293_cast_fp16")]; + tensor mh_q_347_cast_fp16 = add(x = var_13277_cast_fp16, y = var_13293_cast_fp16)[name = string("mh_q_347_cast_fp16")]; + tensor var_13295_cast_fp16 = mul(x = mh_k_345_cast_fp16, y = cos_81_to_fp16)[name = string("op_13295_cast_fp16")]; + tensor var_13300_begin_0 = const()[name = string("op_13300_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13300_end_0 = const()[name = string("op_13300_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_13300_end_mask_0 = const()[name = string("op_13300_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_13300_cast_fp16 = slice_by_index(begin = var_13300_begin_0, end = var_13300_end_0, end_mask = var_13300_end_mask_0, x = mh_k_345_cast_fp16)[name = string("op_13300_cast_fp16")]; + tensor var_13306_begin_0 = const()[name = string("op_13306_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_13306_end_0 = const()[name = string("op_13306_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_13306_end_mask_0 = const()[name = string("op_13306_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_13306_cast_fp16 = slice_by_index(begin = var_13306_begin_0, end = var_13306_end_0, end_mask = var_13306_end_mask_0, x = mh_k_345_cast_fp16)[name = string("op_13306_cast_fp16")]; + fp16 const_885_promoted_to_fp16 = const()[name = string("const_885_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13308_cast_fp16 = mul(x = var_13306_cast_fp16, y = const_885_promoted_to_fp16)[name = string("op_13308_cast_fp16")]; + bool var_13310_interleave_0 = const()[name = string("op_13310_interleave_0"), val = bool(false)]; + tensor var_13310_cast_fp16 = concat(axis = var_13176, interleave = var_13310_interleave_0, values = (var_13308_cast_fp16, var_13300_cast_fp16))[name = string("op_13310_cast_fp16")]; + tensor var_13311_cast_fp16 = mul(x = var_13310_cast_fp16, y = sin_81_to_fp16)[name = string("op_13311_cast_fp16")]; + tensor mh_k_347_cast_fp16 = add(x = var_13295_cast_fp16, y = var_13311_cast_fp16)[name = string("mh_k_347_cast_fp16")]; + tensor var_13315 = const()[name = string("op_13315"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_175_cast_fp16 = reshape(shape = var_13315, x = mh_k_347_cast_fp16)[name = string("current_key_175_cast_fp16")]; + tensor var_13321_to_fp16 = const()[name = string("op_13321_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197376)))]; + tensor var_13322_cast_fp16 = mul(x = obj_383_cast_fp16, y = var_13321_to_fp16)[name = string("op_13322_cast_fp16")]; + tensor var_13319_to_fp16 = const()[name = string("op_13319_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197504)))]; + tensor var_13323_cast_fp16 = mul(x = current_key_175_cast_fp16, y = var_13319_to_fp16)[name = string("op_13323_cast_fp16")]; + tensor key_175_cast_fp16 = add(x = var_13322_cast_fp16, y = var_13323_cast_fp16)[name = string("key_175_cast_fp16")]; + tensor var_13325_to_fp16 = const()[name = string("op_13325_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197376)))]; + tensor var_13326_cast_fp16 = mul(x = obj_385_cast_fp16, y = var_13325_to_fp16)[name = string("op_13326_cast_fp16")]; + tensor var_13327_cast_fp16 = mul(x = current_value_87_cast_fp16, y = var_13319_to_fp16)[name = string("op_13327_cast_fp16")]; + tensor value_87_cast_fp16 = add(x = var_13326_cast_fp16, y = var_13327_cast_fp16)[name = string("value_87_cast_fp16")]; + fp16 var_13334_to_fp16 = const()[name = string("op_13334_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_351_cast_fp16 = mul(x = mh_q_347_cast_fp16, y = var_13334_to_fp16)[name = string("mh_q_351_cast_fp16")]; + tensor var_13336 = const()[name = string("op_13336"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_349_cast_fp16 = reshape(shape = var_13336, x = key_175_cast_fp16)[name = string("mh_k_349_cast_fp16")]; + tensor var_13338 = const()[name = string("op_13338"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_173_cast_fp16 = reshape(shape = var_13338, x = value_87_cast_fp16)[name = string("mh_v_173_cast_fp16")]; + tensor transpose_172_perm_0 = const()[name = string("transpose_172_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_86_reps_0 = const()[name = string("tile_86_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_172_cast_fp16 = transpose(perm = transpose_172_perm_0, x = mh_k_349_cast_fp16)[name = string("transpose_221")]; + tensor tile_86_cast_fp16 = tile(reps = tile_86_reps_0, x = transpose_172_cast_fp16)[name = string("tile_86_cast_fp16")]; + tensor concat_215 = const()[name = string("concat_215"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_172_cast_fp16 = reshape(shape = concat_215, x = tile_86_cast_fp16)[name = string("reshape_172_cast_fp16")]; + tensor transpose_173_perm_0 = const()[name = string("transpose_173_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_216 = const()[name = string("concat_216"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_173_cast_fp16 = transpose(perm = transpose_173_perm_0, x = reshape_172_cast_fp16)[name = string("transpose_220")]; + tensor reshape_173_cast_fp16 = reshape(shape = concat_216, x = transpose_173_cast_fp16)[name = string("reshape_173_cast_fp16")]; + tensor transpose_174_perm_0 = const()[name = string("transpose_174_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_87_reps_0 = const()[name = string("tile_87_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_174_cast_fp16 = transpose(perm = transpose_174_perm_0, x = mh_v_173_cast_fp16)[name = string("transpose_219")]; + tensor tile_87_cast_fp16 = tile(reps = tile_87_reps_0, x = transpose_174_cast_fp16)[name = string("tile_87_cast_fp16")]; + tensor concat_217 = const()[name = string("concat_217"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_174_cast_fp16 = reshape(shape = concat_217, x = tile_87_cast_fp16)[name = string("reshape_174_cast_fp16")]; + tensor transpose_175_perm_0 = const()[name = string("transpose_175_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_218 = const()[name = string("concat_218"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_175_cast_fp16 = transpose(perm = transpose_175_perm_0, x = reshape_174_cast_fp16)[name = string("transpose_218")]; + tensor reshape_175_cast_fp16 = reshape(shape = concat_218, x = transpose_175_cast_fp16)[name = string("reshape_175_cast_fp16")]; + tensor transpose_489_perm_0 = const()[name = string("transpose_489_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_259_transpose_x_1 = const()[name = string("mh_w_259_transpose_x_1"), val = bool(true)]; + bool mh_w_259_transpose_y_1 = const()[name = string("mh_w_259_transpose_y_1"), val = bool(false)]; + tensor transpose_489_cast_fp16 = transpose(perm = transpose_489_perm_0, x = reshape_173_cast_fp16)[name = string("transpose_217")]; + tensor mh_w_259_cast_fp16 = matmul(transpose_x = mh_w_259_transpose_x_1, transpose_y = mh_w_259_transpose_y_1, x = mh_q_351_cast_fp16, y = transpose_489_cast_fp16)[name = string("mh_w_259_cast_fp16")]; + tensor var_13346_to_fp16 = const()[name = string("op_13346_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197632)))]; + tensor mh_w_261_cast_fp16 = add(x = mh_w_259_cast_fp16, y = var_13346_to_fp16)[name = string("mh_w_261_cast_fp16")]; + tensor mh_w_263_cast_fp16 = softmax(axis = var_13166, x = mh_w_261_cast_fp16)[name = string("mh_w_263_cast_fp16")]; + tensor transpose_490_perm_0 = const()[name = string("transpose_490_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_87_transpose_x_1 = const()[name = string("attn_87_transpose_x_1"), val = bool(false)]; + bool attn_87_transpose_y_1 = const()[name = string("attn_87_transpose_y_1"), val = bool(true)]; + tensor transpose_490_cast_fp16 = transpose(perm = transpose_490_perm_0, x = reshape_175_cast_fp16)[name = string("transpose_216")]; + tensor attn_87_cast_fp16 = matmul(transpose_x = attn_87_transpose_x_1, transpose_y = attn_87_transpose_y_1, x = transpose_490_cast_fp16, y = mh_w_263_cast_fp16)[name = string("attn_87_cast_fp16")]; + tensor var_13352 = const()[name = string("op_13352"), val = tensor([1, 2048, 1, 1])]; + tensor input_373_cast_fp16 = reshape(shape = var_13352, x = attn_87_cast_fp16)[name = string("input_373_cast_fp16")]; + string obj_387_pad_type_0 = const()[name = string("obj_387_pad_type_0"), val = string("valid")]; + tensor obj_387_strides_0 = const()[name = string("obj_387_strides_0"), val = tensor([1, 1])]; + tensor obj_387_pad_0 = const()[name = string("obj_387_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_387_dilations_0 = const()[name = string("obj_387_dilations_0"), val = tensor([1, 1])]; + int32 obj_387_groups_0 = const()[name = string("obj_387_groups_0"), val = int32(1)]; + tensor obj_387_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_387_dilations_0, groups = obj_387_groups_0, pad = obj_387_pad_0, pad_type = obj_387_pad_type_0, strides = obj_387_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_373_cast_fp16)[name = string("obj_387_cast_fp16")]; + tensor inputs_365_cast_fp16 = add(x = inputs_359_cast_fp16, y = obj_387_cast_fp16)[name = string("inputs_365_cast_fp16")]; + tensor inputs_sq_365_cast_fp16 = mul(x = inputs_365_cast_fp16, y = inputs_365_cast_fp16)[name = string("inputs_sq_365_cast_fp16")]; + tensor variance_365_axes_0 = const()[name = string("variance_365_axes_0"), val = tensor([1])]; + bool variance_365_keep_dims_0 = const()[name = string("variance_365_keep_dims_0"), val = bool(true)]; + tensor variance_365_cast_fp16 = reduce_mean(axes = variance_365_axes_0, keep_dims = variance_365_keep_dims_0, x = inputs_sq_365_cast_fp16)[name = string("variance_365_cast_fp16")]; + fp16 var_13370_to_fp16 = const()[name = string("op_13370_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13371_cast_fp16 = add(x = variance_365_cast_fp16, y = var_13370_to_fp16)[name = string("op_13371_cast_fp16")]; + fp32 var_13372_epsilon_0 = const()[name = string("op_13372_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13372_cast_fp16 = rsqrt(epsilon = var_13372_epsilon_0, x = var_13371_cast_fp16)[name = string("op_13372_cast_fp16")]; + tensor hidden_states_451_cast_fp16 = mul(x = inputs_365_cast_fp16, y = var_13372_cast_fp16)[name = string("hidden_states_451_cast_fp16")]; + tensor input_375_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_451_cast_fp16)[name = string("input_375_cast_fp16")]; + string input_377_pad_type_0 = const()[name = string("input_377_pad_type_0"), val = string("valid")]; + tensor input_377_strides_0 = const()[name = string("input_377_strides_0"), val = tensor([1, 1])]; + tensor input_377_pad_0 = const()[name = string("input_377_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_377_dilations_0 = const()[name = string("input_377_dilations_0"), val = tensor([1, 1])]; + int32 input_377_groups_0 = const()[name = string("input_377_groups_0"), val = int32(1)]; + tensor input_377_cast_fp16 = conv(dilations = input_377_dilations_0, groups = input_377_groups_0, pad = input_377_pad_0, pad_type = input_377_pad_type_0, strides = input_377_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_375_cast_fp16)[name = string("input_377_cast_fp16")]; + tensor var_13386_cast_fp16 = silu(x = input_377_cast_fp16)[name = string("op_13386_cast_fp16")]; + string var_13392_pad_type_0 = const()[name = string("op_13392_pad_type_0"), val = string("valid")]; + tensor var_13392_strides_0 = const()[name = string("op_13392_strides_0"), val = tensor([1, 1])]; + tensor var_13392_pad_0 = const()[name = string("op_13392_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13392_dilations_0 = const()[name = string("op_13392_dilations_0"), val = tensor([1, 1])]; + int32 var_13392_groups_0 = const()[name = string("op_13392_groups_0"), val = int32(1)]; + tensor var_13392_cast_fp16 = conv(dilations = var_13392_dilations_0, groups = var_13392_groups_0, pad = var_13392_pad_0, pad_type = var_13392_pad_type_0, strides = var_13392_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_375_cast_fp16)[name = string("op_13392_cast_fp16")]; + tensor input_379_cast_fp16 = mul(x = var_13386_cast_fp16, y = var_13392_cast_fp16)[name = string("input_379_cast_fp16")]; + string hidden_states_453_pad_type_0 = const()[name = string("hidden_states_453_pad_type_0"), val = string("valid")]; + tensor hidden_states_453_strides_0 = const()[name = string("hidden_states_453_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_453_pad_0 = const()[name = string("hidden_states_453_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_453_dilations_0 = const()[name = string("hidden_states_453_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_453_groups_0 = const()[name = string("hidden_states_453_groups_0"), val = int32(1)]; + tensor hidden_states_453_cast_fp16 = conv(dilations = hidden_states_453_dilations_0, groups = hidden_states_453_groups_0, pad = hidden_states_453_pad_0, pad_type = hidden_states_453_pad_type_0, strides = hidden_states_453_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_379_cast_fp16)[name = string("hidden_states_453_cast_fp16")]; + tensor inputs_367_cast_fp16 = add(x = inputs_365_cast_fp16, y = hidden_states_453_cast_fp16)[name = string("inputs_367_cast_fp16")]; + tensor obj_391_begin_0 = const()[name = string("obj_391_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_391_end_0 = const()[name = string("obj_391_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_391_end_mask_0 = const()[name = string("obj_391_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_391_cast_fp16 = slice_by_index(begin = obj_391_begin_0, end = obj_391_end_0, end_mask = obj_391_end_mask_0, x = key_caches_17_cast_fp16)[name = string("obj_391_cast_fp16")]; + tensor obj_393_begin_0 = const()[name = string("obj_393_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_393_end_0 = const()[name = string("obj_393_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_393_end_mask_0 = const()[name = string("obj_393_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_393_cast_fp16 = slice_by_index(begin = obj_393_begin_0, end = obj_393_end_0, end_mask = obj_393_end_mask_0, x = value_caches_17_cast_fp16)[name = string("obj_393_cast_fp16")]; + int32 var_13440 = const()[name = string("op_13440"), val = int32(3)]; + int32 var_13450 = const()[name = string("op_13450"), val = int32(-2)]; + tensor inputs_sq_367_cast_fp16 = mul(x = inputs_367_cast_fp16, y = inputs_367_cast_fp16)[name = string("inputs_sq_367_cast_fp16")]; + tensor variance_367_axes_0 = const()[name = string("variance_367_axes_0"), val = tensor([1])]; + bool variance_367_keep_dims_0 = const()[name = string("variance_367_keep_dims_0"), val = bool(true)]; + tensor variance_367_cast_fp16 = reduce_mean(axes = variance_367_axes_0, keep_dims = variance_367_keep_dims_0, x = inputs_sq_367_cast_fp16)[name = string("variance_367_cast_fp16")]; + fp16 var_13464_to_fp16 = const()[name = string("op_13464_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13465_cast_fp16 = add(x = variance_367_cast_fp16, y = var_13464_to_fp16)[name = string("op_13465_cast_fp16")]; + fp32 var_13466_epsilon_0 = const()[name = string("op_13466_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13466_cast_fp16 = rsqrt(epsilon = var_13466_epsilon_0, x = var_13465_cast_fp16)[name = string("op_13466_cast_fp16")]; + tensor hidden_states_455_cast_fp16 = mul(x = inputs_367_cast_fp16, y = var_13466_cast_fp16)[name = string("hidden_states_455_cast_fp16")]; + tensor obj_389_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_455_cast_fp16)[name = string("obj_389_cast_fp16")]; + string query_265_pad_type_0 = const()[name = string("query_265_pad_type_0"), val = string("valid")]; + tensor query_265_strides_0 = const()[name = string("query_265_strides_0"), val = tensor([1, 1])]; + tensor query_265_pad_0 = const()[name = string("query_265_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_265_dilations_0 = const()[name = string("query_265_dilations_0"), val = tensor([1, 1])]; + int32 query_265_groups_0 = const()[name = string("query_265_groups_0"), val = int32(1)]; + tensor query_265_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_265_dilations_0, groups = query_265_groups_0, pad = query_265_pad_0, pad_type = query_265_pad_type_0, strides = query_265_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_389_cast_fp16)[name = string("query_265_cast_fp16")]; + string current_key_177_pad_type_0 = const()[name = string("current_key_177_pad_type_0"), val = string("valid")]; + tensor current_key_177_strides_0 = const()[name = string("current_key_177_strides_0"), val = tensor([1, 1])]; + tensor current_key_177_pad_0 = const()[name = string("current_key_177_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_177_dilations_0 = const()[name = string("current_key_177_dilations_0"), val = tensor([1, 1])]; + int32 current_key_177_groups_0 = const()[name = string("current_key_177_groups_0"), val = int32(1)]; + tensor current_key_177_cast_fp16 = conv(dilations = current_key_177_dilations_0, groups = current_key_177_groups_0, pad = current_key_177_pad_0, pad_type = current_key_177_pad_type_0, strides = current_key_177_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_389_cast_fp16)[name = string("current_key_177_cast_fp16")]; + string current_value_89_pad_type_0 = const()[name = string("current_value_89_pad_type_0"), val = string("valid")]; + tensor current_value_89_strides_0 = const()[name = string("current_value_89_strides_0"), val = tensor([1, 1])]; + tensor current_value_89_pad_0 = const()[name = string("current_value_89_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_89_dilations_0 = const()[name = string("current_value_89_dilations_0"), val = tensor([1, 1])]; + int32 current_value_89_groups_0 = const()[name = string("current_value_89_groups_0"), val = int32(1)]; + tensor current_value_89_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_89_dilations_0, groups = current_value_89_groups_0, pad = current_value_89_pad_0, pad_type = current_value_89_pad_type_0, strides = current_value_89_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_389_cast_fp16)[name = string("current_value_89_cast_fp16")]; + tensor var_13503 = const()[name = string("op_13503"), val = tensor([16, 128, 1, 1])]; + tensor inputs_369_cast_fp16 = reshape(shape = var_13503, x = query_265_cast_fp16)[name = string("inputs_369_cast_fp16")]; + tensor inputs_sq_369_cast_fp16 = mul(x = inputs_369_cast_fp16, y = inputs_369_cast_fp16)[name = string("inputs_sq_369_cast_fp16")]; + tensor variance_369_axes_0 = const()[name = string("variance_369_axes_0"), val = tensor([1])]; + bool variance_369_keep_dims_0 = const()[name = string("variance_369_keep_dims_0"), val = bool(true)]; + tensor variance_369_cast_fp16 = reduce_mean(axes = variance_369_axes_0, keep_dims = variance_369_keep_dims_0, x = inputs_sq_369_cast_fp16)[name = string("variance_369_cast_fp16")]; + fp16 var_13509_to_fp16 = const()[name = string("op_13509_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13510_cast_fp16 = add(x = variance_369_cast_fp16, y = var_13509_to_fp16)[name = string("op_13510_cast_fp16")]; + fp32 var_13511_epsilon_0 = const()[name = string("op_13511_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13511_cast_fp16 = rsqrt(epsilon = var_13511_epsilon_0, x = var_13510_cast_fp16)[name = string("op_13511_cast_fp16")]; + tensor hidden_states_457_cast_fp16 = mul(x = inputs_369_cast_fp16, y = var_13511_cast_fp16)[name = string("hidden_states_457_cast_fp16")]; + tensor query_normed_89_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_457_cast_fp16)[name = string("query_normed_89_cast_fp16")]; + tensor var_13519 = const()[name = string("op_13519"), val = tensor([8, 128, 1, 1])]; + tensor inputs_371_cast_fp16 = reshape(shape = var_13519, x = current_key_177_cast_fp16)[name = string("inputs_371_cast_fp16")]; + tensor inputs_sq_371_cast_fp16 = mul(x = inputs_371_cast_fp16, y = inputs_371_cast_fp16)[name = string("inputs_sq_371_cast_fp16")]; + tensor variance_371_axes_0 = const()[name = string("variance_371_axes_0"), val = tensor([1])]; + bool variance_371_keep_dims_0 = const()[name = string("variance_371_keep_dims_0"), val = bool(true)]; + tensor variance_371_cast_fp16 = reduce_mean(axes = variance_371_axes_0, keep_dims = variance_371_keep_dims_0, x = inputs_sq_371_cast_fp16)[name = string("variance_371_cast_fp16")]; + fp16 var_13525_to_fp16 = const()[name = string("op_13525_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13526_cast_fp16 = add(x = variance_371_cast_fp16, y = var_13525_to_fp16)[name = string("op_13526_cast_fp16")]; + fp32 var_13527_epsilon_0 = const()[name = string("op_13527_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13527_cast_fp16 = rsqrt(epsilon = var_13527_epsilon_0, x = var_13526_cast_fp16)[name = string("op_13527_cast_fp16")]; + tensor hidden_states_459_cast_fp16 = mul(x = inputs_371_cast_fp16, y = var_13527_cast_fp16)[name = string("hidden_states_459_cast_fp16")]; + tensor current_key_normed_89_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_459_cast_fp16)[name = string("current_key_normed_89_cast_fp16")]; + tensor var_13545 = const()[name = string("op_13545"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_353_cast_fp16 = reshape(shape = var_13545, x = query_normed_89_cast_fp16)[name = string("mh_q_353_cast_fp16")]; + tensor var_13547 = const()[name = string("op_13547"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_353_cast_fp16 = reshape(shape = var_13547, x = current_key_normed_89_cast_fp16)[name = string("mh_k_353_cast_fp16")]; + tensor var_13551_cast_fp16 = mul(x = mh_q_353_cast_fp16, y = cos_81_to_fp16)[name = string("op_13551_cast_fp16")]; + tensor var_13556_begin_0 = const()[name = string("op_13556_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13556_end_0 = const()[name = string("op_13556_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_13556_end_mask_0 = const()[name = string("op_13556_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_13556_cast_fp16 = slice_by_index(begin = var_13556_begin_0, end = var_13556_end_0, end_mask = var_13556_end_mask_0, x = mh_q_353_cast_fp16)[name = string("op_13556_cast_fp16")]; + tensor var_13562_begin_0 = const()[name = string("op_13562_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_13562_end_0 = const()[name = string("op_13562_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_13562_end_mask_0 = const()[name = string("op_13562_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_13562_cast_fp16 = slice_by_index(begin = var_13562_begin_0, end = var_13562_end_0, end_mask = var_13562_end_mask_0, x = mh_q_353_cast_fp16)[name = string("op_13562_cast_fp16")]; + fp16 const_902_promoted_to_fp16 = const()[name = string("const_902_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13564_cast_fp16 = mul(x = var_13562_cast_fp16, y = const_902_promoted_to_fp16)[name = string("op_13564_cast_fp16")]; + bool var_13566_interleave_0 = const()[name = string("op_13566_interleave_0"), val = bool(false)]; + tensor var_13566_cast_fp16 = concat(axis = var_13450, interleave = var_13566_interleave_0, values = (var_13564_cast_fp16, var_13556_cast_fp16))[name = string("op_13566_cast_fp16")]; + tensor var_13567_cast_fp16 = mul(x = var_13566_cast_fp16, y = sin_81_to_fp16)[name = string("op_13567_cast_fp16")]; + tensor mh_q_355_cast_fp16 = add(x = var_13551_cast_fp16, y = var_13567_cast_fp16)[name = string("mh_q_355_cast_fp16")]; + tensor var_13569_cast_fp16 = mul(x = mh_k_353_cast_fp16, y = cos_81_to_fp16)[name = string("op_13569_cast_fp16")]; + tensor var_13574_begin_0 = const()[name = string("op_13574_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13574_end_0 = const()[name = string("op_13574_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_13574_end_mask_0 = const()[name = string("op_13574_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_13574_cast_fp16 = slice_by_index(begin = var_13574_begin_0, end = var_13574_end_0, end_mask = var_13574_end_mask_0, x = mh_k_353_cast_fp16)[name = string("op_13574_cast_fp16")]; + tensor var_13580_begin_0 = const()[name = string("op_13580_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_13580_end_0 = const()[name = string("op_13580_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_13580_end_mask_0 = const()[name = string("op_13580_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_13580_cast_fp16 = slice_by_index(begin = var_13580_begin_0, end = var_13580_end_0, end_mask = var_13580_end_mask_0, x = mh_k_353_cast_fp16)[name = string("op_13580_cast_fp16")]; + fp16 const_905_promoted_to_fp16 = const()[name = string("const_905_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_13582_cast_fp16 = mul(x = var_13580_cast_fp16, y = const_905_promoted_to_fp16)[name = string("op_13582_cast_fp16")]; + bool var_13584_interleave_0 = const()[name = string("op_13584_interleave_0"), val = bool(false)]; + tensor var_13584_cast_fp16 = concat(axis = var_13450, interleave = var_13584_interleave_0, values = (var_13582_cast_fp16, var_13574_cast_fp16))[name = string("op_13584_cast_fp16")]; + tensor var_13585_cast_fp16 = mul(x = var_13584_cast_fp16, y = sin_81_to_fp16)[name = string("op_13585_cast_fp16")]; + tensor mh_k_355_cast_fp16 = add(x = var_13569_cast_fp16, y = var_13585_cast_fp16)[name = string("mh_k_355_cast_fp16")]; + tensor var_13589 = const()[name = string("op_13589"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_179_cast_fp16 = reshape(shape = var_13589, x = mh_k_355_cast_fp16)[name = string("current_key_179_cast_fp16")]; + tensor var_13595_to_fp16 = const()[name = string("op_13595_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197376)))]; + tensor var_13596_cast_fp16 = mul(x = obj_391_cast_fp16, y = var_13595_to_fp16)[name = string("op_13596_cast_fp16")]; + tensor var_13593_to_fp16 = const()[name = string("op_13593_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197504)))]; + tensor var_13597_cast_fp16 = mul(x = current_key_179_cast_fp16, y = var_13593_to_fp16)[name = string("op_13597_cast_fp16")]; + tensor key_179_cast_fp16 = add(x = var_13596_cast_fp16, y = var_13597_cast_fp16)[name = string("key_179_cast_fp16")]; + tensor var_13599_to_fp16 = const()[name = string("op_13599_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197376)))]; + tensor var_13600_cast_fp16 = mul(x = obj_393_cast_fp16, y = var_13599_to_fp16)[name = string("op_13600_cast_fp16")]; + tensor var_13601_cast_fp16 = mul(x = current_value_89_cast_fp16, y = var_13593_to_fp16)[name = string("op_13601_cast_fp16")]; + tensor value_89_cast_fp16 = add(x = var_13600_cast_fp16, y = var_13601_cast_fp16)[name = string("value_89_cast_fp16")]; + fp16 var_13608_to_fp16 = const()[name = string("op_13608_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_359_cast_fp16 = mul(x = mh_q_355_cast_fp16, y = var_13608_to_fp16)[name = string("mh_q_359_cast_fp16")]; + tensor var_13610 = const()[name = string("op_13610"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_357_cast_fp16 = reshape(shape = var_13610, x = key_179_cast_fp16)[name = string("mh_k_357_cast_fp16")]; + tensor var_13612 = const()[name = string("op_13612"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_177_cast_fp16 = reshape(shape = var_13612, x = value_89_cast_fp16)[name = string("mh_v_177_cast_fp16")]; + tensor transpose_176_perm_0 = const()[name = string("transpose_176_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_88_reps_0 = const()[name = string("tile_88_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_176_cast_fp16 = transpose(perm = transpose_176_perm_0, x = mh_k_357_cast_fp16)[name = string("transpose_215")]; + tensor tile_88_cast_fp16 = tile(reps = tile_88_reps_0, x = transpose_176_cast_fp16)[name = string("tile_88_cast_fp16")]; + tensor concat_219 = const()[name = string("concat_219"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_176_cast_fp16 = reshape(shape = concat_219, x = tile_88_cast_fp16)[name = string("reshape_176_cast_fp16")]; + tensor transpose_177_perm_0 = const()[name = string("transpose_177_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_220 = const()[name = string("concat_220"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_177_cast_fp16 = transpose(perm = transpose_177_perm_0, x = reshape_176_cast_fp16)[name = string("transpose_214")]; + tensor reshape_177_cast_fp16 = reshape(shape = concat_220, x = transpose_177_cast_fp16)[name = string("reshape_177_cast_fp16")]; + tensor transpose_178_perm_0 = const()[name = string("transpose_178_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_89_reps_0 = const()[name = string("tile_89_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_178_cast_fp16 = transpose(perm = transpose_178_perm_0, x = mh_v_177_cast_fp16)[name = string("transpose_213")]; + tensor tile_89_cast_fp16 = tile(reps = tile_89_reps_0, x = transpose_178_cast_fp16)[name = string("tile_89_cast_fp16")]; + tensor concat_221 = const()[name = string("concat_221"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_178_cast_fp16 = reshape(shape = concat_221, x = tile_89_cast_fp16)[name = string("reshape_178_cast_fp16")]; + tensor transpose_179_perm_0 = const()[name = string("transpose_179_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_222 = const()[name = string("concat_222"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_179_cast_fp16 = transpose(perm = transpose_179_perm_0, x = reshape_178_cast_fp16)[name = string("transpose_212")]; + tensor reshape_179_cast_fp16 = reshape(shape = concat_222, x = transpose_179_cast_fp16)[name = string("reshape_179_cast_fp16")]; + tensor transpose_493_perm_0 = const()[name = string("transpose_493_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_265_transpose_x_1 = const()[name = string("mh_w_265_transpose_x_1"), val = bool(true)]; + bool mh_w_265_transpose_y_1 = const()[name = string("mh_w_265_transpose_y_1"), val = bool(false)]; + tensor transpose_493_cast_fp16 = transpose(perm = transpose_493_perm_0, x = reshape_177_cast_fp16)[name = string("transpose_211")]; + tensor mh_w_265_cast_fp16 = matmul(transpose_x = mh_w_265_transpose_x_1, transpose_y = mh_w_265_transpose_y_1, x = mh_q_359_cast_fp16, y = transpose_493_cast_fp16)[name = string("mh_w_265_cast_fp16")]; + tensor var_13620_to_fp16 = const()[name = string("op_13620_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197632)))]; + tensor mh_w_267_cast_fp16 = add(x = mh_w_265_cast_fp16, y = var_13620_to_fp16)[name = string("mh_w_267_cast_fp16")]; + tensor mh_w_269_cast_fp16 = softmax(axis = var_13440, x = mh_w_267_cast_fp16)[name = string("mh_w_269_cast_fp16")]; + tensor transpose_494_perm_0 = const()[name = string("transpose_494_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_89_transpose_x_1 = const()[name = string("attn_89_transpose_x_1"), val = bool(false)]; + bool attn_89_transpose_y_1 = const()[name = string("attn_89_transpose_y_1"), val = bool(true)]; + tensor transpose_494_cast_fp16 = transpose(perm = transpose_494_perm_0, x = reshape_179_cast_fp16)[name = string("transpose_210")]; + tensor attn_89_cast_fp16 = matmul(transpose_x = attn_89_transpose_x_1, transpose_y = attn_89_transpose_y_1, x = transpose_494_cast_fp16, y = mh_w_269_cast_fp16)[name = string("attn_89_cast_fp16")]; + tensor var_13626 = const()[name = string("op_13626"), val = tensor([1, 2048, 1, 1])]; + tensor input_381_cast_fp16 = reshape(shape = var_13626, x = attn_89_cast_fp16)[name = string("input_381_cast_fp16")]; + string obj_395_pad_type_0 = const()[name = string("obj_395_pad_type_0"), val = string("valid")]; + tensor obj_395_strides_0 = const()[name = string("obj_395_strides_0"), val = tensor([1, 1])]; + tensor obj_395_pad_0 = const()[name = string("obj_395_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_395_dilations_0 = const()[name = string("obj_395_dilations_0"), val = tensor([1, 1])]; + int32 obj_395_groups_0 = const()[name = string("obj_395_groups_0"), val = int32(1)]; + tensor obj_395_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_395_dilations_0, groups = obj_395_groups_0, pad = obj_395_pad_0, pad_type = obj_395_pad_type_0, strides = obj_395_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_381_cast_fp16)[name = string("obj_395_cast_fp16")]; + tensor inputs_373_cast_fp16 = add(x = inputs_367_cast_fp16, y = obj_395_cast_fp16)[name = string("inputs_373_cast_fp16")]; + tensor inputs_sq_373_cast_fp16 = mul(x = inputs_373_cast_fp16, y = inputs_373_cast_fp16)[name = string("inputs_sq_373_cast_fp16")]; + tensor variance_373_axes_0 = const()[name = string("variance_373_axes_0"), val = tensor([1])]; + bool variance_373_keep_dims_0 = const()[name = string("variance_373_keep_dims_0"), val = bool(true)]; + tensor variance_373_cast_fp16 = reduce_mean(axes = variance_373_axes_0, keep_dims = variance_373_keep_dims_0, x = inputs_sq_373_cast_fp16)[name = string("variance_373_cast_fp16")]; + fp16 var_13644_to_fp16 = const()[name = string("op_13644_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13645_cast_fp16 = add(x = variance_373_cast_fp16, y = var_13644_to_fp16)[name = string("op_13645_cast_fp16")]; + fp32 var_13646_epsilon_0 = const()[name = string("op_13646_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13646_cast_fp16 = rsqrt(epsilon = var_13646_epsilon_0, x = var_13645_cast_fp16)[name = string("op_13646_cast_fp16")]; + tensor hidden_states_461_cast_fp16 = mul(x = inputs_373_cast_fp16, y = var_13646_cast_fp16)[name = string("hidden_states_461_cast_fp16")]; + tensor input_383_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_461_cast_fp16)[name = string("input_383_cast_fp16")]; + string input_385_pad_type_0 = const()[name = string("input_385_pad_type_0"), val = string("valid")]; + tensor input_385_strides_0 = const()[name = string("input_385_strides_0"), val = tensor([1, 1])]; + tensor input_385_pad_0 = const()[name = string("input_385_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_385_dilations_0 = const()[name = string("input_385_dilations_0"), val = tensor([1, 1])]; + int32 input_385_groups_0 = const()[name = string("input_385_groups_0"), val = int32(1)]; + tensor input_385_cast_fp16 = conv(dilations = input_385_dilations_0, groups = input_385_groups_0, pad = input_385_pad_0, pad_type = input_385_pad_type_0, strides = input_385_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_383_cast_fp16)[name = string("input_385_cast_fp16")]; + tensor var_13660_cast_fp16 = silu(x = input_385_cast_fp16)[name = string("op_13660_cast_fp16")]; + string var_13666_pad_type_0 = const()[name = string("op_13666_pad_type_0"), val = string("valid")]; + tensor var_13666_strides_0 = const()[name = string("op_13666_strides_0"), val = tensor([1, 1])]; + tensor var_13666_pad_0 = const()[name = string("op_13666_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13666_dilations_0 = const()[name = string("op_13666_dilations_0"), val = tensor([1, 1])]; + int32 var_13666_groups_0 = const()[name = string("op_13666_groups_0"), val = int32(1)]; + tensor var_13666_cast_fp16 = conv(dilations = var_13666_dilations_0, groups = var_13666_groups_0, pad = var_13666_pad_0, pad_type = var_13666_pad_type_0, strides = var_13666_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_383_cast_fp16)[name = string("op_13666_cast_fp16")]; + tensor input_387_cast_fp16 = mul(x = var_13660_cast_fp16, y = var_13666_cast_fp16)[name = string("input_387_cast_fp16")]; + string hidden_states_463_pad_type_0 = const()[name = string("hidden_states_463_pad_type_0"), val = string("valid")]; + tensor hidden_states_463_strides_0 = const()[name = string("hidden_states_463_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_463_pad_0 = const()[name = string("hidden_states_463_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_463_dilations_0 = const()[name = string("hidden_states_463_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_463_groups_0 = const()[name = string("hidden_states_463_groups_0"), val = int32(1)]; + tensor hidden_states_463_cast_fp16 = conv(dilations = hidden_states_463_dilations_0, groups = hidden_states_463_groups_0, pad = hidden_states_463_pad_0, pad_type = hidden_states_463_pad_type_0, strides = hidden_states_463_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_387_cast_fp16)[name = string("hidden_states_463_cast_fp16")]; + tensor inputs_375_cast_fp16 = add(x = inputs_373_cast_fp16, y = hidden_states_463_cast_fp16)[name = string("inputs_375_cast_fp16")]; + int32 var_13694 = const()[name = string("op_13694"), val = int32(1)]; + bool key_caches_19_interleave_0 = const()[name = string("key_caches_19_interleave_0"), val = bool(false)]; + tensor key_caches_19_cast_fp16 = concat(axis = var_13694, interleave = key_caches_19_interleave_0, values = (key_163_cast_fp16, key_167_cast_fp16, key_171_cast_fp16, key_175_cast_fp16, key_179_cast_fp16))[name = string("key_caches_19_cast_fp16")]; + int32 var_13697 = const()[name = string("op_13697"), val = int32(1)]; + bool value_caches_19_interleave_0 = const()[name = string("value_caches_19_interleave_0"), val = bool(false)]; + tensor value_caches_19_cast_fp16 = concat(axis = var_13697, interleave = value_caches_19_interleave_0, values = (value_81_cast_fp16, value_83_cast_fp16, value_85_cast_fp16, value_87_cast_fp16, value_89_cast_fp16))[name = string("value_caches_19_cast_fp16")]; + tensor inputs_sq_375_cast_fp16 = mul(x = inputs_375_cast_fp16, y = inputs_375_cast_fp16)[name = string("inputs_sq_375_cast_fp16")]; + tensor variance_375_axes_0 = const()[name = string("variance_375_axes_0"), val = tensor([1])]; + bool variance_375_keep_dims_0 = const()[name = string("variance_375_keep_dims_0"), val = bool(true)]; + tensor variance_375_cast_fp16 = reduce_mean(axes = variance_375_axes_0, keep_dims = variance_375_keep_dims_0, x = inputs_sq_375_cast_fp16)[name = string("variance_375_cast_fp16")]; + fp16 var_13707_to_fp16 = const()[name = string("op_13707_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13708_cast_fp16 = add(x = variance_375_cast_fp16, y = var_13707_to_fp16)[name = string("op_13708_cast_fp16")]; + fp32 var_13709_epsilon_0 = const()[name = string("op_13709_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13709_cast_fp16 = rsqrt(epsilon = var_13709_epsilon_0, x = var_13708_cast_fp16)[name = string("op_13709_cast_fp16")]; + tensor hidden_states_465_cast_fp16 = mul(x = inputs_375_cast_fp16, y = var_13709_cast_fp16)[name = string("hidden_states_465_cast_fp16")]; + tensor input_389_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_465_cast_fp16)[name = string("input_389_cast_fp16")]; + string logits_29_pad_type_0 = const()[name = string("logits_29_pad_type_0"), val = string("valid")]; + tensor logits_29_strides_0 = const()[name = string("logits_29_strides_0"), val = tensor([1, 1])]; + tensor logits_29_pad_0 = const()[name = string("logits_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_29_dilations_0 = const()[name = string("logits_29_dilations_0"), val = tensor([1, 1])]; + int32 logits_29_groups_0 = const()[name = string("logits_29_groups_0"), val = int32(1)]; + tensor lm_heads_7_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95491136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(97588352))))[name = string("lm_heads_7_weight_to_fp16_palettized")]; + tensor logits_29_cast_fp16 = conv(dilations = logits_29_dilations_0, groups = logits_29_groups_0, pad = logits_29_pad_0, pad_type = logits_29_pad_type_0, strides = logits_29_strides_0, weight = lm_heads_7_weight_to_fp16_palettized, x = input_389_cast_fp16)[name = string("logits_29_cast_fp16")]; + tensor var_13727 = const()[name = string("op_13727"), val = tensor([1, 2048])]; + tensor logits_31_cast_fp16 = reshape(shape = var_13727, x = logits_29_cast_fp16)[name = string("logits_31_cast_fp16")]; + tensor scaled_logits_15_cast_fp16 = real_div(x = logits_31_cast_fp16, y = temperature)[name = string("scaled_logits_15_cast_fp16")]; + int32 var_13737 = const()[name = string("op_13737"), val = int32(100)]; + int32 top_values_15_axis_0 = const()[name = string("top_values_15_axis_0"), val = int32(1)]; + bool top_values_15_ascending_0 = const()[name = string("top_values_15_ascending_0"), val = bool(false)]; + bool top_values_15_sort_0 = const()[name = string("top_values_15_sort_0"), val = bool(true)]; + bool top_values_15_return_indices_0 = const()[name = string("top_values_15_return_indices_0"), val = bool(true)]; + string top_values_15_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_15_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_15_cast_fp16_cast_uint16_0, tensor top_values_15_cast_fp16_cast_uint16_1 = topk(ascending = top_values_15_ascending_0, axis = top_values_15_axis_0, k = var_13737, output_indices_dtype = top_values_15_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_15_return_indices_0, sort = top_values_15_sort_0, x = scaled_logits_15_cast_fp16)[name = string("top_values_15_cast_fp16_cast_uint16")]; + tensor var_13743_cast_fp16 = mul(x = top_values_15_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_13743_cast_fp16")]; + tensor var_13747_cast_fp16 = add(x = var_13743_cast_fp16, y = var_3001_cast_fp16)[name = string("op_13747_cast_fp16")]; + tensor reduce_min_7_axes_0 = const()[name = string("reduce_min_7_axes_0"), val = tensor([1])]; + bool reduce_min_7_keep_dims_0 = const()[name = string("reduce_min_7_keep_dims_0"), val = bool(true)]; + tensor reduce_min_7_cast_fp16 = reduce_min(axes = reduce_min_7_axes_0, keep_dims = reduce_min_7_keep_dims_0, x = var_13747_cast_fp16)[name = string("reduce_min_7_cast_fp16")]; + tensor var_13750_cast_fp16 = greater_equal(x = scaled_logits_15_cast_fp16, y = reduce_min_7_cast_fp16)[name = string("op_13750_cast_fp16")]; + fp16 var_13751_value_0_to_fp16 = const()[name = string("op_13751_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_13751_cast_fp16 = fill_like(ref_tensor = scaled_logits_15_cast_fp16, value = var_13751_value_0_to_fp16)[name = string("op_13751_cast_fp16")]; + tensor masked_logits_15_cast_fp16 = select(a = scaled_logits_15_cast_fp16, b = var_13751_cast_fp16, cond = var_13750_cast_fp16)[name = string("masked_logits_15_cast_fp16")]; + tensor var_13755_begin_0 = const()[name = string("op_13755_begin_0"), val = tensor([7, 0])]; + tensor var_13755_end_0 = const()[name = string("op_13755_end_0"), val = tensor([8, 2048])]; + tensor var_13755_end_mask_0 = const()[name = string("op_13755_end_mask_0"), val = tensor([false, true])]; + tensor var_13755_squeeze_mask_0 = const()[name = string("op_13755_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_13755_cast_fp16 = slice_by_index(begin = var_13755_begin_0, end = var_13755_end_0, end_mask = var_13755_end_mask_0, squeeze_mask = var_13755_squeeze_mask_0, x = gumbel)[name = string("op_13755_cast_fp16")]; + tensor var_13758 = const()[name = string("op_13758"), val = tensor([1, 2048])]; + tensor var_13759_cast_fp16 = reshape(shape = var_13758, x = var_13755_cast_fp16)[name = string("op_13759_cast_fp16")]; + tensor noisy_logits_15_cast_fp16 = add(x = masked_logits_15_cast_fp16, y = var_13759_cast_fp16)[name = string("noisy_logits_15_cast_fp16")]; + int32 code_15_axis_0 = const()[name = string("code_15_axis_0"), val = int32(1)]; + bool code_15_keep_dims_0 = const()[name = string("code_15_keep_dims_0"), val = bool(false)]; + string code_15_output_dtype_0 = const()[name = string("code_15_output_dtype_0"), val = string("int32")]; + tensor code_15_cast_fp16 = reduce_argmax(axis = code_15_axis_0, keep_dims = code_15_keep_dims_0, output_dtype = code_15_output_dtype_0, x = noisy_logits_15_cast_fp16)[name = string("code_15_cast_fp16")]; + int32 var_13770 = const()[name = string("op_13770"), val = int32(14336)]; + tensor input_391 = add(x = code_15_cast_fp16, y = var_13770)[name = string("input_391")]; + int32 code_embed_29_axis_0 = const()[name = string("code_embed_29_axis_0"), val = int32(0)]; + int32 code_embed_29_batch_dims_0 = const()[name = string("code_embed_29_batch_dims_0"), val = int32(0)]; + bool code_embed_29_validate_indices_0 = const()[name = string("code_embed_29_validate_indices_0"), val = bool(false)]; + string input_391_to_uint16_dtype_0 = const()[name = string("input_391_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_391_to_uint16 = cast(dtype = input_391_to_uint16_dtype_0, x = input_391)[name = string("cast_7")]; + tensor code_embed_29_cast_fp16_cast_uint16 = gather(axis = code_embed_29_axis_0, batch_dims = code_embed_29_batch_dims_0, indices = input_391_to_uint16, validate_indices = code_embed_29_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_29_cast_fp16_cast_uint16")]; + tensor var_13774 = const()[name = string("op_13774"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_31_cast_fp16 = reshape(shape = var_13774, x = code_embed_29_cast_fp16_cast_uint16)[name = string("code_embed_31_cast_fp16")]; + tensor embed_sum_17_cast_fp16 = add(x = embed_sum_15_cast_fp16, y = code_embed_31_cast_fp16)[name = string("embed_sum_17_cast_fp16")]; + string inputs_377_pad_type_0 = const()[name = string("inputs_377_pad_type_0"), val = string("valid")]; + tensor inputs_377_strides_0 = const()[name = string("inputs_377_strides_0"), val = tensor([1, 1])]; + tensor inputs_377_pad_0 = const()[name = string("inputs_377_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_377_dilations_0 = const()[name = string("inputs_377_dilations_0"), val = tensor([1, 1])]; + int32 inputs_377_groups_0 = const()[name = string("inputs_377_groups_0"), val = int32(1)]; + tensor inputs_377_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_377_dilations_0, groups = inputs_377_groups_0, pad = inputs_377_pad_0, pad_type = inputs_377_pad_type_0, strides = inputs_377_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_31_cast_fp16)[name = string("inputs_377_cast_fp16")]; + tensor obj_399_begin_0 = const()[name = string("obj_399_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_399_end_0 = const()[name = string("obj_399_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_399_end_mask_0 = const()[name = string("obj_399_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_399_cast_fp16 = slice_by_index(begin = obj_399_begin_0, end = obj_399_end_0, end_mask = obj_399_end_mask_0, x = key_caches_19_cast_fp16)[name = string("obj_399_cast_fp16")]; + tensor obj_401_begin_0 = const()[name = string("obj_401_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_401_end_0 = const()[name = string("obj_401_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_401_end_mask_0 = const()[name = string("obj_401_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_401_cast_fp16 = slice_by_index(begin = obj_401_begin_0, end = obj_401_end_0, end_mask = obj_401_end_mask_0, x = value_caches_19_cast_fp16)[name = string("obj_401_cast_fp16")]; + int32 var_13879 = const()[name = string("op_13879"), val = int32(3)]; + int32 var_13889 = const()[name = string("op_13889"), val = int32(-2)]; + tensor inputs_sq_377_cast_fp16 = mul(x = inputs_377_cast_fp16, y = inputs_377_cast_fp16)[name = string("inputs_sq_377_cast_fp16")]; + tensor variance_377_axes_0 = const()[name = string("variance_377_axes_0"), val = tensor([1])]; + bool variance_377_keep_dims_0 = const()[name = string("variance_377_keep_dims_0"), val = bool(true)]; + tensor variance_377_cast_fp16 = reduce_mean(axes = variance_377_axes_0, keep_dims = variance_377_keep_dims_0, x = inputs_sq_377_cast_fp16)[name = string("variance_377_cast_fp16")]; + fp16 var_13903_to_fp16 = const()[name = string("op_13903_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13904_cast_fp16 = add(x = variance_377_cast_fp16, y = var_13903_to_fp16)[name = string("op_13904_cast_fp16")]; + fp32 var_13905_epsilon_0 = const()[name = string("op_13905_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13905_cast_fp16 = rsqrt(epsilon = var_13905_epsilon_0, x = var_13904_cast_fp16)[name = string("op_13905_cast_fp16")]; + tensor hidden_states_467_cast_fp16 = mul(x = inputs_377_cast_fp16, y = var_13905_cast_fp16)[name = string("hidden_states_467_cast_fp16")]; + tensor obj_397_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_467_cast_fp16)[name = string("obj_397_cast_fp16")]; + string query_271_pad_type_0 = const()[name = string("query_271_pad_type_0"), val = string("valid")]; + tensor query_271_strides_0 = const()[name = string("query_271_strides_0"), val = tensor([1, 1])]; + tensor query_271_pad_0 = const()[name = string("query_271_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_271_dilations_0 = const()[name = string("query_271_dilations_0"), val = tensor([1, 1])]; + int32 query_271_groups_0 = const()[name = string("query_271_groups_0"), val = int32(1)]; + tensor query_271_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_271_dilations_0, groups = query_271_groups_0, pad = query_271_pad_0, pad_type = query_271_pad_type_0, strides = query_271_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_397_cast_fp16)[name = string("query_271_cast_fp16")]; + string current_key_181_pad_type_0 = const()[name = string("current_key_181_pad_type_0"), val = string("valid")]; + tensor current_key_181_strides_0 = const()[name = string("current_key_181_strides_0"), val = tensor([1, 1])]; + tensor current_key_181_pad_0 = const()[name = string("current_key_181_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_181_dilations_0 = const()[name = string("current_key_181_dilations_0"), val = tensor([1, 1])]; + int32 current_key_181_groups_0 = const()[name = string("current_key_181_groups_0"), val = int32(1)]; + tensor current_key_181_cast_fp16 = conv(dilations = current_key_181_dilations_0, groups = current_key_181_groups_0, pad = current_key_181_pad_0, pad_type = current_key_181_pad_type_0, strides = current_key_181_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_397_cast_fp16)[name = string("current_key_181_cast_fp16")]; + string current_value_91_pad_type_0 = const()[name = string("current_value_91_pad_type_0"), val = string("valid")]; + tensor current_value_91_strides_0 = const()[name = string("current_value_91_strides_0"), val = tensor([1, 1])]; + tensor current_value_91_pad_0 = const()[name = string("current_value_91_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_91_dilations_0 = const()[name = string("current_value_91_dilations_0"), val = tensor([1, 1])]; + int32 current_value_91_groups_0 = const()[name = string("current_value_91_groups_0"), val = int32(1)]; + tensor current_value_91_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_91_dilations_0, groups = current_value_91_groups_0, pad = current_value_91_pad_0, pad_type = current_value_91_pad_type_0, strides = current_value_91_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_397_cast_fp16)[name = string("current_value_91_cast_fp16")]; + tensor var_13942 = const()[name = string("op_13942"), val = tensor([16, 128, 1, 1])]; + tensor inputs_379_cast_fp16 = reshape(shape = var_13942, x = query_271_cast_fp16)[name = string("inputs_379_cast_fp16")]; + tensor inputs_sq_379_cast_fp16 = mul(x = inputs_379_cast_fp16, y = inputs_379_cast_fp16)[name = string("inputs_sq_379_cast_fp16")]; + tensor variance_379_axes_0 = const()[name = string("variance_379_axes_0"), val = tensor([1])]; + bool variance_379_keep_dims_0 = const()[name = string("variance_379_keep_dims_0"), val = bool(true)]; + tensor variance_379_cast_fp16 = reduce_mean(axes = variance_379_axes_0, keep_dims = variance_379_keep_dims_0, x = inputs_sq_379_cast_fp16)[name = string("variance_379_cast_fp16")]; + fp16 var_13948_to_fp16 = const()[name = string("op_13948_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13949_cast_fp16 = add(x = variance_379_cast_fp16, y = var_13948_to_fp16)[name = string("op_13949_cast_fp16")]; + fp32 var_13950_epsilon_0 = const()[name = string("op_13950_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13950_cast_fp16 = rsqrt(epsilon = var_13950_epsilon_0, x = var_13949_cast_fp16)[name = string("op_13950_cast_fp16")]; + tensor hidden_states_469_cast_fp16 = mul(x = inputs_379_cast_fp16, y = var_13950_cast_fp16)[name = string("hidden_states_469_cast_fp16")]; + tensor query_normed_91_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_469_cast_fp16)[name = string("query_normed_91_cast_fp16")]; + tensor var_13958 = const()[name = string("op_13958"), val = tensor([8, 128, 1, 1])]; + tensor inputs_381_cast_fp16 = reshape(shape = var_13958, x = current_key_181_cast_fp16)[name = string("inputs_381_cast_fp16")]; + tensor inputs_sq_381_cast_fp16 = mul(x = inputs_381_cast_fp16, y = inputs_381_cast_fp16)[name = string("inputs_sq_381_cast_fp16")]; + tensor variance_381_axes_0 = const()[name = string("variance_381_axes_0"), val = tensor([1])]; + bool variance_381_keep_dims_0 = const()[name = string("variance_381_keep_dims_0"), val = bool(true)]; + tensor variance_381_cast_fp16 = reduce_mean(axes = variance_381_axes_0, keep_dims = variance_381_keep_dims_0, x = inputs_sq_381_cast_fp16)[name = string("variance_381_cast_fp16")]; + fp16 var_13964_to_fp16 = const()[name = string("op_13964_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_13965_cast_fp16 = add(x = variance_381_cast_fp16, y = var_13964_to_fp16)[name = string("op_13965_cast_fp16")]; + fp32 var_13966_epsilon_0 = const()[name = string("op_13966_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_13966_cast_fp16 = rsqrt(epsilon = var_13966_epsilon_0, x = var_13965_cast_fp16)[name = string("op_13966_cast_fp16")]; + tensor hidden_states_471_cast_fp16 = mul(x = inputs_381_cast_fp16, y = var_13966_cast_fp16)[name = string("hidden_states_471_cast_fp16")]; + tensor current_key_normed_91_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_471_cast_fp16)[name = string("current_key_normed_91_cast_fp16")]; + tensor var_13984 = const()[name = string("op_13984"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_361_cast_fp16 = reshape(shape = var_13984, x = query_normed_91_cast_fp16)[name = string("mh_q_361_cast_fp16")]; + tensor var_13986 = const()[name = string("op_13986"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_361_cast_fp16 = reshape(shape = var_13986, x = current_key_normed_91_cast_fp16)[name = string("mh_k_361_cast_fp16")]; + tensor cos_91_to_fp16 = const()[name = string("cos_91_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175197760)))]; + tensor var_13990_cast_fp16 = mul(x = mh_q_361_cast_fp16, y = cos_91_to_fp16)[name = string("op_13990_cast_fp16")]; + tensor var_13995_begin_0 = const()[name = string("op_13995_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_13995_end_0 = const()[name = string("op_13995_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_13995_end_mask_0 = const()[name = string("op_13995_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_13995_cast_fp16 = slice_by_index(begin = var_13995_begin_0, end = var_13995_end_0, end_mask = var_13995_end_mask_0, x = mh_q_361_cast_fp16)[name = string("op_13995_cast_fp16")]; + tensor var_14001_begin_0 = const()[name = string("op_14001_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_14001_end_0 = const()[name = string("op_14001_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_14001_end_mask_0 = const()[name = string("op_14001_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_14001_cast_fp16 = slice_by_index(begin = var_14001_begin_0, end = var_14001_end_0, end_mask = var_14001_end_mask_0, x = mh_q_361_cast_fp16)[name = string("op_14001_cast_fp16")]; + fp16 const_923_promoted_to_fp16 = const()[name = string("const_923_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14003_cast_fp16 = mul(x = var_14001_cast_fp16, y = const_923_promoted_to_fp16)[name = string("op_14003_cast_fp16")]; + bool var_14005_interleave_0 = const()[name = string("op_14005_interleave_0"), val = bool(false)]; + tensor var_14005_cast_fp16 = concat(axis = var_13889, interleave = var_14005_interleave_0, values = (var_14003_cast_fp16, var_13995_cast_fp16))[name = string("op_14005_cast_fp16")]; + tensor sin_91_to_fp16 = const()[name = string("sin_91_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198080)))]; + tensor var_14006_cast_fp16 = mul(x = var_14005_cast_fp16, y = sin_91_to_fp16)[name = string("op_14006_cast_fp16")]; + tensor mh_q_363_cast_fp16 = add(x = var_13990_cast_fp16, y = var_14006_cast_fp16)[name = string("mh_q_363_cast_fp16")]; + tensor var_14008_cast_fp16 = mul(x = mh_k_361_cast_fp16, y = cos_91_to_fp16)[name = string("op_14008_cast_fp16")]; + tensor var_14013_begin_0 = const()[name = string("op_14013_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14013_end_0 = const()[name = string("op_14013_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_14013_end_mask_0 = const()[name = string("op_14013_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_14013_cast_fp16 = slice_by_index(begin = var_14013_begin_0, end = var_14013_end_0, end_mask = var_14013_end_mask_0, x = mh_k_361_cast_fp16)[name = string("op_14013_cast_fp16")]; + tensor var_14019_begin_0 = const()[name = string("op_14019_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_14019_end_0 = const()[name = string("op_14019_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_14019_end_mask_0 = const()[name = string("op_14019_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_14019_cast_fp16 = slice_by_index(begin = var_14019_begin_0, end = var_14019_end_0, end_mask = var_14019_end_mask_0, x = mh_k_361_cast_fp16)[name = string("op_14019_cast_fp16")]; + fp16 const_926_promoted_to_fp16 = const()[name = string("const_926_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14021_cast_fp16 = mul(x = var_14019_cast_fp16, y = const_926_promoted_to_fp16)[name = string("op_14021_cast_fp16")]; + bool var_14023_interleave_0 = const()[name = string("op_14023_interleave_0"), val = bool(false)]; + tensor var_14023_cast_fp16 = concat(axis = var_13889, interleave = var_14023_interleave_0, values = (var_14021_cast_fp16, var_14013_cast_fp16))[name = string("op_14023_cast_fp16")]; + tensor var_14024_cast_fp16 = mul(x = var_14023_cast_fp16, y = sin_91_to_fp16)[name = string("op_14024_cast_fp16")]; + tensor mh_k_363_cast_fp16 = add(x = var_14008_cast_fp16, y = var_14024_cast_fp16)[name = string("mh_k_363_cast_fp16")]; + tensor var_14028 = const()[name = string("op_14028"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_183_cast_fp16 = reshape(shape = var_14028, x = mh_k_363_cast_fp16)[name = string("current_key_183_cast_fp16")]; + tensor var_14034_to_fp16 = const()[name = string("op_14034_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198400)))]; + tensor var_14035_cast_fp16 = mul(x = obj_399_cast_fp16, y = var_14034_to_fp16)[name = string("op_14035_cast_fp16")]; + tensor var_14032_to_fp16 = const()[name = string("op_14032_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198528)))]; + tensor var_14036_cast_fp16 = mul(x = current_key_183_cast_fp16, y = var_14032_to_fp16)[name = string("op_14036_cast_fp16")]; + tensor key_183_cast_fp16 = add(x = var_14035_cast_fp16, y = var_14036_cast_fp16)[name = string("key_183_cast_fp16")]; + tensor var_14038_to_fp16 = const()[name = string("op_14038_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198400)))]; + tensor var_14039_cast_fp16 = mul(x = obj_401_cast_fp16, y = var_14038_to_fp16)[name = string("op_14039_cast_fp16")]; + tensor var_14040_cast_fp16 = mul(x = current_value_91_cast_fp16, y = var_14032_to_fp16)[name = string("op_14040_cast_fp16")]; + tensor value_91_cast_fp16 = add(x = var_14039_cast_fp16, y = var_14040_cast_fp16)[name = string("value_91_cast_fp16")]; + fp16 var_14047_to_fp16 = const()[name = string("op_14047_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_367_cast_fp16 = mul(x = mh_q_363_cast_fp16, y = var_14047_to_fp16)[name = string("mh_q_367_cast_fp16")]; + tensor var_14049 = const()[name = string("op_14049"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_365_cast_fp16 = reshape(shape = var_14049, x = key_183_cast_fp16)[name = string("mh_k_365_cast_fp16")]; + tensor var_14051 = const()[name = string("op_14051"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_181_cast_fp16 = reshape(shape = var_14051, x = value_91_cast_fp16)[name = string("mh_v_181_cast_fp16")]; + tensor transpose_180_perm_0 = const()[name = string("transpose_180_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_90_reps_0 = const()[name = string("tile_90_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_180_cast_fp16 = transpose(perm = transpose_180_perm_0, x = mh_k_365_cast_fp16)[name = string("transpose_209")]; + tensor tile_90_cast_fp16 = tile(reps = tile_90_reps_0, x = transpose_180_cast_fp16)[name = string("tile_90_cast_fp16")]; + tensor concat_228 = const()[name = string("concat_228"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_180_cast_fp16 = reshape(shape = concat_228, x = tile_90_cast_fp16)[name = string("reshape_180_cast_fp16")]; + tensor transpose_181_perm_0 = const()[name = string("transpose_181_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_229 = const()[name = string("concat_229"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_181_cast_fp16 = transpose(perm = transpose_181_perm_0, x = reshape_180_cast_fp16)[name = string("transpose_208")]; + tensor reshape_181_cast_fp16 = reshape(shape = concat_229, x = transpose_181_cast_fp16)[name = string("reshape_181_cast_fp16")]; + tensor transpose_182_perm_0 = const()[name = string("transpose_182_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_91_reps_0 = const()[name = string("tile_91_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_182_cast_fp16 = transpose(perm = transpose_182_perm_0, x = mh_v_181_cast_fp16)[name = string("transpose_207")]; + tensor tile_91_cast_fp16 = tile(reps = tile_91_reps_0, x = transpose_182_cast_fp16)[name = string("tile_91_cast_fp16")]; + tensor concat_230 = const()[name = string("concat_230"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_182_cast_fp16 = reshape(shape = concat_230, x = tile_91_cast_fp16)[name = string("reshape_182_cast_fp16")]; + tensor transpose_183_perm_0 = const()[name = string("transpose_183_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_231 = const()[name = string("concat_231"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_183_cast_fp16 = transpose(perm = transpose_183_perm_0, x = reshape_182_cast_fp16)[name = string("transpose_206")]; + tensor reshape_183_cast_fp16 = reshape(shape = concat_231, x = transpose_183_cast_fp16)[name = string("reshape_183_cast_fp16")]; + tensor transpose_497_perm_0 = const()[name = string("transpose_497_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_271_transpose_x_1 = const()[name = string("mh_w_271_transpose_x_1"), val = bool(true)]; + bool mh_w_271_transpose_y_1 = const()[name = string("mh_w_271_transpose_y_1"), val = bool(false)]; + tensor transpose_497_cast_fp16 = transpose(perm = transpose_497_perm_0, x = reshape_181_cast_fp16)[name = string("transpose_205")]; + tensor mh_w_271_cast_fp16 = matmul(transpose_x = mh_w_271_transpose_x_1, transpose_y = mh_w_271_transpose_y_1, x = mh_q_367_cast_fp16, y = transpose_497_cast_fp16)[name = string("mh_w_271_cast_fp16")]; + tensor var_14059_to_fp16 = const()[name = string("op_14059_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198656)))]; + tensor mh_w_273_cast_fp16 = add(x = mh_w_271_cast_fp16, y = var_14059_to_fp16)[name = string("mh_w_273_cast_fp16")]; + tensor mh_w_275_cast_fp16 = softmax(axis = var_13879, x = mh_w_273_cast_fp16)[name = string("mh_w_275_cast_fp16")]; + tensor transpose_498_perm_0 = const()[name = string("transpose_498_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_91_transpose_x_1 = const()[name = string("attn_91_transpose_x_1"), val = bool(false)]; + bool attn_91_transpose_y_1 = const()[name = string("attn_91_transpose_y_1"), val = bool(true)]; + tensor transpose_498_cast_fp16 = transpose(perm = transpose_498_perm_0, x = reshape_183_cast_fp16)[name = string("transpose_204")]; + tensor attn_91_cast_fp16 = matmul(transpose_x = attn_91_transpose_x_1, transpose_y = attn_91_transpose_y_1, x = transpose_498_cast_fp16, y = mh_w_275_cast_fp16)[name = string("attn_91_cast_fp16")]; + tensor var_14065 = const()[name = string("op_14065"), val = tensor([1, 2048, 1, 1])]; + tensor input_393_cast_fp16 = reshape(shape = var_14065, x = attn_91_cast_fp16)[name = string("input_393_cast_fp16")]; + string obj_407_pad_type_0 = const()[name = string("obj_407_pad_type_0"), val = string("valid")]; + tensor obj_407_strides_0 = const()[name = string("obj_407_strides_0"), val = tensor([1, 1])]; + tensor obj_407_pad_0 = const()[name = string("obj_407_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_407_dilations_0 = const()[name = string("obj_407_dilations_0"), val = tensor([1, 1])]; + int32 obj_407_groups_0 = const()[name = string("obj_407_groups_0"), val = int32(1)]; + tensor obj_407_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_407_dilations_0, groups = obj_407_groups_0, pad = obj_407_pad_0, pad_type = obj_407_pad_type_0, strides = obj_407_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_393_cast_fp16)[name = string("obj_407_cast_fp16")]; + tensor inputs_383_cast_fp16 = add(x = inputs_377_cast_fp16, y = obj_407_cast_fp16)[name = string("inputs_383_cast_fp16")]; + tensor inputs_sq_383_cast_fp16 = mul(x = inputs_383_cast_fp16, y = inputs_383_cast_fp16)[name = string("inputs_sq_383_cast_fp16")]; + tensor variance_383_axes_0 = const()[name = string("variance_383_axes_0"), val = tensor([1])]; + bool variance_383_keep_dims_0 = const()[name = string("variance_383_keep_dims_0"), val = bool(true)]; + tensor variance_383_cast_fp16 = reduce_mean(axes = variance_383_axes_0, keep_dims = variance_383_keep_dims_0, x = inputs_sq_383_cast_fp16)[name = string("variance_383_cast_fp16")]; + fp16 var_14083_to_fp16 = const()[name = string("op_14083_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14084_cast_fp16 = add(x = variance_383_cast_fp16, y = var_14083_to_fp16)[name = string("op_14084_cast_fp16")]; + fp32 var_14085_epsilon_0 = const()[name = string("op_14085_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14085_cast_fp16 = rsqrt(epsilon = var_14085_epsilon_0, x = var_14084_cast_fp16)[name = string("op_14085_cast_fp16")]; + tensor hidden_states_473_cast_fp16 = mul(x = inputs_383_cast_fp16, y = var_14085_cast_fp16)[name = string("hidden_states_473_cast_fp16")]; + tensor input_395_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_473_cast_fp16)[name = string("input_395_cast_fp16")]; + string input_397_pad_type_0 = const()[name = string("input_397_pad_type_0"), val = string("valid")]; + tensor input_397_strides_0 = const()[name = string("input_397_strides_0"), val = tensor([1, 1])]; + tensor input_397_pad_0 = const()[name = string("input_397_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_397_dilations_0 = const()[name = string("input_397_dilations_0"), val = tensor([1, 1])]; + int32 input_397_groups_0 = const()[name = string("input_397_groups_0"), val = int32(1)]; + tensor input_397_cast_fp16 = conv(dilations = input_397_dilations_0, groups = input_397_groups_0, pad = input_397_pad_0, pad_type = input_397_pad_type_0, strides = input_397_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_395_cast_fp16)[name = string("input_397_cast_fp16")]; + tensor var_14099_cast_fp16 = silu(x = input_397_cast_fp16)[name = string("op_14099_cast_fp16")]; + string var_14105_pad_type_0 = const()[name = string("op_14105_pad_type_0"), val = string("valid")]; + tensor var_14105_strides_0 = const()[name = string("op_14105_strides_0"), val = tensor([1, 1])]; + tensor var_14105_pad_0 = const()[name = string("op_14105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14105_dilations_0 = const()[name = string("op_14105_dilations_0"), val = tensor([1, 1])]; + int32 var_14105_groups_0 = const()[name = string("op_14105_groups_0"), val = int32(1)]; + tensor var_14105_cast_fp16 = conv(dilations = var_14105_dilations_0, groups = var_14105_groups_0, pad = var_14105_pad_0, pad_type = var_14105_pad_type_0, strides = var_14105_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_395_cast_fp16)[name = string("op_14105_cast_fp16")]; + tensor input_399_cast_fp16 = mul(x = var_14099_cast_fp16, y = var_14105_cast_fp16)[name = string("input_399_cast_fp16")]; + string hidden_states_475_pad_type_0 = const()[name = string("hidden_states_475_pad_type_0"), val = string("valid")]; + tensor hidden_states_475_strides_0 = const()[name = string("hidden_states_475_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_475_pad_0 = const()[name = string("hidden_states_475_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_475_dilations_0 = const()[name = string("hidden_states_475_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_475_groups_0 = const()[name = string("hidden_states_475_groups_0"), val = int32(1)]; + tensor hidden_states_475_cast_fp16 = conv(dilations = hidden_states_475_dilations_0, groups = hidden_states_475_groups_0, pad = hidden_states_475_pad_0, pad_type = hidden_states_475_pad_type_0, strides = hidden_states_475_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_399_cast_fp16)[name = string("hidden_states_475_cast_fp16")]; + tensor inputs_385_cast_fp16 = add(x = inputs_383_cast_fp16, y = hidden_states_475_cast_fp16)[name = string("inputs_385_cast_fp16")]; + tensor obj_411_begin_0 = const()[name = string("obj_411_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_411_end_0 = const()[name = string("obj_411_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_411_end_mask_0 = const()[name = string("obj_411_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_411_cast_fp16 = slice_by_index(begin = obj_411_begin_0, end = obj_411_end_0, end_mask = obj_411_end_mask_0, x = key_caches_19_cast_fp16)[name = string("obj_411_cast_fp16")]; + tensor obj_413_begin_0 = const()[name = string("obj_413_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_413_end_0 = const()[name = string("obj_413_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_413_end_mask_0 = const()[name = string("obj_413_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_413_cast_fp16 = slice_by_index(begin = obj_413_begin_0, end = obj_413_end_0, end_mask = obj_413_end_mask_0, x = value_caches_19_cast_fp16)[name = string("obj_413_cast_fp16")]; + int32 var_14153 = const()[name = string("op_14153"), val = int32(3)]; + int32 var_14163 = const()[name = string("op_14163"), val = int32(-2)]; + tensor inputs_sq_385_cast_fp16 = mul(x = inputs_385_cast_fp16, y = inputs_385_cast_fp16)[name = string("inputs_sq_385_cast_fp16")]; + tensor variance_385_axes_0 = const()[name = string("variance_385_axes_0"), val = tensor([1])]; + bool variance_385_keep_dims_0 = const()[name = string("variance_385_keep_dims_0"), val = bool(true)]; + tensor variance_385_cast_fp16 = reduce_mean(axes = variance_385_axes_0, keep_dims = variance_385_keep_dims_0, x = inputs_sq_385_cast_fp16)[name = string("variance_385_cast_fp16")]; + fp16 var_14177_to_fp16 = const()[name = string("op_14177_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14178_cast_fp16 = add(x = variance_385_cast_fp16, y = var_14177_to_fp16)[name = string("op_14178_cast_fp16")]; + fp32 var_14179_epsilon_0 = const()[name = string("op_14179_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14179_cast_fp16 = rsqrt(epsilon = var_14179_epsilon_0, x = var_14178_cast_fp16)[name = string("op_14179_cast_fp16")]; + tensor hidden_states_477_cast_fp16 = mul(x = inputs_385_cast_fp16, y = var_14179_cast_fp16)[name = string("hidden_states_477_cast_fp16")]; + tensor obj_409_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_477_cast_fp16)[name = string("obj_409_cast_fp16")]; + string query_277_pad_type_0 = const()[name = string("query_277_pad_type_0"), val = string("valid")]; + tensor query_277_strides_0 = const()[name = string("query_277_strides_0"), val = tensor([1, 1])]; + tensor query_277_pad_0 = const()[name = string("query_277_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_277_dilations_0 = const()[name = string("query_277_dilations_0"), val = tensor([1, 1])]; + int32 query_277_groups_0 = const()[name = string("query_277_groups_0"), val = int32(1)]; + tensor query_277_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_277_dilations_0, groups = query_277_groups_0, pad = query_277_pad_0, pad_type = query_277_pad_type_0, strides = query_277_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_409_cast_fp16)[name = string("query_277_cast_fp16")]; + string current_key_185_pad_type_0 = const()[name = string("current_key_185_pad_type_0"), val = string("valid")]; + tensor current_key_185_strides_0 = const()[name = string("current_key_185_strides_0"), val = tensor([1, 1])]; + tensor current_key_185_pad_0 = const()[name = string("current_key_185_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_185_dilations_0 = const()[name = string("current_key_185_dilations_0"), val = tensor([1, 1])]; + int32 current_key_185_groups_0 = const()[name = string("current_key_185_groups_0"), val = int32(1)]; + tensor current_key_185_cast_fp16 = conv(dilations = current_key_185_dilations_0, groups = current_key_185_groups_0, pad = current_key_185_pad_0, pad_type = current_key_185_pad_type_0, strides = current_key_185_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_409_cast_fp16)[name = string("current_key_185_cast_fp16")]; + string current_value_93_pad_type_0 = const()[name = string("current_value_93_pad_type_0"), val = string("valid")]; + tensor current_value_93_strides_0 = const()[name = string("current_value_93_strides_0"), val = tensor([1, 1])]; + tensor current_value_93_pad_0 = const()[name = string("current_value_93_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_93_dilations_0 = const()[name = string("current_value_93_dilations_0"), val = tensor([1, 1])]; + int32 current_value_93_groups_0 = const()[name = string("current_value_93_groups_0"), val = int32(1)]; + tensor current_value_93_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_93_dilations_0, groups = current_value_93_groups_0, pad = current_value_93_pad_0, pad_type = current_value_93_pad_type_0, strides = current_value_93_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_409_cast_fp16)[name = string("current_value_93_cast_fp16")]; + tensor var_14216 = const()[name = string("op_14216"), val = tensor([16, 128, 1, 1])]; + tensor inputs_387_cast_fp16 = reshape(shape = var_14216, x = query_277_cast_fp16)[name = string("inputs_387_cast_fp16")]; + tensor inputs_sq_387_cast_fp16 = mul(x = inputs_387_cast_fp16, y = inputs_387_cast_fp16)[name = string("inputs_sq_387_cast_fp16")]; + tensor variance_387_axes_0 = const()[name = string("variance_387_axes_0"), val = tensor([1])]; + bool variance_387_keep_dims_0 = const()[name = string("variance_387_keep_dims_0"), val = bool(true)]; + tensor variance_387_cast_fp16 = reduce_mean(axes = variance_387_axes_0, keep_dims = variance_387_keep_dims_0, x = inputs_sq_387_cast_fp16)[name = string("variance_387_cast_fp16")]; + fp16 var_14222_to_fp16 = const()[name = string("op_14222_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14223_cast_fp16 = add(x = variance_387_cast_fp16, y = var_14222_to_fp16)[name = string("op_14223_cast_fp16")]; + fp32 var_14224_epsilon_0 = const()[name = string("op_14224_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14224_cast_fp16 = rsqrt(epsilon = var_14224_epsilon_0, x = var_14223_cast_fp16)[name = string("op_14224_cast_fp16")]; + tensor hidden_states_479_cast_fp16 = mul(x = inputs_387_cast_fp16, y = var_14224_cast_fp16)[name = string("hidden_states_479_cast_fp16")]; + tensor query_normed_93_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_479_cast_fp16)[name = string("query_normed_93_cast_fp16")]; + tensor var_14232 = const()[name = string("op_14232"), val = tensor([8, 128, 1, 1])]; + tensor inputs_389_cast_fp16 = reshape(shape = var_14232, x = current_key_185_cast_fp16)[name = string("inputs_389_cast_fp16")]; + tensor inputs_sq_389_cast_fp16 = mul(x = inputs_389_cast_fp16, y = inputs_389_cast_fp16)[name = string("inputs_sq_389_cast_fp16")]; + tensor variance_389_axes_0 = const()[name = string("variance_389_axes_0"), val = tensor([1])]; + bool variance_389_keep_dims_0 = const()[name = string("variance_389_keep_dims_0"), val = bool(true)]; + tensor variance_389_cast_fp16 = reduce_mean(axes = variance_389_axes_0, keep_dims = variance_389_keep_dims_0, x = inputs_sq_389_cast_fp16)[name = string("variance_389_cast_fp16")]; + fp16 var_14238_to_fp16 = const()[name = string("op_14238_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14239_cast_fp16 = add(x = variance_389_cast_fp16, y = var_14238_to_fp16)[name = string("op_14239_cast_fp16")]; + fp32 var_14240_epsilon_0 = const()[name = string("op_14240_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14240_cast_fp16 = rsqrt(epsilon = var_14240_epsilon_0, x = var_14239_cast_fp16)[name = string("op_14240_cast_fp16")]; + tensor hidden_states_481_cast_fp16 = mul(x = inputs_389_cast_fp16, y = var_14240_cast_fp16)[name = string("hidden_states_481_cast_fp16")]; + tensor current_key_normed_93_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_481_cast_fp16)[name = string("current_key_normed_93_cast_fp16")]; + tensor var_14258 = const()[name = string("op_14258"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_369_cast_fp16 = reshape(shape = var_14258, x = query_normed_93_cast_fp16)[name = string("mh_q_369_cast_fp16")]; + tensor var_14260 = const()[name = string("op_14260"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_369_cast_fp16 = reshape(shape = var_14260, x = current_key_normed_93_cast_fp16)[name = string("mh_k_369_cast_fp16")]; + tensor var_14264_cast_fp16 = mul(x = mh_q_369_cast_fp16, y = cos_91_to_fp16)[name = string("op_14264_cast_fp16")]; + tensor var_14269_begin_0 = const()[name = string("op_14269_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14269_end_0 = const()[name = string("op_14269_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_14269_end_mask_0 = const()[name = string("op_14269_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_14269_cast_fp16 = slice_by_index(begin = var_14269_begin_0, end = var_14269_end_0, end_mask = var_14269_end_mask_0, x = mh_q_369_cast_fp16)[name = string("op_14269_cast_fp16")]; + tensor var_14275_begin_0 = const()[name = string("op_14275_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_14275_end_0 = const()[name = string("op_14275_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_14275_end_mask_0 = const()[name = string("op_14275_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_14275_cast_fp16 = slice_by_index(begin = var_14275_begin_0, end = var_14275_end_0, end_mask = var_14275_end_mask_0, x = mh_q_369_cast_fp16)[name = string("op_14275_cast_fp16")]; + fp16 const_943_promoted_to_fp16 = const()[name = string("const_943_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14277_cast_fp16 = mul(x = var_14275_cast_fp16, y = const_943_promoted_to_fp16)[name = string("op_14277_cast_fp16")]; + bool var_14279_interleave_0 = const()[name = string("op_14279_interleave_0"), val = bool(false)]; + tensor var_14279_cast_fp16 = concat(axis = var_14163, interleave = var_14279_interleave_0, values = (var_14277_cast_fp16, var_14269_cast_fp16))[name = string("op_14279_cast_fp16")]; + tensor var_14280_cast_fp16 = mul(x = var_14279_cast_fp16, y = sin_91_to_fp16)[name = string("op_14280_cast_fp16")]; + tensor mh_q_371_cast_fp16 = add(x = var_14264_cast_fp16, y = var_14280_cast_fp16)[name = string("mh_q_371_cast_fp16")]; + tensor var_14282_cast_fp16 = mul(x = mh_k_369_cast_fp16, y = cos_91_to_fp16)[name = string("op_14282_cast_fp16")]; + tensor var_14287_begin_0 = const()[name = string("op_14287_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14287_end_0 = const()[name = string("op_14287_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_14287_end_mask_0 = const()[name = string("op_14287_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_14287_cast_fp16 = slice_by_index(begin = var_14287_begin_0, end = var_14287_end_0, end_mask = var_14287_end_mask_0, x = mh_k_369_cast_fp16)[name = string("op_14287_cast_fp16")]; + tensor var_14293_begin_0 = const()[name = string("op_14293_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_14293_end_0 = const()[name = string("op_14293_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_14293_end_mask_0 = const()[name = string("op_14293_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_14293_cast_fp16 = slice_by_index(begin = var_14293_begin_0, end = var_14293_end_0, end_mask = var_14293_end_mask_0, x = mh_k_369_cast_fp16)[name = string("op_14293_cast_fp16")]; + fp16 const_946_promoted_to_fp16 = const()[name = string("const_946_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14295_cast_fp16 = mul(x = var_14293_cast_fp16, y = const_946_promoted_to_fp16)[name = string("op_14295_cast_fp16")]; + bool var_14297_interleave_0 = const()[name = string("op_14297_interleave_0"), val = bool(false)]; + tensor var_14297_cast_fp16 = concat(axis = var_14163, interleave = var_14297_interleave_0, values = (var_14295_cast_fp16, var_14287_cast_fp16))[name = string("op_14297_cast_fp16")]; + tensor var_14298_cast_fp16 = mul(x = var_14297_cast_fp16, y = sin_91_to_fp16)[name = string("op_14298_cast_fp16")]; + tensor mh_k_371_cast_fp16 = add(x = var_14282_cast_fp16, y = var_14298_cast_fp16)[name = string("mh_k_371_cast_fp16")]; + tensor var_14302 = const()[name = string("op_14302"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_187_cast_fp16 = reshape(shape = var_14302, x = mh_k_371_cast_fp16)[name = string("current_key_187_cast_fp16")]; + tensor var_14308_to_fp16 = const()[name = string("op_14308_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198400)))]; + tensor var_14309_cast_fp16 = mul(x = obj_411_cast_fp16, y = var_14308_to_fp16)[name = string("op_14309_cast_fp16")]; + tensor var_14306_to_fp16 = const()[name = string("op_14306_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198528)))]; + tensor var_14310_cast_fp16 = mul(x = current_key_187_cast_fp16, y = var_14306_to_fp16)[name = string("op_14310_cast_fp16")]; + tensor key_187_cast_fp16 = add(x = var_14309_cast_fp16, y = var_14310_cast_fp16)[name = string("key_187_cast_fp16")]; + tensor var_14312_to_fp16 = const()[name = string("op_14312_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198400)))]; + tensor var_14313_cast_fp16 = mul(x = obj_413_cast_fp16, y = var_14312_to_fp16)[name = string("op_14313_cast_fp16")]; + tensor var_14314_cast_fp16 = mul(x = current_value_93_cast_fp16, y = var_14306_to_fp16)[name = string("op_14314_cast_fp16")]; + tensor value_93_cast_fp16 = add(x = var_14313_cast_fp16, y = var_14314_cast_fp16)[name = string("value_93_cast_fp16")]; + fp16 var_14321_to_fp16 = const()[name = string("op_14321_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_375_cast_fp16 = mul(x = mh_q_371_cast_fp16, y = var_14321_to_fp16)[name = string("mh_q_375_cast_fp16")]; + tensor var_14323 = const()[name = string("op_14323"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_373_cast_fp16 = reshape(shape = var_14323, x = key_187_cast_fp16)[name = string("mh_k_373_cast_fp16")]; + tensor var_14325 = const()[name = string("op_14325"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_185_cast_fp16 = reshape(shape = var_14325, x = value_93_cast_fp16)[name = string("mh_v_185_cast_fp16")]; + tensor transpose_184_perm_0 = const()[name = string("transpose_184_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_92_reps_0 = const()[name = string("tile_92_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_184_cast_fp16 = transpose(perm = transpose_184_perm_0, x = mh_k_373_cast_fp16)[name = string("transpose_203")]; + tensor tile_92_cast_fp16 = tile(reps = tile_92_reps_0, x = transpose_184_cast_fp16)[name = string("tile_92_cast_fp16")]; + tensor concat_232 = const()[name = string("concat_232"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_184_cast_fp16 = reshape(shape = concat_232, x = tile_92_cast_fp16)[name = string("reshape_184_cast_fp16")]; + tensor transpose_185_perm_0 = const()[name = string("transpose_185_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_233 = const()[name = string("concat_233"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_185_cast_fp16 = transpose(perm = transpose_185_perm_0, x = reshape_184_cast_fp16)[name = string("transpose_202")]; + tensor reshape_185_cast_fp16 = reshape(shape = concat_233, x = transpose_185_cast_fp16)[name = string("reshape_185_cast_fp16")]; + tensor transpose_186_perm_0 = const()[name = string("transpose_186_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_93_reps_0 = const()[name = string("tile_93_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_186_cast_fp16 = transpose(perm = transpose_186_perm_0, x = mh_v_185_cast_fp16)[name = string("transpose_201")]; + tensor tile_93_cast_fp16 = tile(reps = tile_93_reps_0, x = transpose_186_cast_fp16)[name = string("tile_93_cast_fp16")]; + tensor concat_234 = const()[name = string("concat_234"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_186_cast_fp16 = reshape(shape = concat_234, x = tile_93_cast_fp16)[name = string("reshape_186_cast_fp16")]; + tensor transpose_187_perm_0 = const()[name = string("transpose_187_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_235 = const()[name = string("concat_235"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_187_cast_fp16 = transpose(perm = transpose_187_perm_0, x = reshape_186_cast_fp16)[name = string("transpose_200")]; + tensor reshape_187_cast_fp16 = reshape(shape = concat_235, x = transpose_187_cast_fp16)[name = string("reshape_187_cast_fp16")]; + tensor transpose_501_perm_0 = const()[name = string("transpose_501_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_277_transpose_x_1 = const()[name = string("mh_w_277_transpose_x_1"), val = bool(true)]; + bool mh_w_277_transpose_y_1 = const()[name = string("mh_w_277_transpose_y_1"), val = bool(false)]; + tensor transpose_501_cast_fp16 = transpose(perm = transpose_501_perm_0, x = reshape_185_cast_fp16)[name = string("transpose_199")]; + tensor mh_w_277_cast_fp16 = matmul(transpose_x = mh_w_277_transpose_x_1, transpose_y = mh_w_277_transpose_y_1, x = mh_q_375_cast_fp16, y = transpose_501_cast_fp16)[name = string("mh_w_277_cast_fp16")]; + tensor var_14333_to_fp16 = const()[name = string("op_14333_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198656)))]; + tensor mh_w_279_cast_fp16 = add(x = mh_w_277_cast_fp16, y = var_14333_to_fp16)[name = string("mh_w_279_cast_fp16")]; + tensor mh_w_281_cast_fp16 = softmax(axis = var_14153, x = mh_w_279_cast_fp16)[name = string("mh_w_281_cast_fp16")]; + tensor transpose_502_perm_0 = const()[name = string("transpose_502_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_93_transpose_x_1 = const()[name = string("attn_93_transpose_x_1"), val = bool(false)]; + bool attn_93_transpose_y_1 = const()[name = string("attn_93_transpose_y_1"), val = bool(true)]; + tensor transpose_502_cast_fp16 = transpose(perm = transpose_502_perm_0, x = reshape_187_cast_fp16)[name = string("transpose_198")]; + tensor attn_93_cast_fp16 = matmul(transpose_x = attn_93_transpose_x_1, transpose_y = attn_93_transpose_y_1, x = transpose_502_cast_fp16, y = mh_w_281_cast_fp16)[name = string("attn_93_cast_fp16")]; + tensor var_14339 = const()[name = string("op_14339"), val = tensor([1, 2048, 1, 1])]; + tensor input_401_cast_fp16 = reshape(shape = var_14339, x = attn_93_cast_fp16)[name = string("input_401_cast_fp16")]; + string obj_415_pad_type_0 = const()[name = string("obj_415_pad_type_0"), val = string("valid")]; + tensor obj_415_strides_0 = const()[name = string("obj_415_strides_0"), val = tensor([1, 1])]; + tensor obj_415_pad_0 = const()[name = string("obj_415_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_415_dilations_0 = const()[name = string("obj_415_dilations_0"), val = tensor([1, 1])]; + int32 obj_415_groups_0 = const()[name = string("obj_415_groups_0"), val = int32(1)]; + tensor obj_415_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_415_dilations_0, groups = obj_415_groups_0, pad = obj_415_pad_0, pad_type = obj_415_pad_type_0, strides = obj_415_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_401_cast_fp16)[name = string("obj_415_cast_fp16")]; + tensor inputs_391_cast_fp16 = add(x = inputs_385_cast_fp16, y = obj_415_cast_fp16)[name = string("inputs_391_cast_fp16")]; + tensor inputs_sq_391_cast_fp16 = mul(x = inputs_391_cast_fp16, y = inputs_391_cast_fp16)[name = string("inputs_sq_391_cast_fp16")]; + tensor variance_391_axes_0 = const()[name = string("variance_391_axes_0"), val = tensor([1])]; + bool variance_391_keep_dims_0 = const()[name = string("variance_391_keep_dims_0"), val = bool(true)]; + tensor variance_391_cast_fp16 = reduce_mean(axes = variance_391_axes_0, keep_dims = variance_391_keep_dims_0, x = inputs_sq_391_cast_fp16)[name = string("variance_391_cast_fp16")]; + fp16 var_14357_to_fp16 = const()[name = string("op_14357_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14358_cast_fp16 = add(x = variance_391_cast_fp16, y = var_14357_to_fp16)[name = string("op_14358_cast_fp16")]; + fp32 var_14359_epsilon_0 = const()[name = string("op_14359_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14359_cast_fp16 = rsqrt(epsilon = var_14359_epsilon_0, x = var_14358_cast_fp16)[name = string("op_14359_cast_fp16")]; + tensor hidden_states_483_cast_fp16 = mul(x = inputs_391_cast_fp16, y = var_14359_cast_fp16)[name = string("hidden_states_483_cast_fp16")]; + tensor input_403_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_483_cast_fp16)[name = string("input_403_cast_fp16")]; + string input_405_pad_type_0 = const()[name = string("input_405_pad_type_0"), val = string("valid")]; + tensor input_405_strides_0 = const()[name = string("input_405_strides_0"), val = tensor([1, 1])]; + tensor input_405_pad_0 = const()[name = string("input_405_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_405_dilations_0 = const()[name = string("input_405_dilations_0"), val = tensor([1, 1])]; + int32 input_405_groups_0 = const()[name = string("input_405_groups_0"), val = int32(1)]; + tensor input_405_cast_fp16 = conv(dilations = input_405_dilations_0, groups = input_405_groups_0, pad = input_405_pad_0, pad_type = input_405_pad_type_0, strides = input_405_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_403_cast_fp16)[name = string("input_405_cast_fp16")]; + tensor var_14373_cast_fp16 = silu(x = input_405_cast_fp16)[name = string("op_14373_cast_fp16")]; + string var_14379_pad_type_0 = const()[name = string("op_14379_pad_type_0"), val = string("valid")]; + tensor var_14379_strides_0 = const()[name = string("op_14379_strides_0"), val = tensor([1, 1])]; + tensor var_14379_pad_0 = const()[name = string("op_14379_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14379_dilations_0 = const()[name = string("op_14379_dilations_0"), val = tensor([1, 1])]; + int32 var_14379_groups_0 = const()[name = string("op_14379_groups_0"), val = int32(1)]; + tensor var_14379_cast_fp16 = conv(dilations = var_14379_dilations_0, groups = var_14379_groups_0, pad = var_14379_pad_0, pad_type = var_14379_pad_type_0, strides = var_14379_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_403_cast_fp16)[name = string("op_14379_cast_fp16")]; + tensor input_407_cast_fp16 = mul(x = var_14373_cast_fp16, y = var_14379_cast_fp16)[name = string("input_407_cast_fp16")]; + string hidden_states_485_pad_type_0 = const()[name = string("hidden_states_485_pad_type_0"), val = string("valid")]; + tensor hidden_states_485_strides_0 = const()[name = string("hidden_states_485_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_485_pad_0 = const()[name = string("hidden_states_485_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_485_dilations_0 = const()[name = string("hidden_states_485_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_485_groups_0 = const()[name = string("hidden_states_485_groups_0"), val = int32(1)]; + tensor hidden_states_485_cast_fp16 = conv(dilations = hidden_states_485_dilations_0, groups = hidden_states_485_groups_0, pad = hidden_states_485_pad_0, pad_type = hidden_states_485_pad_type_0, strides = hidden_states_485_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_407_cast_fp16)[name = string("hidden_states_485_cast_fp16")]; + tensor inputs_393_cast_fp16 = add(x = inputs_391_cast_fp16, y = hidden_states_485_cast_fp16)[name = string("inputs_393_cast_fp16")]; + tensor obj_419_begin_0 = const()[name = string("obj_419_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_419_end_0 = const()[name = string("obj_419_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_419_end_mask_0 = const()[name = string("obj_419_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_419_cast_fp16 = slice_by_index(begin = obj_419_begin_0, end = obj_419_end_0, end_mask = obj_419_end_mask_0, x = key_caches_19_cast_fp16)[name = string("obj_419_cast_fp16")]; + tensor obj_421_begin_0 = const()[name = string("obj_421_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_421_end_0 = const()[name = string("obj_421_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_421_end_mask_0 = const()[name = string("obj_421_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_421_cast_fp16 = slice_by_index(begin = obj_421_begin_0, end = obj_421_end_0, end_mask = obj_421_end_mask_0, x = value_caches_19_cast_fp16)[name = string("obj_421_cast_fp16")]; + int32 var_14427 = const()[name = string("op_14427"), val = int32(3)]; + int32 var_14437 = const()[name = string("op_14437"), val = int32(-2)]; + tensor inputs_sq_393_cast_fp16 = mul(x = inputs_393_cast_fp16, y = inputs_393_cast_fp16)[name = string("inputs_sq_393_cast_fp16")]; + tensor variance_393_axes_0 = const()[name = string("variance_393_axes_0"), val = tensor([1])]; + bool variance_393_keep_dims_0 = const()[name = string("variance_393_keep_dims_0"), val = bool(true)]; + tensor variance_393_cast_fp16 = reduce_mean(axes = variance_393_axes_0, keep_dims = variance_393_keep_dims_0, x = inputs_sq_393_cast_fp16)[name = string("variance_393_cast_fp16")]; + fp16 var_14451_to_fp16 = const()[name = string("op_14451_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14452_cast_fp16 = add(x = variance_393_cast_fp16, y = var_14451_to_fp16)[name = string("op_14452_cast_fp16")]; + fp32 var_14453_epsilon_0 = const()[name = string("op_14453_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14453_cast_fp16 = rsqrt(epsilon = var_14453_epsilon_0, x = var_14452_cast_fp16)[name = string("op_14453_cast_fp16")]; + tensor hidden_states_487_cast_fp16 = mul(x = inputs_393_cast_fp16, y = var_14453_cast_fp16)[name = string("hidden_states_487_cast_fp16")]; + tensor obj_417_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_487_cast_fp16)[name = string("obj_417_cast_fp16")]; + string query_283_pad_type_0 = const()[name = string("query_283_pad_type_0"), val = string("valid")]; + tensor query_283_strides_0 = const()[name = string("query_283_strides_0"), val = tensor([1, 1])]; + tensor query_283_pad_0 = const()[name = string("query_283_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_283_dilations_0 = const()[name = string("query_283_dilations_0"), val = tensor([1, 1])]; + int32 query_283_groups_0 = const()[name = string("query_283_groups_0"), val = int32(1)]; + tensor query_283_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_283_dilations_0, groups = query_283_groups_0, pad = query_283_pad_0, pad_type = query_283_pad_type_0, strides = query_283_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_417_cast_fp16)[name = string("query_283_cast_fp16")]; + string current_key_189_pad_type_0 = const()[name = string("current_key_189_pad_type_0"), val = string("valid")]; + tensor current_key_189_strides_0 = const()[name = string("current_key_189_strides_0"), val = tensor([1, 1])]; + tensor current_key_189_pad_0 = const()[name = string("current_key_189_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_189_dilations_0 = const()[name = string("current_key_189_dilations_0"), val = tensor([1, 1])]; + int32 current_key_189_groups_0 = const()[name = string("current_key_189_groups_0"), val = int32(1)]; + tensor current_key_189_cast_fp16 = conv(dilations = current_key_189_dilations_0, groups = current_key_189_groups_0, pad = current_key_189_pad_0, pad_type = current_key_189_pad_type_0, strides = current_key_189_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_417_cast_fp16)[name = string("current_key_189_cast_fp16")]; + string current_value_95_pad_type_0 = const()[name = string("current_value_95_pad_type_0"), val = string("valid")]; + tensor current_value_95_strides_0 = const()[name = string("current_value_95_strides_0"), val = tensor([1, 1])]; + tensor current_value_95_pad_0 = const()[name = string("current_value_95_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_95_dilations_0 = const()[name = string("current_value_95_dilations_0"), val = tensor([1, 1])]; + int32 current_value_95_groups_0 = const()[name = string("current_value_95_groups_0"), val = int32(1)]; + tensor current_value_95_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_95_dilations_0, groups = current_value_95_groups_0, pad = current_value_95_pad_0, pad_type = current_value_95_pad_type_0, strides = current_value_95_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_417_cast_fp16)[name = string("current_value_95_cast_fp16")]; + tensor var_14490 = const()[name = string("op_14490"), val = tensor([16, 128, 1, 1])]; + tensor inputs_395_cast_fp16 = reshape(shape = var_14490, x = query_283_cast_fp16)[name = string("inputs_395_cast_fp16")]; + tensor inputs_sq_395_cast_fp16 = mul(x = inputs_395_cast_fp16, y = inputs_395_cast_fp16)[name = string("inputs_sq_395_cast_fp16")]; + tensor variance_395_axes_0 = const()[name = string("variance_395_axes_0"), val = tensor([1])]; + bool variance_395_keep_dims_0 = const()[name = string("variance_395_keep_dims_0"), val = bool(true)]; + tensor variance_395_cast_fp16 = reduce_mean(axes = variance_395_axes_0, keep_dims = variance_395_keep_dims_0, x = inputs_sq_395_cast_fp16)[name = string("variance_395_cast_fp16")]; + fp16 var_14496_to_fp16 = const()[name = string("op_14496_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14497_cast_fp16 = add(x = variance_395_cast_fp16, y = var_14496_to_fp16)[name = string("op_14497_cast_fp16")]; + fp32 var_14498_epsilon_0 = const()[name = string("op_14498_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14498_cast_fp16 = rsqrt(epsilon = var_14498_epsilon_0, x = var_14497_cast_fp16)[name = string("op_14498_cast_fp16")]; + tensor hidden_states_489_cast_fp16 = mul(x = inputs_395_cast_fp16, y = var_14498_cast_fp16)[name = string("hidden_states_489_cast_fp16")]; + tensor query_normed_95_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_489_cast_fp16)[name = string("query_normed_95_cast_fp16")]; + tensor var_14506 = const()[name = string("op_14506"), val = tensor([8, 128, 1, 1])]; + tensor inputs_397_cast_fp16 = reshape(shape = var_14506, x = current_key_189_cast_fp16)[name = string("inputs_397_cast_fp16")]; + tensor inputs_sq_397_cast_fp16 = mul(x = inputs_397_cast_fp16, y = inputs_397_cast_fp16)[name = string("inputs_sq_397_cast_fp16")]; + tensor variance_397_axes_0 = const()[name = string("variance_397_axes_0"), val = tensor([1])]; + bool variance_397_keep_dims_0 = const()[name = string("variance_397_keep_dims_0"), val = bool(true)]; + tensor variance_397_cast_fp16 = reduce_mean(axes = variance_397_axes_0, keep_dims = variance_397_keep_dims_0, x = inputs_sq_397_cast_fp16)[name = string("variance_397_cast_fp16")]; + fp16 var_14512_to_fp16 = const()[name = string("op_14512_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14513_cast_fp16 = add(x = variance_397_cast_fp16, y = var_14512_to_fp16)[name = string("op_14513_cast_fp16")]; + fp32 var_14514_epsilon_0 = const()[name = string("op_14514_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14514_cast_fp16 = rsqrt(epsilon = var_14514_epsilon_0, x = var_14513_cast_fp16)[name = string("op_14514_cast_fp16")]; + tensor hidden_states_491_cast_fp16 = mul(x = inputs_397_cast_fp16, y = var_14514_cast_fp16)[name = string("hidden_states_491_cast_fp16")]; + tensor current_key_normed_95_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_491_cast_fp16)[name = string("current_key_normed_95_cast_fp16")]; + tensor var_14532 = const()[name = string("op_14532"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_377_cast_fp16 = reshape(shape = var_14532, x = query_normed_95_cast_fp16)[name = string("mh_q_377_cast_fp16")]; + tensor var_14534 = const()[name = string("op_14534"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_377_cast_fp16 = reshape(shape = var_14534, x = current_key_normed_95_cast_fp16)[name = string("mh_k_377_cast_fp16")]; + tensor var_14538_cast_fp16 = mul(x = mh_q_377_cast_fp16, y = cos_91_to_fp16)[name = string("op_14538_cast_fp16")]; + tensor var_14543_begin_0 = const()[name = string("op_14543_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14543_end_0 = const()[name = string("op_14543_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_14543_end_mask_0 = const()[name = string("op_14543_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_14543_cast_fp16 = slice_by_index(begin = var_14543_begin_0, end = var_14543_end_0, end_mask = var_14543_end_mask_0, x = mh_q_377_cast_fp16)[name = string("op_14543_cast_fp16")]; + tensor var_14549_begin_0 = const()[name = string("op_14549_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_14549_end_0 = const()[name = string("op_14549_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_14549_end_mask_0 = const()[name = string("op_14549_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_14549_cast_fp16 = slice_by_index(begin = var_14549_begin_0, end = var_14549_end_0, end_mask = var_14549_end_mask_0, x = mh_q_377_cast_fp16)[name = string("op_14549_cast_fp16")]; + fp16 const_963_promoted_to_fp16 = const()[name = string("const_963_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14551_cast_fp16 = mul(x = var_14549_cast_fp16, y = const_963_promoted_to_fp16)[name = string("op_14551_cast_fp16")]; + bool var_14553_interleave_0 = const()[name = string("op_14553_interleave_0"), val = bool(false)]; + tensor var_14553_cast_fp16 = concat(axis = var_14437, interleave = var_14553_interleave_0, values = (var_14551_cast_fp16, var_14543_cast_fp16))[name = string("op_14553_cast_fp16")]; + tensor var_14554_cast_fp16 = mul(x = var_14553_cast_fp16, y = sin_91_to_fp16)[name = string("op_14554_cast_fp16")]; + tensor mh_q_379_cast_fp16 = add(x = var_14538_cast_fp16, y = var_14554_cast_fp16)[name = string("mh_q_379_cast_fp16")]; + tensor var_14556_cast_fp16 = mul(x = mh_k_377_cast_fp16, y = cos_91_to_fp16)[name = string("op_14556_cast_fp16")]; + tensor var_14561_begin_0 = const()[name = string("op_14561_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14561_end_0 = const()[name = string("op_14561_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_14561_end_mask_0 = const()[name = string("op_14561_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_14561_cast_fp16 = slice_by_index(begin = var_14561_begin_0, end = var_14561_end_0, end_mask = var_14561_end_mask_0, x = mh_k_377_cast_fp16)[name = string("op_14561_cast_fp16")]; + tensor var_14567_begin_0 = const()[name = string("op_14567_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_14567_end_0 = const()[name = string("op_14567_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_14567_end_mask_0 = const()[name = string("op_14567_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_14567_cast_fp16 = slice_by_index(begin = var_14567_begin_0, end = var_14567_end_0, end_mask = var_14567_end_mask_0, x = mh_k_377_cast_fp16)[name = string("op_14567_cast_fp16")]; + fp16 const_966_promoted_to_fp16 = const()[name = string("const_966_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14569_cast_fp16 = mul(x = var_14567_cast_fp16, y = const_966_promoted_to_fp16)[name = string("op_14569_cast_fp16")]; + bool var_14571_interleave_0 = const()[name = string("op_14571_interleave_0"), val = bool(false)]; + tensor var_14571_cast_fp16 = concat(axis = var_14437, interleave = var_14571_interleave_0, values = (var_14569_cast_fp16, var_14561_cast_fp16))[name = string("op_14571_cast_fp16")]; + tensor var_14572_cast_fp16 = mul(x = var_14571_cast_fp16, y = sin_91_to_fp16)[name = string("op_14572_cast_fp16")]; + tensor mh_k_379_cast_fp16 = add(x = var_14556_cast_fp16, y = var_14572_cast_fp16)[name = string("mh_k_379_cast_fp16")]; + tensor var_14576 = const()[name = string("op_14576"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_191_cast_fp16 = reshape(shape = var_14576, x = mh_k_379_cast_fp16)[name = string("current_key_191_cast_fp16")]; + tensor var_14582_to_fp16 = const()[name = string("op_14582_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198400)))]; + tensor var_14583_cast_fp16 = mul(x = obj_419_cast_fp16, y = var_14582_to_fp16)[name = string("op_14583_cast_fp16")]; + tensor var_14580_to_fp16 = const()[name = string("op_14580_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198528)))]; + tensor var_14584_cast_fp16 = mul(x = current_key_191_cast_fp16, y = var_14580_to_fp16)[name = string("op_14584_cast_fp16")]; + tensor key_191_cast_fp16 = add(x = var_14583_cast_fp16, y = var_14584_cast_fp16)[name = string("key_191_cast_fp16")]; + tensor var_14586_to_fp16 = const()[name = string("op_14586_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198400)))]; + tensor var_14587_cast_fp16 = mul(x = obj_421_cast_fp16, y = var_14586_to_fp16)[name = string("op_14587_cast_fp16")]; + tensor var_14588_cast_fp16 = mul(x = current_value_95_cast_fp16, y = var_14580_to_fp16)[name = string("op_14588_cast_fp16")]; + tensor value_95_cast_fp16 = add(x = var_14587_cast_fp16, y = var_14588_cast_fp16)[name = string("value_95_cast_fp16")]; + fp16 var_14595_to_fp16 = const()[name = string("op_14595_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_383_cast_fp16 = mul(x = mh_q_379_cast_fp16, y = var_14595_to_fp16)[name = string("mh_q_383_cast_fp16")]; + tensor var_14597 = const()[name = string("op_14597"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_381_cast_fp16 = reshape(shape = var_14597, x = key_191_cast_fp16)[name = string("mh_k_381_cast_fp16")]; + tensor var_14599 = const()[name = string("op_14599"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_189_cast_fp16 = reshape(shape = var_14599, x = value_95_cast_fp16)[name = string("mh_v_189_cast_fp16")]; + tensor transpose_188_perm_0 = const()[name = string("transpose_188_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_94_reps_0 = const()[name = string("tile_94_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_188_cast_fp16 = transpose(perm = transpose_188_perm_0, x = mh_k_381_cast_fp16)[name = string("transpose_197")]; + tensor tile_94_cast_fp16 = tile(reps = tile_94_reps_0, x = transpose_188_cast_fp16)[name = string("tile_94_cast_fp16")]; + tensor concat_236 = const()[name = string("concat_236"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_188_cast_fp16 = reshape(shape = concat_236, x = tile_94_cast_fp16)[name = string("reshape_188_cast_fp16")]; + tensor transpose_189_perm_0 = const()[name = string("transpose_189_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_237 = const()[name = string("concat_237"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_189_cast_fp16 = transpose(perm = transpose_189_perm_0, x = reshape_188_cast_fp16)[name = string("transpose_196")]; + tensor reshape_189_cast_fp16 = reshape(shape = concat_237, x = transpose_189_cast_fp16)[name = string("reshape_189_cast_fp16")]; + tensor transpose_190_perm_0 = const()[name = string("transpose_190_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_95_reps_0 = const()[name = string("tile_95_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_190_cast_fp16 = transpose(perm = transpose_190_perm_0, x = mh_v_189_cast_fp16)[name = string("transpose_195")]; + tensor tile_95_cast_fp16 = tile(reps = tile_95_reps_0, x = transpose_190_cast_fp16)[name = string("tile_95_cast_fp16")]; + tensor concat_238 = const()[name = string("concat_238"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_190_cast_fp16 = reshape(shape = concat_238, x = tile_95_cast_fp16)[name = string("reshape_190_cast_fp16")]; + tensor transpose_191_perm_0 = const()[name = string("transpose_191_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_239 = const()[name = string("concat_239"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_191_cast_fp16 = transpose(perm = transpose_191_perm_0, x = reshape_190_cast_fp16)[name = string("transpose_194")]; + tensor reshape_191_cast_fp16 = reshape(shape = concat_239, x = transpose_191_cast_fp16)[name = string("reshape_191_cast_fp16")]; + tensor transpose_505_perm_0 = const()[name = string("transpose_505_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_283_transpose_x_1 = const()[name = string("mh_w_283_transpose_x_1"), val = bool(true)]; + bool mh_w_283_transpose_y_1 = const()[name = string("mh_w_283_transpose_y_1"), val = bool(false)]; + tensor transpose_505_cast_fp16 = transpose(perm = transpose_505_perm_0, x = reshape_189_cast_fp16)[name = string("transpose_193")]; + tensor mh_w_283_cast_fp16 = matmul(transpose_x = mh_w_283_transpose_x_1, transpose_y = mh_w_283_transpose_y_1, x = mh_q_383_cast_fp16, y = transpose_505_cast_fp16)[name = string("mh_w_283_cast_fp16")]; + tensor var_14607_to_fp16 = const()[name = string("op_14607_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198656)))]; + tensor mh_w_285_cast_fp16 = add(x = mh_w_283_cast_fp16, y = var_14607_to_fp16)[name = string("mh_w_285_cast_fp16")]; + tensor mh_w_287_cast_fp16 = softmax(axis = var_14427, x = mh_w_285_cast_fp16)[name = string("mh_w_287_cast_fp16")]; + tensor transpose_506_perm_0 = const()[name = string("transpose_506_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_95_transpose_x_1 = const()[name = string("attn_95_transpose_x_1"), val = bool(false)]; + bool attn_95_transpose_y_1 = const()[name = string("attn_95_transpose_y_1"), val = bool(true)]; + tensor transpose_506_cast_fp16 = transpose(perm = transpose_506_perm_0, x = reshape_191_cast_fp16)[name = string("transpose_192")]; + tensor attn_95_cast_fp16 = matmul(transpose_x = attn_95_transpose_x_1, transpose_y = attn_95_transpose_y_1, x = transpose_506_cast_fp16, y = mh_w_287_cast_fp16)[name = string("attn_95_cast_fp16")]; + tensor var_14613 = const()[name = string("op_14613"), val = tensor([1, 2048, 1, 1])]; + tensor input_409_cast_fp16 = reshape(shape = var_14613, x = attn_95_cast_fp16)[name = string("input_409_cast_fp16")]; + string obj_423_pad_type_0 = const()[name = string("obj_423_pad_type_0"), val = string("valid")]; + tensor obj_423_strides_0 = const()[name = string("obj_423_strides_0"), val = tensor([1, 1])]; + tensor obj_423_pad_0 = const()[name = string("obj_423_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_423_dilations_0 = const()[name = string("obj_423_dilations_0"), val = tensor([1, 1])]; + int32 obj_423_groups_0 = const()[name = string("obj_423_groups_0"), val = int32(1)]; + tensor obj_423_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_423_dilations_0, groups = obj_423_groups_0, pad = obj_423_pad_0, pad_type = obj_423_pad_type_0, strides = obj_423_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_409_cast_fp16)[name = string("obj_423_cast_fp16")]; + tensor inputs_399_cast_fp16 = add(x = inputs_393_cast_fp16, y = obj_423_cast_fp16)[name = string("inputs_399_cast_fp16")]; + tensor inputs_sq_399_cast_fp16 = mul(x = inputs_399_cast_fp16, y = inputs_399_cast_fp16)[name = string("inputs_sq_399_cast_fp16")]; + tensor variance_399_axes_0 = const()[name = string("variance_399_axes_0"), val = tensor([1])]; + bool variance_399_keep_dims_0 = const()[name = string("variance_399_keep_dims_0"), val = bool(true)]; + tensor variance_399_cast_fp16 = reduce_mean(axes = variance_399_axes_0, keep_dims = variance_399_keep_dims_0, x = inputs_sq_399_cast_fp16)[name = string("variance_399_cast_fp16")]; + fp16 var_14631_to_fp16 = const()[name = string("op_14631_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14632_cast_fp16 = add(x = variance_399_cast_fp16, y = var_14631_to_fp16)[name = string("op_14632_cast_fp16")]; + fp32 var_14633_epsilon_0 = const()[name = string("op_14633_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14633_cast_fp16 = rsqrt(epsilon = var_14633_epsilon_0, x = var_14632_cast_fp16)[name = string("op_14633_cast_fp16")]; + tensor hidden_states_493_cast_fp16 = mul(x = inputs_399_cast_fp16, y = var_14633_cast_fp16)[name = string("hidden_states_493_cast_fp16")]; + tensor input_411_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_493_cast_fp16)[name = string("input_411_cast_fp16")]; + string input_413_pad_type_0 = const()[name = string("input_413_pad_type_0"), val = string("valid")]; + tensor input_413_strides_0 = const()[name = string("input_413_strides_0"), val = tensor([1, 1])]; + tensor input_413_pad_0 = const()[name = string("input_413_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_413_dilations_0 = const()[name = string("input_413_dilations_0"), val = tensor([1, 1])]; + int32 input_413_groups_0 = const()[name = string("input_413_groups_0"), val = int32(1)]; + tensor input_413_cast_fp16 = conv(dilations = input_413_dilations_0, groups = input_413_groups_0, pad = input_413_pad_0, pad_type = input_413_pad_type_0, strides = input_413_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_411_cast_fp16)[name = string("input_413_cast_fp16")]; + tensor var_14647_cast_fp16 = silu(x = input_413_cast_fp16)[name = string("op_14647_cast_fp16")]; + string var_14653_pad_type_0 = const()[name = string("op_14653_pad_type_0"), val = string("valid")]; + tensor var_14653_strides_0 = const()[name = string("op_14653_strides_0"), val = tensor([1, 1])]; + tensor var_14653_pad_0 = const()[name = string("op_14653_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14653_dilations_0 = const()[name = string("op_14653_dilations_0"), val = tensor([1, 1])]; + int32 var_14653_groups_0 = const()[name = string("op_14653_groups_0"), val = int32(1)]; + tensor var_14653_cast_fp16 = conv(dilations = var_14653_dilations_0, groups = var_14653_groups_0, pad = var_14653_pad_0, pad_type = var_14653_pad_type_0, strides = var_14653_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_411_cast_fp16)[name = string("op_14653_cast_fp16")]; + tensor input_415_cast_fp16 = mul(x = var_14647_cast_fp16, y = var_14653_cast_fp16)[name = string("input_415_cast_fp16")]; + string hidden_states_495_pad_type_0 = const()[name = string("hidden_states_495_pad_type_0"), val = string("valid")]; + tensor hidden_states_495_strides_0 = const()[name = string("hidden_states_495_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_495_pad_0 = const()[name = string("hidden_states_495_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_495_dilations_0 = const()[name = string("hidden_states_495_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_495_groups_0 = const()[name = string("hidden_states_495_groups_0"), val = int32(1)]; + tensor hidden_states_495_cast_fp16 = conv(dilations = hidden_states_495_dilations_0, groups = hidden_states_495_groups_0, pad = hidden_states_495_pad_0, pad_type = hidden_states_495_pad_type_0, strides = hidden_states_495_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_415_cast_fp16)[name = string("hidden_states_495_cast_fp16")]; + tensor inputs_401_cast_fp16 = add(x = inputs_399_cast_fp16, y = hidden_states_495_cast_fp16)[name = string("inputs_401_cast_fp16")]; + tensor obj_427_begin_0 = const()[name = string("obj_427_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_427_end_0 = const()[name = string("obj_427_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_427_end_mask_0 = const()[name = string("obj_427_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_427_cast_fp16 = slice_by_index(begin = obj_427_begin_0, end = obj_427_end_0, end_mask = obj_427_end_mask_0, x = key_caches_19_cast_fp16)[name = string("obj_427_cast_fp16")]; + tensor obj_429_begin_0 = const()[name = string("obj_429_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_429_end_0 = const()[name = string("obj_429_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_429_end_mask_0 = const()[name = string("obj_429_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_429_cast_fp16 = slice_by_index(begin = obj_429_begin_0, end = obj_429_end_0, end_mask = obj_429_end_mask_0, x = value_caches_19_cast_fp16)[name = string("obj_429_cast_fp16")]; + int32 var_14701 = const()[name = string("op_14701"), val = int32(3)]; + int32 var_14711 = const()[name = string("op_14711"), val = int32(-2)]; + tensor inputs_sq_401_cast_fp16 = mul(x = inputs_401_cast_fp16, y = inputs_401_cast_fp16)[name = string("inputs_sq_401_cast_fp16")]; + tensor variance_401_axes_0 = const()[name = string("variance_401_axes_0"), val = tensor([1])]; + bool variance_401_keep_dims_0 = const()[name = string("variance_401_keep_dims_0"), val = bool(true)]; + tensor variance_401_cast_fp16 = reduce_mean(axes = variance_401_axes_0, keep_dims = variance_401_keep_dims_0, x = inputs_sq_401_cast_fp16)[name = string("variance_401_cast_fp16")]; + fp16 var_14725_to_fp16 = const()[name = string("op_14725_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14726_cast_fp16 = add(x = variance_401_cast_fp16, y = var_14725_to_fp16)[name = string("op_14726_cast_fp16")]; + fp32 var_14727_epsilon_0 = const()[name = string("op_14727_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14727_cast_fp16 = rsqrt(epsilon = var_14727_epsilon_0, x = var_14726_cast_fp16)[name = string("op_14727_cast_fp16")]; + tensor hidden_states_497_cast_fp16 = mul(x = inputs_401_cast_fp16, y = var_14727_cast_fp16)[name = string("hidden_states_497_cast_fp16")]; + tensor obj_425_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_497_cast_fp16)[name = string("obj_425_cast_fp16")]; + string query_289_pad_type_0 = const()[name = string("query_289_pad_type_0"), val = string("valid")]; + tensor query_289_strides_0 = const()[name = string("query_289_strides_0"), val = tensor([1, 1])]; + tensor query_289_pad_0 = const()[name = string("query_289_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_289_dilations_0 = const()[name = string("query_289_dilations_0"), val = tensor([1, 1])]; + int32 query_289_groups_0 = const()[name = string("query_289_groups_0"), val = int32(1)]; + tensor query_289_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_289_dilations_0, groups = query_289_groups_0, pad = query_289_pad_0, pad_type = query_289_pad_type_0, strides = query_289_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_425_cast_fp16)[name = string("query_289_cast_fp16")]; + string current_key_193_pad_type_0 = const()[name = string("current_key_193_pad_type_0"), val = string("valid")]; + tensor current_key_193_strides_0 = const()[name = string("current_key_193_strides_0"), val = tensor([1, 1])]; + tensor current_key_193_pad_0 = const()[name = string("current_key_193_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_193_dilations_0 = const()[name = string("current_key_193_dilations_0"), val = tensor([1, 1])]; + int32 current_key_193_groups_0 = const()[name = string("current_key_193_groups_0"), val = int32(1)]; + tensor current_key_193_cast_fp16 = conv(dilations = current_key_193_dilations_0, groups = current_key_193_groups_0, pad = current_key_193_pad_0, pad_type = current_key_193_pad_type_0, strides = current_key_193_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_425_cast_fp16)[name = string("current_key_193_cast_fp16")]; + string current_value_97_pad_type_0 = const()[name = string("current_value_97_pad_type_0"), val = string("valid")]; + tensor current_value_97_strides_0 = const()[name = string("current_value_97_strides_0"), val = tensor([1, 1])]; + tensor current_value_97_pad_0 = const()[name = string("current_value_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_97_dilations_0 = const()[name = string("current_value_97_dilations_0"), val = tensor([1, 1])]; + int32 current_value_97_groups_0 = const()[name = string("current_value_97_groups_0"), val = int32(1)]; + tensor current_value_97_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_97_dilations_0, groups = current_value_97_groups_0, pad = current_value_97_pad_0, pad_type = current_value_97_pad_type_0, strides = current_value_97_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_425_cast_fp16)[name = string("current_value_97_cast_fp16")]; + tensor var_14764 = const()[name = string("op_14764"), val = tensor([16, 128, 1, 1])]; + tensor inputs_403_cast_fp16 = reshape(shape = var_14764, x = query_289_cast_fp16)[name = string("inputs_403_cast_fp16")]; + tensor inputs_sq_403_cast_fp16 = mul(x = inputs_403_cast_fp16, y = inputs_403_cast_fp16)[name = string("inputs_sq_403_cast_fp16")]; + tensor variance_403_axes_0 = const()[name = string("variance_403_axes_0"), val = tensor([1])]; + bool variance_403_keep_dims_0 = const()[name = string("variance_403_keep_dims_0"), val = bool(true)]; + tensor variance_403_cast_fp16 = reduce_mean(axes = variance_403_axes_0, keep_dims = variance_403_keep_dims_0, x = inputs_sq_403_cast_fp16)[name = string("variance_403_cast_fp16")]; + fp16 var_14770_to_fp16 = const()[name = string("op_14770_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14771_cast_fp16 = add(x = variance_403_cast_fp16, y = var_14770_to_fp16)[name = string("op_14771_cast_fp16")]; + fp32 var_14772_epsilon_0 = const()[name = string("op_14772_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14772_cast_fp16 = rsqrt(epsilon = var_14772_epsilon_0, x = var_14771_cast_fp16)[name = string("op_14772_cast_fp16")]; + tensor hidden_states_499_cast_fp16 = mul(x = inputs_403_cast_fp16, y = var_14772_cast_fp16)[name = string("hidden_states_499_cast_fp16")]; + tensor query_normed_97_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_499_cast_fp16)[name = string("query_normed_97_cast_fp16")]; + tensor var_14780 = const()[name = string("op_14780"), val = tensor([8, 128, 1, 1])]; + tensor inputs_405_cast_fp16 = reshape(shape = var_14780, x = current_key_193_cast_fp16)[name = string("inputs_405_cast_fp16")]; + tensor inputs_sq_405_cast_fp16 = mul(x = inputs_405_cast_fp16, y = inputs_405_cast_fp16)[name = string("inputs_sq_405_cast_fp16")]; + tensor variance_405_axes_0 = const()[name = string("variance_405_axes_0"), val = tensor([1])]; + bool variance_405_keep_dims_0 = const()[name = string("variance_405_keep_dims_0"), val = bool(true)]; + tensor variance_405_cast_fp16 = reduce_mean(axes = variance_405_axes_0, keep_dims = variance_405_keep_dims_0, x = inputs_sq_405_cast_fp16)[name = string("variance_405_cast_fp16")]; + fp16 var_14786_to_fp16 = const()[name = string("op_14786_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14787_cast_fp16 = add(x = variance_405_cast_fp16, y = var_14786_to_fp16)[name = string("op_14787_cast_fp16")]; + fp32 var_14788_epsilon_0 = const()[name = string("op_14788_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14788_cast_fp16 = rsqrt(epsilon = var_14788_epsilon_0, x = var_14787_cast_fp16)[name = string("op_14788_cast_fp16")]; + tensor hidden_states_501_cast_fp16 = mul(x = inputs_405_cast_fp16, y = var_14788_cast_fp16)[name = string("hidden_states_501_cast_fp16")]; + tensor current_key_normed_97_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_501_cast_fp16)[name = string("current_key_normed_97_cast_fp16")]; + tensor var_14806 = const()[name = string("op_14806"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_385_cast_fp16 = reshape(shape = var_14806, x = query_normed_97_cast_fp16)[name = string("mh_q_385_cast_fp16")]; + tensor var_14808 = const()[name = string("op_14808"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_385_cast_fp16 = reshape(shape = var_14808, x = current_key_normed_97_cast_fp16)[name = string("mh_k_385_cast_fp16")]; + tensor var_14812_cast_fp16 = mul(x = mh_q_385_cast_fp16, y = cos_91_to_fp16)[name = string("op_14812_cast_fp16")]; + tensor var_14817_begin_0 = const()[name = string("op_14817_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14817_end_0 = const()[name = string("op_14817_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_14817_end_mask_0 = const()[name = string("op_14817_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_14817_cast_fp16 = slice_by_index(begin = var_14817_begin_0, end = var_14817_end_0, end_mask = var_14817_end_mask_0, x = mh_q_385_cast_fp16)[name = string("op_14817_cast_fp16")]; + tensor var_14823_begin_0 = const()[name = string("op_14823_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_14823_end_0 = const()[name = string("op_14823_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_14823_end_mask_0 = const()[name = string("op_14823_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_14823_cast_fp16 = slice_by_index(begin = var_14823_begin_0, end = var_14823_end_0, end_mask = var_14823_end_mask_0, x = mh_q_385_cast_fp16)[name = string("op_14823_cast_fp16")]; + fp16 const_983_promoted_to_fp16 = const()[name = string("const_983_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14825_cast_fp16 = mul(x = var_14823_cast_fp16, y = const_983_promoted_to_fp16)[name = string("op_14825_cast_fp16")]; + bool var_14827_interleave_0 = const()[name = string("op_14827_interleave_0"), val = bool(false)]; + tensor var_14827_cast_fp16 = concat(axis = var_14711, interleave = var_14827_interleave_0, values = (var_14825_cast_fp16, var_14817_cast_fp16))[name = string("op_14827_cast_fp16")]; + tensor var_14828_cast_fp16 = mul(x = var_14827_cast_fp16, y = sin_91_to_fp16)[name = string("op_14828_cast_fp16")]; + tensor mh_q_387_cast_fp16 = add(x = var_14812_cast_fp16, y = var_14828_cast_fp16)[name = string("mh_q_387_cast_fp16")]; + tensor var_14830_cast_fp16 = mul(x = mh_k_385_cast_fp16, y = cos_91_to_fp16)[name = string("op_14830_cast_fp16")]; + tensor var_14835_begin_0 = const()[name = string("op_14835_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14835_end_0 = const()[name = string("op_14835_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_14835_end_mask_0 = const()[name = string("op_14835_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_14835_cast_fp16 = slice_by_index(begin = var_14835_begin_0, end = var_14835_end_0, end_mask = var_14835_end_mask_0, x = mh_k_385_cast_fp16)[name = string("op_14835_cast_fp16")]; + tensor var_14841_begin_0 = const()[name = string("op_14841_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_14841_end_0 = const()[name = string("op_14841_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_14841_end_mask_0 = const()[name = string("op_14841_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_14841_cast_fp16 = slice_by_index(begin = var_14841_begin_0, end = var_14841_end_0, end_mask = var_14841_end_mask_0, x = mh_k_385_cast_fp16)[name = string("op_14841_cast_fp16")]; + fp16 const_986_promoted_to_fp16 = const()[name = string("const_986_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_14843_cast_fp16 = mul(x = var_14841_cast_fp16, y = const_986_promoted_to_fp16)[name = string("op_14843_cast_fp16")]; + bool var_14845_interleave_0 = const()[name = string("op_14845_interleave_0"), val = bool(false)]; + tensor var_14845_cast_fp16 = concat(axis = var_14711, interleave = var_14845_interleave_0, values = (var_14843_cast_fp16, var_14835_cast_fp16))[name = string("op_14845_cast_fp16")]; + tensor var_14846_cast_fp16 = mul(x = var_14845_cast_fp16, y = sin_91_to_fp16)[name = string("op_14846_cast_fp16")]; + tensor mh_k_387_cast_fp16 = add(x = var_14830_cast_fp16, y = var_14846_cast_fp16)[name = string("mh_k_387_cast_fp16")]; + tensor var_14850 = const()[name = string("op_14850"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_195_cast_fp16 = reshape(shape = var_14850, x = mh_k_387_cast_fp16)[name = string("current_key_195_cast_fp16")]; + tensor var_14856_to_fp16 = const()[name = string("op_14856_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198400)))]; + tensor var_14857_cast_fp16 = mul(x = obj_427_cast_fp16, y = var_14856_to_fp16)[name = string("op_14857_cast_fp16")]; + tensor var_14854_to_fp16 = const()[name = string("op_14854_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198528)))]; + tensor var_14858_cast_fp16 = mul(x = current_key_195_cast_fp16, y = var_14854_to_fp16)[name = string("op_14858_cast_fp16")]; + tensor key_195_cast_fp16 = add(x = var_14857_cast_fp16, y = var_14858_cast_fp16)[name = string("key_195_cast_fp16")]; + tensor var_14860_to_fp16 = const()[name = string("op_14860_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198400)))]; + tensor var_14861_cast_fp16 = mul(x = obj_429_cast_fp16, y = var_14860_to_fp16)[name = string("op_14861_cast_fp16")]; + tensor var_14862_cast_fp16 = mul(x = current_value_97_cast_fp16, y = var_14854_to_fp16)[name = string("op_14862_cast_fp16")]; + tensor value_97_cast_fp16 = add(x = var_14861_cast_fp16, y = var_14862_cast_fp16)[name = string("value_97_cast_fp16")]; + fp16 var_14869_to_fp16 = const()[name = string("op_14869_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_391_cast_fp16 = mul(x = mh_q_387_cast_fp16, y = var_14869_to_fp16)[name = string("mh_q_391_cast_fp16")]; + tensor var_14871 = const()[name = string("op_14871"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_389_cast_fp16 = reshape(shape = var_14871, x = key_195_cast_fp16)[name = string("mh_k_389_cast_fp16")]; + tensor var_14873 = const()[name = string("op_14873"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_193_cast_fp16 = reshape(shape = var_14873, x = value_97_cast_fp16)[name = string("mh_v_193_cast_fp16")]; + tensor transpose_192_perm_0 = const()[name = string("transpose_192_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_96_reps_0 = const()[name = string("tile_96_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_192_cast_fp16 = transpose(perm = transpose_192_perm_0, x = mh_k_389_cast_fp16)[name = string("transpose_191")]; + tensor tile_96_cast_fp16 = tile(reps = tile_96_reps_0, x = transpose_192_cast_fp16)[name = string("tile_96_cast_fp16")]; + tensor concat_240 = const()[name = string("concat_240"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_192_cast_fp16 = reshape(shape = concat_240, x = tile_96_cast_fp16)[name = string("reshape_192_cast_fp16")]; + tensor transpose_193_perm_0 = const()[name = string("transpose_193_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_241 = const()[name = string("concat_241"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_193_cast_fp16 = transpose(perm = transpose_193_perm_0, x = reshape_192_cast_fp16)[name = string("transpose_190")]; + tensor reshape_193_cast_fp16 = reshape(shape = concat_241, x = transpose_193_cast_fp16)[name = string("reshape_193_cast_fp16")]; + tensor transpose_194_perm_0 = const()[name = string("transpose_194_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_97_reps_0 = const()[name = string("tile_97_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_194_cast_fp16 = transpose(perm = transpose_194_perm_0, x = mh_v_193_cast_fp16)[name = string("transpose_189")]; + tensor tile_97_cast_fp16 = tile(reps = tile_97_reps_0, x = transpose_194_cast_fp16)[name = string("tile_97_cast_fp16")]; + tensor concat_242 = const()[name = string("concat_242"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_194_cast_fp16 = reshape(shape = concat_242, x = tile_97_cast_fp16)[name = string("reshape_194_cast_fp16")]; + tensor transpose_195_perm_0 = const()[name = string("transpose_195_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_243 = const()[name = string("concat_243"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_195_cast_fp16 = transpose(perm = transpose_195_perm_0, x = reshape_194_cast_fp16)[name = string("transpose_188")]; + tensor reshape_195_cast_fp16 = reshape(shape = concat_243, x = transpose_195_cast_fp16)[name = string("reshape_195_cast_fp16")]; + tensor transpose_509_perm_0 = const()[name = string("transpose_509_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_289_transpose_x_1 = const()[name = string("mh_w_289_transpose_x_1"), val = bool(true)]; + bool mh_w_289_transpose_y_1 = const()[name = string("mh_w_289_transpose_y_1"), val = bool(false)]; + tensor transpose_509_cast_fp16 = transpose(perm = transpose_509_perm_0, x = reshape_193_cast_fp16)[name = string("transpose_187")]; + tensor mh_w_289_cast_fp16 = matmul(transpose_x = mh_w_289_transpose_x_1, transpose_y = mh_w_289_transpose_y_1, x = mh_q_391_cast_fp16, y = transpose_509_cast_fp16)[name = string("mh_w_289_cast_fp16")]; + tensor var_14881_to_fp16 = const()[name = string("op_14881_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198656)))]; + tensor mh_w_291_cast_fp16 = add(x = mh_w_289_cast_fp16, y = var_14881_to_fp16)[name = string("mh_w_291_cast_fp16")]; + tensor mh_w_293_cast_fp16 = softmax(axis = var_14701, x = mh_w_291_cast_fp16)[name = string("mh_w_293_cast_fp16")]; + tensor transpose_510_perm_0 = const()[name = string("transpose_510_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_97_transpose_x_1 = const()[name = string("attn_97_transpose_x_1"), val = bool(false)]; + bool attn_97_transpose_y_1 = const()[name = string("attn_97_transpose_y_1"), val = bool(true)]; + tensor transpose_510_cast_fp16 = transpose(perm = transpose_510_perm_0, x = reshape_195_cast_fp16)[name = string("transpose_186")]; + tensor attn_97_cast_fp16 = matmul(transpose_x = attn_97_transpose_x_1, transpose_y = attn_97_transpose_y_1, x = transpose_510_cast_fp16, y = mh_w_293_cast_fp16)[name = string("attn_97_cast_fp16")]; + tensor var_14887 = const()[name = string("op_14887"), val = tensor([1, 2048, 1, 1])]; + tensor input_417_cast_fp16 = reshape(shape = var_14887, x = attn_97_cast_fp16)[name = string("input_417_cast_fp16")]; + string obj_431_pad_type_0 = const()[name = string("obj_431_pad_type_0"), val = string("valid")]; + tensor obj_431_strides_0 = const()[name = string("obj_431_strides_0"), val = tensor([1, 1])]; + tensor obj_431_pad_0 = const()[name = string("obj_431_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_431_dilations_0 = const()[name = string("obj_431_dilations_0"), val = tensor([1, 1])]; + int32 obj_431_groups_0 = const()[name = string("obj_431_groups_0"), val = int32(1)]; + tensor obj_431_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_431_dilations_0, groups = obj_431_groups_0, pad = obj_431_pad_0, pad_type = obj_431_pad_type_0, strides = obj_431_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_417_cast_fp16)[name = string("obj_431_cast_fp16")]; + tensor inputs_407_cast_fp16 = add(x = inputs_401_cast_fp16, y = obj_431_cast_fp16)[name = string("inputs_407_cast_fp16")]; + tensor inputs_sq_407_cast_fp16 = mul(x = inputs_407_cast_fp16, y = inputs_407_cast_fp16)[name = string("inputs_sq_407_cast_fp16")]; + tensor variance_407_axes_0 = const()[name = string("variance_407_axes_0"), val = tensor([1])]; + bool variance_407_keep_dims_0 = const()[name = string("variance_407_keep_dims_0"), val = bool(true)]; + tensor variance_407_cast_fp16 = reduce_mean(axes = variance_407_axes_0, keep_dims = variance_407_keep_dims_0, x = inputs_sq_407_cast_fp16)[name = string("variance_407_cast_fp16")]; + fp16 var_14905_to_fp16 = const()[name = string("op_14905_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_14906_cast_fp16 = add(x = variance_407_cast_fp16, y = var_14905_to_fp16)[name = string("op_14906_cast_fp16")]; + fp32 var_14907_epsilon_0 = const()[name = string("op_14907_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_14907_cast_fp16 = rsqrt(epsilon = var_14907_epsilon_0, x = var_14906_cast_fp16)[name = string("op_14907_cast_fp16")]; + tensor hidden_states_503_cast_fp16 = mul(x = inputs_407_cast_fp16, y = var_14907_cast_fp16)[name = string("hidden_states_503_cast_fp16")]; + tensor input_419_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_503_cast_fp16)[name = string("input_419_cast_fp16")]; + string input_421_pad_type_0 = const()[name = string("input_421_pad_type_0"), val = string("valid")]; + tensor input_421_strides_0 = const()[name = string("input_421_strides_0"), val = tensor([1, 1])]; + tensor input_421_pad_0 = const()[name = string("input_421_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_421_dilations_0 = const()[name = string("input_421_dilations_0"), val = tensor([1, 1])]; + int32 input_421_groups_0 = const()[name = string("input_421_groups_0"), val = int32(1)]; + tensor input_421_cast_fp16 = conv(dilations = input_421_dilations_0, groups = input_421_groups_0, pad = input_421_pad_0, pad_type = input_421_pad_type_0, strides = input_421_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_419_cast_fp16)[name = string("input_421_cast_fp16")]; + tensor var_14921_cast_fp16 = silu(x = input_421_cast_fp16)[name = string("op_14921_cast_fp16")]; + string var_14927_pad_type_0 = const()[name = string("op_14927_pad_type_0"), val = string("valid")]; + tensor var_14927_strides_0 = const()[name = string("op_14927_strides_0"), val = tensor([1, 1])]; + tensor var_14927_pad_0 = const()[name = string("op_14927_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_14927_dilations_0 = const()[name = string("op_14927_dilations_0"), val = tensor([1, 1])]; + int32 var_14927_groups_0 = const()[name = string("op_14927_groups_0"), val = int32(1)]; + tensor var_14927_cast_fp16 = conv(dilations = var_14927_dilations_0, groups = var_14927_groups_0, pad = var_14927_pad_0, pad_type = var_14927_pad_type_0, strides = var_14927_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_419_cast_fp16)[name = string("op_14927_cast_fp16")]; + tensor input_423_cast_fp16 = mul(x = var_14921_cast_fp16, y = var_14927_cast_fp16)[name = string("input_423_cast_fp16")]; + string hidden_states_505_pad_type_0 = const()[name = string("hidden_states_505_pad_type_0"), val = string("valid")]; + tensor hidden_states_505_strides_0 = const()[name = string("hidden_states_505_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_505_pad_0 = const()[name = string("hidden_states_505_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_505_dilations_0 = const()[name = string("hidden_states_505_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_505_groups_0 = const()[name = string("hidden_states_505_groups_0"), val = int32(1)]; + tensor hidden_states_505_cast_fp16 = conv(dilations = hidden_states_505_dilations_0, groups = hidden_states_505_groups_0, pad = hidden_states_505_pad_0, pad_type = hidden_states_505_pad_type_0, strides = hidden_states_505_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_423_cast_fp16)[name = string("hidden_states_505_cast_fp16")]; + tensor inputs_409_cast_fp16 = add(x = inputs_407_cast_fp16, y = hidden_states_505_cast_fp16)[name = string("inputs_409_cast_fp16")]; + tensor obj_435_begin_0 = const()[name = string("obj_435_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_435_end_0 = const()[name = string("obj_435_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_435_end_mask_0 = const()[name = string("obj_435_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_435_cast_fp16 = slice_by_index(begin = obj_435_begin_0, end = obj_435_end_0, end_mask = obj_435_end_mask_0, x = key_caches_19_cast_fp16)[name = string("obj_435_cast_fp16")]; + tensor obj_437_begin_0 = const()[name = string("obj_437_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_437_end_0 = const()[name = string("obj_437_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_437_end_mask_0 = const()[name = string("obj_437_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_437_cast_fp16 = slice_by_index(begin = obj_437_begin_0, end = obj_437_end_0, end_mask = obj_437_end_mask_0, x = value_caches_19_cast_fp16)[name = string("obj_437_cast_fp16")]; + int32 var_14975 = const()[name = string("op_14975"), val = int32(3)]; + int32 var_14985 = const()[name = string("op_14985"), val = int32(-2)]; + tensor inputs_sq_409_cast_fp16 = mul(x = inputs_409_cast_fp16, y = inputs_409_cast_fp16)[name = string("inputs_sq_409_cast_fp16")]; + tensor variance_409_axes_0 = const()[name = string("variance_409_axes_0"), val = tensor([1])]; + bool variance_409_keep_dims_0 = const()[name = string("variance_409_keep_dims_0"), val = bool(true)]; + tensor variance_409_cast_fp16 = reduce_mean(axes = variance_409_axes_0, keep_dims = variance_409_keep_dims_0, x = inputs_sq_409_cast_fp16)[name = string("variance_409_cast_fp16")]; + fp16 var_14999_to_fp16 = const()[name = string("op_14999_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15000_cast_fp16 = add(x = variance_409_cast_fp16, y = var_14999_to_fp16)[name = string("op_15000_cast_fp16")]; + fp32 var_15001_epsilon_0 = const()[name = string("op_15001_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15001_cast_fp16 = rsqrt(epsilon = var_15001_epsilon_0, x = var_15000_cast_fp16)[name = string("op_15001_cast_fp16")]; + tensor hidden_states_507_cast_fp16 = mul(x = inputs_409_cast_fp16, y = var_15001_cast_fp16)[name = string("hidden_states_507_cast_fp16")]; + tensor obj_433_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_507_cast_fp16)[name = string("obj_433_cast_fp16")]; + string query_295_pad_type_0 = const()[name = string("query_295_pad_type_0"), val = string("valid")]; + tensor query_295_strides_0 = const()[name = string("query_295_strides_0"), val = tensor([1, 1])]; + tensor query_295_pad_0 = const()[name = string("query_295_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_295_dilations_0 = const()[name = string("query_295_dilations_0"), val = tensor([1, 1])]; + int32 query_295_groups_0 = const()[name = string("query_295_groups_0"), val = int32(1)]; + tensor query_295_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_295_dilations_0, groups = query_295_groups_0, pad = query_295_pad_0, pad_type = query_295_pad_type_0, strides = query_295_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_433_cast_fp16)[name = string("query_295_cast_fp16")]; + string current_key_197_pad_type_0 = const()[name = string("current_key_197_pad_type_0"), val = string("valid")]; + tensor current_key_197_strides_0 = const()[name = string("current_key_197_strides_0"), val = tensor([1, 1])]; + tensor current_key_197_pad_0 = const()[name = string("current_key_197_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_197_dilations_0 = const()[name = string("current_key_197_dilations_0"), val = tensor([1, 1])]; + int32 current_key_197_groups_0 = const()[name = string("current_key_197_groups_0"), val = int32(1)]; + tensor current_key_197_cast_fp16 = conv(dilations = current_key_197_dilations_0, groups = current_key_197_groups_0, pad = current_key_197_pad_0, pad_type = current_key_197_pad_type_0, strides = current_key_197_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_433_cast_fp16)[name = string("current_key_197_cast_fp16")]; + string current_value_99_pad_type_0 = const()[name = string("current_value_99_pad_type_0"), val = string("valid")]; + tensor current_value_99_strides_0 = const()[name = string("current_value_99_strides_0"), val = tensor([1, 1])]; + tensor current_value_99_pad_0 = const()[name = string("current_value_99_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_99_dilations_0 = const()[name = string("current_value_99_dilations_0"), val = tensor([1, 1])]; + int32 current_value_99_groups_0 = const()[name = string("current_value_99_groups_0"), val = int32(1)]; + tensor current_value_99_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_99_dilations_0, groups = current_value_99_groups_0, pad = current_value_99_pad_0, pad_type = current_value_99_pad_type_0, strides = current_value_99_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_433_cast_fp16)[name = string("current_value_99_cast_fp16")]; + tensor var_15038 = const()[name = string("op_15038"), val = tensor([16, 128, 1, 1])]; + tensor inputs_411_cast_fp16 = reshape(shape = var_15038, x = query_295_cast_fp16)[name = string("inputs_411_cast_fp16")]; + tensor inputs_sq_411_cast_fp16 = mul(x = inputs_411_cast_fp16, y = inputs_411_cast_fp16)[name = string("inputs_sq_411_cast_fp16")]; + tensor variance_411_axes_0 = const()[name = string("variance_411_axes_0"), val = tensor([1])]; + bool variance_411_keep_dims_0 = const()[name = string("variance_411_keep_dims_0"), val = bool(true)]; + tensor variance_411_cast_fp16 = reduce_mean(axes = variance_411_axes_0, keep_dims = variance_411_keep_dims_0, x = inputs_sq_411_cast_fp16)[name = string("variance_411_cast_fp16")]; + fp16 var_15044_to_fp16 = const()[name = string("op_15044_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15045_cast_fp16 = add(x = variance_411_cast_fp16, y = var_15044_to_fp16)[name = string("op_15045_cast_fp16")]; + fp32 var_15046_epsilon_0 = const()[name = string("op_15046_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15046_cast_fp16 = rsqrt(epsilon = var_15046_epsilon_0, x = var_15045_cast_fp16)[name = string("op_15046_cast_fp16")]; + tensor hidden_states_509_cast_fp16 = mul(x = inputs_411_cast_fp16, y = var_15046_cast_fp16)[name = string("hidden_states_509_cast_fp16")]; + tensor query_normed_99_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_509_cast_fp16)[name = string("query_normed_99_cast_fp16")]; + tensor var_15054 = const()[name = string("op_15054"), val = tensor([8, 128, 1, 1])]; + tensor inputs_413_cast_fp16 = reshape(shape = var_15054, x = current_key_197_cast_fp16)[name = string("inputs_413_cast_fp16")]; + tensor inputs_sq_413_cast_fp16 = mul(x = inputs_413_cast_fp16, y = inputs_413_cast_fp16)[name = string("inputs_sq_413_cast_fp16")]; + tensor variance_413_axes_0 = const()[name = string("variance_413_axes_0"), val = tensor([1])]; + bool variance_413_keep_dims_0 = const()[name = string("variance_413_keep_dims_0"), val = bool(true)]; + tensor variance_413_cast_fp16 = reduce_mean(axes = variance_413_axes_0, keep_dims = variance_413_keep_dims_0, x = inputs_sq_413_cast_fp16)[name = string("variance_413_cast_fp16")]; + fp16 var_15060_to_fp16 = const()[name = string("op_15060_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15061_cast_fp16 = add(x = variance_413_cast_fp16, y = var_15060_to_fp16)[name = string("op_15061_cast_fp16")]; + fp32 var_15062_epsilon_0 = const()[name = string("op_15062_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15062_cast_fp16 = rsqrt(epsilon = var_15062_epsilon_0, x = var_15061_cast_fp16)[name = string("op_15062_cast_fp16")]; + tensor hidden_states_511_cast_fp16 = mul(x = inputs_413_cast_fp16, y = var_15062_cast_fp16)[name = string("hidden_states_511_cast_fp16")]; + tensor current_key_normed_99_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_511_cast_fp16)[name = string("current_key_normed_99_cast_fp16")]; + tensor var_15080 = const()[name = string("op_15080"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_393_cast_fp16 = reshape(shape = var_15080, x = query_normed_99_cast_fp16)[name = string("mh_q_393_cast_fp16")]; + tensor var_15082 = const()[name = string("op_15082"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_393_cast_fp16 = reshape(shape = var_15082, x = current_key_normed_99_cast_fp16)[name = string("mh_k_393_cast_fp16")]; + tensor var_15086_cast_fp16 = mul(x = mh_q_393_cast_fp16, y = cos_91_to_fp16)[name = string("op_15086_cast_fp16")]; + tensor var_15091_begin_0 = const()[name = string("op_15091_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15091_end_0 = const()[name = string("op_15091_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_15091_end_mask_0 = const()[name = string("op_15091_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_15091_cast_fp16 = slice_by_index(begin = var_15091_begin_0, end = var_15091_end_0, end_mask = var_15091_end_mask_0, x = mh_q_393_cast_fp16)[name = string("op_15091_cast_fp16")]; + tensor var_15097_begin_0 = const()[name = string("op_15097_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_15097_end_0 = const()[name = string("op_15097_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_15097_end_mask_0 = const()[name = string("op_15097_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_15097_cast_fp16 = slice_by_index(begin = var_15097_begin_0, end = var_15097_end_0, end_mask = var_15097_end_mask_0, x = mh_q_393_cast_fp16)[name = string("op_15097_cast_fp16")]; + fp16 const_1003_promoted_to_fp16 = const()[name = string("const_1003_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15099_cast_fp16 = mul(x = var_15097_cast_fp16, y = const_1003_promoted_to_fp16)[name = string("op_15099_cast_fp16")]; + bool var_15101_interleave_0 = const()[name = string("op_15101_interleave_0"), val = bool(false)]; + tensor var_15101_cast_fp16 = concat(axis = var_14985, interleave = var_15101_interleave_0, values = (var_15099_cast_fp16, var_15091_cast_fp16))[name = string("op_15101_cast_fp16")]; + tensor var_15102_cast_fp16 = mul(x = var_15101_cast_fp16, y = sin_91_to_fp16)[name = string("op_15102_cast_fp16")]; + tensor mh_q_395_cast_fp16 = add(x = var_15086_cast_fp16, y = var_15102_cast_fp16)[name = string("mh_q_395_cast_fp16")]; + tensor var_15104_cast_fp16 = mul(x = mh_k_393_cast_fp16, y = cos_91_to_fp16)[name = string("op_15104_cast_fp16")]; + tensor var_15109_begin_0 = const()[name = string("op_15109_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15109_end_0 = const()[name = string("op_15109_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_15109_end_mask_0 = const()[name = string("op_15109_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_15109_cast_fp16 = slice_by_index(begin = var_15109_begin_0, end = var_15109_end_0, end_mask = var_15109_end_mask_0, x = mh_k_393_cast_fp16)[name = string("op_15109_cast_fp16")]; + tensor var_15115_begin_0 = const()[name = string("op_15115_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_15115_end_0 = const()[name = string("op_15115_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_15115_end_mask_0 = const()[name = string("op_15115_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_15115_cast_fp16 = slice_by_index(begin = var_15115_begin_0, end = var_15115_end_0, end_mask = var_15115_end_mask_0, x = mh_k_393_cast_fp16)[name = string("op_15115_cast_fp16")]; + fp16 const_1006_promoted_to_fp16 = const()[name = string("const_1006_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15117_cast_fp16 = mul(x = var_15115_cast_fp16, y = const_1006_promoted_to_fp16)[name = string("op_15117_cast_fp16")]; + bool var_15119_interleave_0 = const()[name = string("op_15119_interleave_0"), val = bool(false)]; + tensor var_15119_cast_fp16 = concat(axis = var_14985, interleave = var_15119_interleave_0, values = (var_15117_cast_fp16, var_15109_cast_fp16))[name = string("op_15119_cast_fp16")]; + tensor var_15120_cast_fp16 = mul(x = var_15119_cast_fp16, y = sin_91_to_fp16)[name = string("op_15120_cast_fp16")]; + tensor mh_k_395_cast_fp16 = add(x = var_15104_cast_fp16, y = var_15120_cast_fp16)[name = string("mh_k_395_cast_fp16")]; + tensor var_15124 = const()[name = string("op_15124"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_199_cast_fp16 = reshape(shape = var_15124, x = mh_k_395_cast_fp16)[name = string("current_key_199_cast_fp16")]; + tensor var_15130_to_fp16 = const()[name = string("op_15130_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198400)))]; + tensor var_15131_cast_fp16 = mul(x = obj_435_cast_fp16, y = var_15130_to_fp16)[name = string("op_15131_cast_fp16")]; + tensor var_15128_to_fp16 = const()[name = string("op_15128_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198528)))]; + tensor var_15132_cast_fp16 = mul(x = current_key_199_cast_fp16, y = var_15128_to_fp16)[name = string("op_15132_cast_fp16")]; + tensor key_199_cast_fp16 = add(x = var_15131_cast_fp16, y = var_15132_cast_fp16)[name = string("key_199_cast_fp16")]; + tensor var_15134_to_fp16 = const()[name = string("op_15134_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198400)))]; + tensor var_15135_cast_fp16 = mul(x = obj_437_cast_fp16, y = var_15134_to_fp16)[name = string("op_15135_cast_fp16")]; + tensor var_15136_cast_fp16 = mul(x = current_value_99_cast_fp16, y = var_15128_to_fp16)[name = string("op_15136_cast_fp16")]; + tensor value_99_cast_fp16 = add(x = var_15135_cast_fp16, y = var_15136_cast_fp16)[name = string("value_99_cast_fp16")]; + fp16 var_15143_to_fp16 = const()[name = string("op_15143_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_399_cast_fp16 = mul(x = mh_q_395_cast_fp16, y = var_15143_to_fp16)[name = string("mh_q_399_cast_fp16")]; + tensor var_15145 = const()[name = string("op_15145"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_397_cast_fp16 = reshape(shape = var_15145, x = key_199_cast_fp16)[name = string("mh_k_397_cast_fp16")]; + tensor var_15147 = const()[name = string("op_15147"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_197_cast_fp16 = reshape(shape = var_15147, x = value_99_cast_fp16)[name = string("mh_v_197_cast_fp16")]; + tensor transpose_196_perm_0 = const()[name = string("transpose_196_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_98_reps_0 = const()[name = string("tile_98_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_196_cast_fp16 = transpose(perm = transpose_196_perm_0, x = mh_k_397_cast_fp16)[name = string("transpose_185")]; + tensor tile_98_cast_fp16 = tile(reps = tile_98_reps_0, x = transpose_196_cast_fp16)[name = string("tile_98_cast_fp16")]; + tensor concat_244 = const()[name = string("concat_244"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_196_cast_fp16 = reshape(shape = concat_244, x = tile_98_cast_fp16)[name = string("reshape_196_cast_fp16")]; + tensor transpose_197_perm_0 = const()[name = string("transpose_197_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_245 = const()[name = string("concat_245"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_197_cast_fp16 = transpose(perm = transpose_197_perm_0, x = reshape_196_cast_fp16)[name = string("transpose_184")]; + tensor reshape_197_cast_fp16 = reshape(shape = concat_245, x = transpose_197_cast_fp16)[name = string("reshape_197_cast_fp16")]; + tensor transpose_198_perm_0 = const()[name = string("transpose_198_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_99_reps_0 = const()[name = string("tile_99_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_198_cast_fp16 = transpose(perm = transpose_198_perm_0, x = mh_v_197_cast_fp16)[name = string("transpose_183")]; + tensor tile_99_cast_fp16 = tile(reps = tile_99_reps_0, x = transpose_198_cast_fp16)[name = string("tile_99_cast_fp16")]; + tensor concat_246 = const()[name = string("concat_246"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_198_cast_fp16 = reshape(shape = concat_246, x = tile_99_cast_fp16)[name = string("reshape_198_cast_fp16")]; + tensor transpose_199_perm_0 = const()[name = string("transpose_199_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_247 = const()[name = string("concat_247"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_199_cast_fp16 = transpose(perm = transpose_199_perm_0, x = reshape_198_cast_fp16)[name = string("transpose_182")]; + tensor reshape_199_cast_fp16 = reshape(shape = concat_247, x = transpose_199_cast_fp16)[name = string("reshape_199_cast_fp16")]; + tensor transpose_513_perm_0 = const()[name = string("transpose_513_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_295_transpose_x_1 = const()[name = string("mh_w_295_transpose_x_1"), val = bool(true)]; + bool mh_w_295_transpose_y_1 = const()[name = string("mh_w_295_transpose_y_1"), val = bool(false)]; + tensor transpose_513_cast_fp16 = transpose(perm = transpose_513_perm_0, x = reshape_197_cast_fp16)[name = string("transpose_181")]; + tensor mh_w_295_cast_fp16 = matmul(transpose_x = mh_w_295_transpose_x_1, transpose_y = mh_w_295_transpose_y_1, x = mh_q_399_cast_fp16, y = transpose_513_cast_fp16)[name = string("mh_w_295_cast_fp16")]; + tensor var_15155_to_fp16 = const()[name = string("op_15155_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198656)))]; + tensor mh_w_297_cast_fp16 = add(x = mh_w_295_cast_fp16, y = var_15155_to_fp16)[name = string("mh_w_297_cast_fp16")]; + tensor mh_w_299_cast_fp16 = softmax(axis = var_14975, x = mh_w_297_cast_fp16)[name = string("mh_w_299_cast_fp16")]; + tensor transpose_514_perm_0 = const()[name = string("transpose_514_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_99_transpose_x_1 = const()[name = string("attn_99_transpose_x_1"), val = bool(false)]; + bool attn_99_transpose_y_1 = const()[name = string("attn_99_transpose_y_1"), val = bool(true)]; + tensor transpose_514_cast_fp16 = transpose(perm = transpose_514_perm_0, x = reshape_199_cast_fp16)[name = string("transpose_180")]; + tensor attn_99_cast_fp16 = matmul(transpose_x = attn_99_transpose_x_1, transpose_y = attn_99_transpose_y_1, x = transpose_514_cast_fp16, y = mh_w_299_cast_fp16)[name = string("attn_99_cast_fp16")]; + tensor var_15161 = const()[name = string("op_15161"), val = tensor([1, 2048, 1, 1])]; + tensor input_425_cast_fp16 = reshape(shape = var_15161, x = attn_99_cast_fp16)[name = string("input_425_cast_fp16")]; + string obj_439_pad_type_0 = const()[name = string("obj_439_pad_type_0"), val = string("valid")]; + tensor obj_439_strides_0 = const()[name = string("obj_439_strides_0"), val = tensor([1, 1])]; + tensor obj_439_pad_0 = const()[name = string("obj_439_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_439_dilations_0 = const()[name = string("obj_439_dilations_0"), val = tensor([1, 1])]; + int32 obj_439_groups_0 = const()[name = string("obj_439_groups_0"), val = int32(1)]; + tensor obj_439_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_439_dilations_0, groups = obj_439_groups_0, pad = obj_439_pad_0, pad_type = obj_439_pad_type_0, strides = obj_439_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_425_cast_fp16)[name = string("obj_439_cast_fp16")]; + tensor inputs_415_cast_fp16 = add(x = inputs_409_cast_fp16, y = obj_439_cast_fp16)[name = string("inputs_415_cast_fp16")]; + tensor inputs_sq_415_cast_fp16 = mul(x = inputs_415_cast_fp16, y = inputs_415_cast_fp16)[name = string("inputs_sq_415_cast_fp16")]; + tensor variance_415_axes_0 = const()[name = string("variance_415_axes_0"), val = tensor([1])]; + bool variance_415_keep_dims_0 = const()[name = string("variance_415_keep_dims_0"), val = bool(true)]; + tensor variance_415_cast_fp16 = reduce_mean(axes = variance_415_axes_0, keep_dims = variance_415_keep_dims_0, x = inputs_sq_415_cast_fp16)[name = string("variance_415_cast_fp16")]; + fp16 var_15179_to_fp16 = const()[name = string("op_15179_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15180_cast_fp16 = add(x = variance_415_cast_fp16, y = var_15179_to_fp16)[name = string("op_15180_cast_fp16")]; + fp32 var_15181_epsilon_0 = const()[name = string("op_15181_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15181_cast_fp16 = rsqrt(epsilon = var_15181_epsilon_0, x = var_15180_cast_fp16)[name = string("op_15181_cast_fp16")]; + tensor hidden_states_513_cast_fp16 = mul(x = inputs_415_cast_fp16, y = var_15181_cast_fp16)[name = string("hidden_states_513_cast_fp16")]; + tensor input_427_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_513_cast_fp16)[name = string("input_427_cast_fp16")]; + string input_429_pad_type_0 = const()[name = string("input_429_pad_type_0"), val = string("valid")]; + tensor input_429_strides_0 = const()[name = string("input_429_strides_0"), val = tensor([1, 1])]; + tensor input_429_pad_0 = const()[name = string("input_429_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_429_dilations_0 = const()[name = string("input_429_dilations_0"), val = tensor([1, 1])]; + int32 input_429_groups_0 = const()[name = string("input_429_groups_0"), val = int32(1)]; + tensor input_429_cast_fp16 = conv(dilations = input_429_dilations_0, groups = input_429_groups_0, pad = input_429_pad_0, pad_type = input_429_pad_type_0, strides = input_429_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_427_cast_fp16)[name = string("input_429_cast_fp16")]; + tensor var_15195_cast_fp16 = silu(x = input_429_cast_fp16)[name = string("op_15195_cast_fp16")]; + string var_15201_pad_type_0 = const()[name = string("op_15201_pad_type_0"), val = string("valid")]; + tensor var_15201_strides_0 = const()[name = string("op_15201_strides_0"), val = tensor([1, 1])]; + tensor var_15201_pad_0 = const()[name = string("op_15201_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15201_dilations_0 = const()[name = string("op_15201_dilations_0"), val = tensor([1, 1])]; + int32 var_15201_groups_0 = const()[name = string("op_15201_groups_0"), val = int32(1)]; + tensor var_15201_cast_fp16 = conv(dilations = var_15201_dilations_0, groups = var_15201_groups_0, pad = var_15201_pad_0, pad_type = var_15201_pad_type_0, strides = var_15201_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_427_cast_fp16)[name = string("op_15201_cast_fp16")]; + tensor input_431_cast_fp16 = mul(x = var_15195_cast_fp16, y = var_15201_cast_fp16)[name = string("input_431_cast_fp16")]; + string hidden_states_515_pad_type_0 = const()[name = string("hidden_states_515_pad_type_0"), val = string("valid")]; + tensor hidden_states_515_strides_0 = const()[name = string("hidden_states_515_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_515_pad_0 = const()[name = string("hidden_states_515_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_515_dilations_0 = const()[name = string("hidden_states_515_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_515_groups_0 = const()[name = string("hidden_states_515_groups_0"), val = int32(1)]; + tensor hidden_states_515_cast_fp16 = conv(dilations = hidden_states_515_dilations_0, groups = hidden_states_515_groups_0, pad = hidden_states_515_pad_0, pad_type = hidden_states_515_pad_type_0, strides = hidden_states_515_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_431_cast_fp16)[name = string("hidden_states_515_cast_fp16")]; + tensor inputs_417_cast_fp16 = add(x = inputs_415_cast_fp16, y = hidden_states_515_cast_fp16)[name = string("inputs_417_cast_fp16")]; + int32 var_15229 = const()[name = string("op_15229"), val = int32(1)]; + bool key_caches_21_interleave_0 = const()[name = string("key_caches_21_interleave_0"), val = bool(false)]; + tensor key_caches_21_cast_fp16 = concat(axis = var_15229, interleave = key_caches_21_interleave_0, values = (key_183_cast_fp16, key_187_cast_fp16, key_191_cast_fp16, key_195_cast_fp16, key_199_cast_fp16))[name = string("key_caches_21_cast_fp16")]; + int32 var_15232 = const()[name = string("op_15232"), val = int32(1)]; + bool value_caches_21_interleave_0 = const()[name = string("value_caches_21_interleave_0"), val = bool(false)]; + tensor value_caches_21_cast_fp16 = concat(axis = var_15232, interleave = value_caches_21_interleave_0, values = (value_91_cast_fp16, value_93_cast_fp16, value_95_cast_fp16, value_97_cast_fp16, value_99_cast_fp16))[name = string("value_caches_21_cast_fp16")]; + tensor inputs_sq_417_cast_fp16 = mul(x = inputs_417_cast_fp16, y = inputs_417_cast_fp16)[name = string("inputs_sq_417_cast_fp16")]; + tensor variance_417_axes_0 = const()[name = string("variance_417_axes_0"), val = tensor([1])]; + bool variance_417_keep_dims_0 = const()[name = string("variance_417_keep_dims_0"), val = bool(true)]; + tensor variance_417_cast_fp16 = reduce_mean(axes = variance_417_axes_0, keep_dims = variance_417_keep_dims_0, x = inputs_sq_417_cast_fp16)[name = string("variance_417_cast_fp16")]; + fp16 var_15242_to_fp16 = const()[name = string("op_15242_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15243_cast_fp16 = add(x = variance_417_cast_fp16, y = var_15242_to_fp16)[name = string("op_15243_cast_fp16")]; + fp32 var_15244_epsilon_0 = const()[name = string("op_15244_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15244_cast_fp16 = rsqrt(epsilon = var_15244_epsilon_0, x = var_15243_cast_fp16)[name = string("op_15244_cast_fp16")]; + tensor hidden_states_517_cast_fp16 = mul(x = inputs_417_cast_fp16, y = var_15244_cast_fp16)[name = string("hidden_states_517_cast_fp16")]; + tensor input_433_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_517_cast_fp16)[name = string("input_433_cast_fp16")]; + string logits_33_pad_type_0 = const()[name = string("logits_33_pad_type_0"), val = string("valid")]; + tensor logits_33_strides_0 = const()[name = string("logits_33_strides_0"), val = tensor([1, 1])]; + tensor logits_33_pad_0 = const()[name = string("logits_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_33_dilations_0 = const()[name = string("logits_33_dilations_0"), val = tensor([1, 1])]; + int32 logits_33_groups_0 = const()[name = string("logits_33_groups_0"), val = int32(1)]; + tensor lm_heads_8_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(97588928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(99686144))))[name = string("lm_heads_8_weight_to_fp16_palettized")]; + tensor logits_33_cast_fp16 = conv(dilations = logits_33_dilations_0, groups = logits_33_groups_0, pad = logits_33_pad_0, pad_type = logits_33_pad_type_0, strides = logits_33_strides_0, weight = lm_heads_8_weight_to_fp16_palettized, x = input_433_cast_fp16)[name = string("logits_33_cast_fp16")]; + tensor var_15262 = const()[name = string("op_15262"), val = tensor([1, 2048])]; + tensor logits_35_cast_fp16 = reshape(shape = var_15262, x = logits_33_cast_fp16)[name = string("logits_35_cast_fp16")]; + tensor scaled_logits_17_cast_fp16 = real_div(x = logits_35_cast_fp16, y = temperature)[name = string("scaled_logits_17_cast_fp16")]; + int32 var_15272 = const()[name = string("op_15272"), val = int32(100)]; + int32 top_values_17_axis_0 = const()[name = string("top_values_17_axis_0"), val = int32(1)]; + bool top_values_17_ascending_0 = const()[name = string("top_values_17_ascending_0"), val = bool(false)]; + bool top_values_17_sort_0 = const()[name = string("top_values_17_sort_0"), val = bool(true)]; + bool top_values_17_return_indices_0 = const()[name = string("top_values_17_return_indices_0"), val = bool(true)]; + string top_values_17_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_17_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_17_cast_fp16_cast_uint16_0, tensor top_values_17_cast_fp16_cast_uint16_1 = topk(ascending = top_values_17_ascending_0, axis = top_values_17_axis_0, k = var_15272, output_indices_dtype = top_values_17_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_17_return_indices_0, sort = top_values_17_sort_0, x = scaled_logits_17_cast_fp16)[name = string("top_values_17_cast_fp16_cast_uint16")]; + tensor var_15278_cast_fp16 = mul(x = top_values_17_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_15278_cast_fp16")]; + tensor var_15282_cast_fp16 = add(x = var_15278_cast_fp16, y = var_3001_cast_fp16)[name = string("op_15282_cast_fp16")]; + tensor reduce_min_8_axes_0 = const()[name = string("reduce_min_8_axes_0"), val = tensor([1])]; + bool reduce_min_8_keep_dims_0 = const()[name = string("reduce_min_8_keep_dims_0"), val = bool(true)]; + tensor reduce_min_8_cast_fp16 = reduce_min(axes = reduce_min_8_axes_0, keep_dims = reduce_min_8_keep_dims_0, x = var_15282_cast_fp16)[name = string("reduce_min_8_cast_fp16")]; + tensor var_15285_cast_fp16 = greater_equal(x = scaled_logits_17_cast_fp16, y = reduce_min_8_cast_fp16)[name = string("op_15285_cast_fp16")]; + fp16 var_15286_value_0_to_fp16 = const()[name = string("op_15286_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_15286_cast_fp16 = fill_like(ref_tensor = scaled_logits_17_cast_fp16, value = var_15286_value_0_to_fp16)[name = string("op_15286_cast_fp16")]; + tensor masked_logits_17_cast_fp16 = select(a = scaled_logits_17_cast_fp16, b = var_15286_cast_fp16, cond = var_15285_cast_fp16)[name = string("masked_logits_17_cast_fp16")]; + tensor var_15290_begin_0 = const()[name = string("op_15290_begin_0"), val = tensor([8, 0])]; + tensor var_15290_end_0 = const()[name = string("op_15290_end_0"), val = tensor([9, 2048])]; + tensor var_15290_end_mask_0 = const()[name = string("op_15290_end_mask_0"), val = tensor([false, true])]; + tensor var_15290_squeeze_mask_0 = const()[name = string("op_15290_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_15290_cast_fp16 = slice_by_index(begin = var_15290_begin_0, end = var_15290_end_0, end_mask = var_15290_end_mask_0, squeeze_mask = var_15290_squeeze_mask_0, x = gumbel)[name = string("op_15290_cast_fp16")]; + tensor var_15293 = const()[name = string("op_15293"), val = tensor([1, 2048])]; + tensor var_15294_cast_fp16 = reshape(shape = var_15293, x = var_15290_cast_fp16)[name = string("op_15294_cast_fp16")]; + tensor noisy_logits_17_cast_fp16 = add(x = masked_logits_17_cast_fp16, y = var_15294_cast_fp16)[name = string("noisy_logits_17_cast_fp16")]; + int32 code_17_axis_0 = const()[name = string("code_17_axis_0"), val = int32(1)]; + bool code_17_keep_dims_0 = const()[name = string("code_17_keep_dims_0"), val = bool(false)]; + string code_17_output_dtype_0 = const()[name = string("code_17_output_dtype_0"), val = string("int32")]; + tensor code_17_cast_fp16 = reduce_argmax(axis = code_17_axis_0, keep_dims = code_17_keep_dims_0, output_dtype = code_17_output_dtype_0, x = noisy_logits_17_cast_fp16)[name = string("code_17_cast_fp16")]; + int32 var_15305 = const()[name = string("op_15305"), val = int32(16384)]; + tensor input_435 = add(x = code_17_cast_fp16, y = var_15305)[name = string("input_435")]; + int32 code_embed_33_axis_0 = const()[name = string("code_embed_33_axis_0"), val = int32(0)]; + int32 code_embed_33_batch_dims_0 = const()[name = string("code_embed_33_batch_dims_0"), val = int32(0)]; + bool code_embed_33_validate_indices_0 = const()[name = string("code_embed_33_validate_indices_0"), val = bool(false)]; + string input_435_to_uint16_dtype_0 = const()[name = string("input_435_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_435_to_uint16 = cast(dtype = input_435_to_uint16_dtype_0, x = input_435)[name = string("cast_6")]; + tensor code_embed_33_cast_fp16_cast_uint16 = gather(axis = code_embed_33_axis_0, batch_dims = code_embed_33_batch_dims_0, indices = input_435_to_uint16, validate_indices = code_embed_33_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_33_cast_fp16_cast_uint16")]; + tensor var_15309 = const()[name = string("op_15309"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_35_cast_fp16 = reshape(shape = var_15309, x = code_embed_33_cast_fp16_cast_uint16)[name = string("code_embed_35_cast_fp16")]; + tensor embed_sum_19_cast_fp16 = add(x = embed_sum_17_cast_fp16, y = code_embed_35_cast_fp16)[name = string("embed_sum_19_cast_fp16")]; + string inputs_419_pad_type_0 = const()[name = string("inputs_419_pad_type_0"), val = string("valid")]; + tensor inputs_419_strides_0 = const()[name = string("inputs_419_strides_0"), val = tensor([1, 1])]; + tensor inputs_419_pad_0 = const()[name = string("inputs_419_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_419_dilations_0 = const()[name = string("inputs_419_dilations_0"), val = tensor([1, 1])]; + int32 inputs_419_groups_0 = const()[name = string("inputs_419_groups_0"), val = int32(1)]; + tensor inputs_419_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_419_dilations_0, groups = inputs_419_groups_0, pad = inputs_419_pad_0, pad_type = inputs_419_pad_type_0, strides = inputs_419_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_35_cast_fp16)[name = string("inputs_419_cast_fp16")]; + tensor obj_443_begin_0 = const()[name = string("obj_443_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_443_end_0 = const()[name = string("obj_443_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_443_end_mask_0 = const()[name = string("obj_443_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_443_cast_fp16 = slice_by_index(begin = obj_443_begin_0, end = obj_443_end_0, end_mask = obj_443_end_mask_0, x = key_caches_21_cast_fp16)[name = string("obj_443_cast_fp16")]; + tensor obj_445_begin_0 = const()[name = string("obj_445_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_445_end_0 = const()[name = string("obj_445_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_445_end_mask_0 = const()[name = string("obj_445_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_445_cast_fp16 = slice_by_index(begin = obj_445_begin_0, end = obj_445_end_0, end_mask = obj_445_end_mask_0, x = value_caches_21_cast_fp16)[name = string("obj_445_cast_fp16")]; + int32 var_15414 = const()[name = string("op_15414"), val = int32(3)]; + int32 var_15424 = const()[name = string("op_15424"), val = int32(-2)]; + tensor inputs_sq_419_cast_fp16 = mul(x = inputs_419_cast_fp16, y = inputs_419_cast_fp16)[name = string("inputs_sq_419_cast_fp16")]; + tensor variance_419_axes_0 = const()[name = string("variance_419_axes_0"), val = tensor([1])]; + bool variance_419_keep_dims_0 = const()[name = string("variance_419_keep_dims_0"), val = bool(true)]; + tensor variance_419_cast_fp16 = reduce_mean(axes = variance_419_axes_0, keep_dims = variance_419_keep_dims_0, x = inputs_sq_419_cast_fp16)[name = string("variance_419_cast_fp16")]; + fp16 var_15438_to_fp16 = const()[name = string("op_15438_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15439_cast_fp16 = add(x = variance_419_cast_fp16, y = var_15438_to_fp16)[name = string("op_15439_cast_fp16")]; + fp32 var_15440_epsilon_0 = const()[name = string("op_15440_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15440_cast_fp16 = rsqrt(epsilon = var_15440_epsilon_0, x = var_15439_cast_fp16)[name = string("op_15440_cast_fp16")]; + tensor hidden_states_519_cast_fp16 = mul(x = inputs_419_cast_fp16, y = var_15440_cast_fp16)[name = string("hidden_states_519_cast_fp16")]; + tensor obj_441_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_519_cast_fp16)[name = string("obj_441_cast_fp16")]; + string query_301_pad_type_0 = const()[name = string("query_301_pad_type_0"), val = string("valid")]; + tensor query_301_strides_0 = const()[name = string("query_301_strides_0"), val = tensor([1, 1])]; + tensor query_301_pad_0 = const()[name = string("query_301_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_301_dilations_0 = const()[name = string("query_301_dilations_0"), val = tensor([1, 1])]; + int32 query_301_groups_0 = const()[name = string("query_301_groups_0"), val = int32(1)]; + tensor query_301_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_301_dilations_0, groups = query_301_groups_0, pad = query_301_pad_0, pad_type = query_301_pad_type_0, strides = query_301_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_441_cast_fp16)[name = string("query_301_cast_fp16")]; + string current_key_201_pad_type_0 = const()[name = string("current_key_201_pad_type_0"), val = string("valid")]; + tensor current_key_201_strides_0 = const()[name = string("current_key_201_strides_0"), val = tensor([1, 1])]; + tensor current_key_201_pad_0 = const()[name = string("current_key_201_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_201_dilations_0 = const()[name = string("current_key_201_dilations_0"), val = tensor([1, 1])]; + int32 current_key_201_groups_0 = const()[name = string("current_key_201_groups_0"), val = int32(1)]; + tensor current_key_201_cast_fp16 = conv(dilations = current_key_201_dilations_0, groups = current_key_201_groups_0, pad = current_key_201_pad_0, pad_type = current_key_201_pad_type_0, strides = current_key_201_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_441_cast_fp16)[name = string("current_key_201_cast_fp16")]; + string current_value_101_pad_type_0 = const()[name = string("current_value_101_pad_type_0"), val = string("valid")]; + tensor current_value_101_strides_0 = const()[name = string("current_value_101_strides_0"), val = tensor([1, 1])]; + tensor current_value_101_pad_0 = const()[name = string("current_value_101_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_101_dilations_0 = const()[name = string("current_value_101_dilations_0"), val = tensor([1, 1])]; + int32 current_value_101_groups_0 = const()[name = string("current_value_101_groups_0"), val = int32(1)]; + tensor current_value_101_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_101_dilations_0, groups = current_value_101_groups_0, pad = current_value_101_pad_0, pad_type = current_value_101_pad_type_0, strides = current_value_101_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_441_cast_fp16)[name = string("current_value_101_cast_fp16")]; + tensor var_15477 = const()[name = string("op_15477"), val = tensor([16, 128, 1, 1])]; + tensor inputs_421_cast_fp16 = reshape(shape = var_15477, x = query_301_cast_fp16)[name = string("inputs_421_cast_fp16")]; + tensor inputs_sq_421_cast_fp16 = mul(x = inputs_421_cast_fp16, y = inputs_421_cast_fp16)[name = string("inputs_sq_421_cast_fp16")]; + tensor variance_421_axes_0 = const()[name = string("variance_421_axes_0"), val = tensor([1])]; + bool variance_421_keep_dims_0 = const()[name = string("variance_421_keep_dims_0"), val = bool(true)]; + tensor variance_421_cast_fp16 = reduce_mean(axes = variance_421_axes_0, keep_dims = variance_421_keep_dims_0, x = inputs_sq_421_cast_fp16)[name = string("variance_421_cast_fp16")]; + fp16 var_15483_to_fp16 = const()[name = string("op_15483_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15484_cast_fp16 = add(x = variance_421_cast_fp16, y = var_15483_to_fp16)[name = string("op_15484_cast_fp16")]; + fp32 var_15485_epsilon_0 = const()[name = string("op_15485_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15485_cast_fp16 = rsqrt(epsilon = var_15485_epsilon_0, x = var_15484_cast_fp16)[name = string("op_15485_cast_fp16")]; + tensor hidden_states_521_cast_fp16 = mul(x = inputs_421_cast_fp16, y = var_15485_cast_fp16)[name = string("hidden_states_521_cast_fp16")]; + tensor query_normed_101_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_521_cast_fp16)[name = string("query_normed_101_cast_fp16")]; + tensor var_15493 = const()[name = string("op_15493"), val = tensor([8, 128, 1, 1])]; + tensor inputs_423_cast_fp16 = reshape(shape = var_15493, x = current_key_201_cast_fp16)[name = string("inputs_423_cast_fp16")]; + tensor inputs_sq_423_cast_fp16 = mul(x = inputs_423_cast_fp16, y = inputs_423_cast_fp16)[name = string("inputs_sq_423_cast_fp16")]; + tensor variance_423_axes_0 = const()[name = string("variance_423_axes_0"), val = tensor([1])]; + bool variance_423_keep_dims_0 = const()[name = string("variance_423_keep_dims_0"), val = bool(true)]; + tensor variance_423_cast_fp16 = reduce_mean(axes = variance_423_axes_0, keep_dims = variance_423_keep_dims_0, x = inputs_sq_423_cast_fp16)[name = string("variance_423_cast_fp16")]; + fp16 var_15499_to_fp16 = const()[name = string("op_15499_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15500_cast_fp16 = add(x = variance_423_cast_fp16, y = var_15499_to_fp16)[name = string("op_15500_cast_fp16")]; + fp32 var_15501_epsilon_0 = const()[name = string("op_15501_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15501_cast_fp16 = rsqrt(epsilon = var_15501_epsilon_0, x = var_15500_cast_fp16)[name = string("op_15501_cast_fp16")]; + tensor hidden_states_523_cast_fp16 = mul(x = inputs_423_cast_fp16, y = var_15501_cast_fp16)[name = string("hidden_states_523_cast_fp16")]; + tensor current_key_normed_101_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_523_cast_fp16)[name = string("current_key_normed_101_cast_fp16")]; + tensor var_15519 = const()[name = string("op_15519"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_401_cast_fp16 = reshape(shape = var_15519, x = query_normed_101_cast_fp16)[name = string("mh_q_401_cast_fp16")]; + tensor var_15521 = const()[name = string("op_15521"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_401_cast_fp16 = reshape(shape = var_15521, x = current_key_normed_101_cast_fp16)[name = string("mh_k_401_cast_fp16")]; + tensor cos_101_to_fp16 = const()[name = string("cos_101_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175198784)))]; + tensor var_15525_cast_fp16 = mul(x = mh_q_401_cast_fp16, y = cos_101_to_fp16)[name = string("op_15525_cast_fp16")]; + tensor var_15530_begin_0 = const()[name = string("op_15530_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15530_end_0 = const()[name = string("op_15530_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_15530_end_mask_0 = const()[name = string("op_15530_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_15530_cast_fp16 = slice_by_index(begin = var_15530_begin_0, end = var_15530_end_0, end_mask = var_15530_end_mask_0, x = mh_q_401_cast_fp16)[name = string("op_15530_cast_fp16")]; + tensor var_15536_begin_0 = const()[name = string("op_15536_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_15536_end_0 = const()[name = string("op_15536_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_15536_end_mask_0 = const()[name = string("op_15536_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_15536_cast_fp16 = slice_by_index(begin = var_15536_begin_0, end = var_15536_end_0, end_mask = var_15536_end_mask_0, x = mh_q_401_cast_fp16)[name = string("op_15536_cast_fp16")]; + fp16 const_1024_promoted_to_fp16 = const()[name = string("const_1024_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15538_cast_fp16 = mul(x = var_15536_cast_fp16, y = const_1024_promoted_to_fp16)[name = string("op_15538_cast_fp16")]; + bool var_15540_interleave_0 = const()[name = string("op_15540_interleave_0"), val = bool(false)]; + tensor var_15540_cast_fp16 = concat(axis = var_15424, interleave = var_15540_interleave_0, values = (var_15538_cast_fp16, var_15530_cast_fp16))[name = string("op_15540_cast_fp16")]; + tensor sin_101_to_fp16 = const()[name = string("sin_101_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199104)))]; + tensor var_15541_cast_fp16 = mul(x = var_15540_cast_fp16, y = sin_101_to_fp16)[name = string("op_15541_cast_fp16")]; + tensor mh_q_403_cast_fp16 = add(x = var_15525_cast_fp16, y = var_15541_cast_fp16)[name = string("mh_q_403_cast_fp16")]; + tensor var_15543_cast_fp16 = mul(x = mh_k_401_cast_fp16, y = cos_101_to_fp16)[name = string("op_15543_cast_fp16")]; + tensor var_15548_begin_0 = const()[name = string("op_15548_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15548_end_0 = const()[name = string("op_15548_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_15548_end_mask_0 = const()[name = string("op_15548_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_15548_cast_fp16 = slice_by_index(begin = var_15548_begin_0, end = var_15548_end_0, end_mask = var_15548_end_mask_0, x = mh_k_401_cast_fp16)[name = string("op_15548_cast_fp16")]; + tensor var_15554_begin_0 = const()[name = string("op_15554_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_15554_end_0 = const()[name = string("op_15554_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_15554_end_mask_0 = const()[name = string("op_15554_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_15554_cast_fp16 = slice_by_index(begin = var_15554_begin_0, end = var_15554_end_0, end_mask = var_15554_end_mask_0, x = mh_k_401_cast_fp16)[name = string("op_15554_cast_fp16")]; + fp16 const_1027_promoted_to_fp16 = const()[name = string("const_1027_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15556_cast_fp16 = mul(x = var_15554_cast_fp16, y = const_1027_promoted_to_fp16)[name = string("op_15556_cast_fp16")]; + bool var_15558_interleave_0 = const()[name = string("op_15558_interleave_0"), val = bool(false)]; + tensor var_15558_cast_fp16 = concat(axis = var_15424, interleave = var_15558_interleave_0, values = (var_15556_cast_fp16, var_15548_cast_fp16))[name = string("op_15558_cast_fp16")]; + tensor var_15559_cast_fp16 = mul(x = var_15558_cast_fp16, y = sin_101_to_fp16)[name = string("op_15559_cast_fp16")]; + tensor mh_k_403_cast_fp16 = add(x = var_15543_cast_fp16, y = var_15559_cast_fp16)[name = string("mh_k_403_cast_fp16")]; + tensor var_15563 = const()[name = string("op_15563"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_203_cast_fp16 = reshape(shape = var_15563, x = mh_k_403_cast_fp16)[name = string("current_key_203_cast_fp16")]; + tensor var_15569_to_fp16 = const()[name = string("op_15569_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199424)))]; + tensor var_15570_cast_fp16 = mul(x = obj_443_cast_fp16, y = var_15569_to_fp16)[name = string("op_15570_cast_fp16")]; + tensor var_15567_to_fp16 = const()[name = string("op_15567_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199552)))]; + tensor var_15571_cast_fp16 = mul(x = current_key_203_cast_fp16, y = var_15567_to_fp16)[name = string("op_15571_cast_fp16")]; + tensor key_203_cast_fp16 = add(x = var_15570_cast_fp16, y = var_15571_cast_fp16)[name = string("key_203_cast_fp16")]; + tensor var_15573_to_fp16 = const()[name = string("op_15573_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199424)))]; + tensor var_15574_cast_fp16 = mul(x = obj_445_cast_fp16, y = var_15573_to_fp16)[name = string("op_15574_cast_fp16")]; + tensor var_15575_cast_fp16 = mul(x = current_value_101_cast_fp16, y = var_15567_to_fp16)[name = string("op_15575_cast_fp16")]; + tensor value_101_cast_fp16 = add(x = var_15574_cast_fp16, y = var_15575_cast_fp16)[name = string("value_101_cast_fp16")]; + fp16 var_15582_to_fp16 = const()[name = string("op_15582_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_407_cast_fp16 = mul(x = mh_q_403_cast_fp16, y = var_15582_to_fp16)[name = string("mh_q_407_cast_fp16")]; + tensor var_15584 = const()[name = string("op_15584"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_405_cast_fp16 = reshape(shape = var_15584, x = key_203_cast_fp16)[name = string("mh_k_405_cast_fp16")]; + tensor var_15586 = const()[name = string("op_15586"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_201_cast_fp16 = reshape(shape = var_15586, x = value_101_cast_fp16)[name = string("mh_v_201_cast_fp16")]; + tensor transpose_200_perm_0 = const()[name = string("transpose_200_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_100_reps_0 = const()[name = string("tile_100_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_200_cast_fp16 = transpose(perm = transpose_200_perm_0, x = mh_k_405_cast_fp16)[name = string("transpose_179")]; + tensor tile_100_cast_fp16 = tile(reps = tile_100_reps_0, x = transpose_200_cast_fp16)[name = string("tile_100_cast_fp16")]; + tensor concat_253 = const()[name = string("concat_253"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_200_cast_fp16 = reshape(shape = concat_253, x = tile_100_cast_fp16)[name = string("reshape_200_cast_fp16")]; + tensor transpose_201_perm_0 = const()[name = string("transpose_201_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_254 = const()[name = string("concat_254"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_201_cast_fp16 = transpose(perm = transpose_201_perm_0, x = reshape_200_cast_fp16)[name = string("transpose_178")]; + tensor reshape_201_cast_fp16 = reshape(shape = concat_254, x = transpose_201_cast_fp16)[name = string("reshape_201_cast_fp16")]; + tensor transpose_202_perm_0 = const()[name = string("transpose_202_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_101_reps_0 = const()[name = string("tile_101_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_202_cast_fp16 = transpose(perm = transpose_202_perm_0, x = mh_v_201_cast_fp16)[name = string("transpose_177")]; + tensor tile_101_cast_fp16 = tile(reps = tile_101_reps_0, x = transpose_202_cast_fp16)[name = string("tile_101_cast_fp16")]; + tensor concat_255 = const()[name = string("concat_255"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_202_cast_fp16 = reshape(shape = concat_255, x = tile_101_cast_fp16)[name = string("reshape_202_cast_fp16")]; + tensor transpose_203_perm_0 = const()[name = string("transpose_203_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_256 = const()[name = string("concat_256"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_203_cast_fp16 = transpose(perm = transpose_203_perm_0, x = reshape_202_cast_fp16)[name = string("transpose_176")]; + tensor reshape_203_cast_fp16 = reshape(shape = concat_256, x = transpose_203_cast_fp16)[name = string("reshape_203_cast_fp16")]; + tensor transpose_517_perm_0 = const()[name = string("transpose_517_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_301_transpose_x_1 = const()[name = string("mh_w_301_transpose_x_1"), val = bool(true)]; + bool mh_w_301_transpose_y_1 = const()[name = string("mh_w_301_transpose_y_1"), val = bool(false)]; + tensor transpose_517_cast_fp16 = transpose(perm = transpose_517_perm_0, x = reshape_201_cast_fp16)[name = string("transpose_175")]; + tensor mh_w_301_cast_fp16 = matmul(transpose_x = mh_w_301_transpose_x_1, transpose_y = mh_w_301_transpose_y_1, x = mh_q_407_cast_fp16, y = transpose_517_cast_fp16)[name = string("mh_w_301_cast_fp16")]; + tensor var_15594_to_fp16 = const()[name = string("op_15594_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199680)))]; + tensor mh_w_303_cast_fp16 = add(x = mh_w_301_cast_fp16, y = var_15594_to_fp16)[name = string("mh_w_303_cast_fp16")]; + tensor mh_w_305_cast_fp16 = softmax(axis = var_15414, x = mh_w_303_cast_fp16)[name = string("mh_w_305_cast_fp16")]; + tensor transpose_518_perm_0 = const()[name = string("transpose_518_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_101_transpose_x_1 = const()[name = string("attn_101_transpose_x_1"), val = bool(false)]; + bool attn_101_transpose_y_1 = const()[name = string("attn_101_transpose_y_1"), val = bool(true)]; + tensor transpose_518_cast_fp16 = transpose(perm = transpose_518_perm_0, x = reshape_203_cast_fp16)[name = string("transpose_174")]; + tensor attn_101_cast_fp16 = matmul(transpose_x = attn_101_transpose_x_1, transpose_y = attn_101_transpose_y_1, x = transpose_518_cast_fp16, y = mh_w_305_cast_fp16)[name = string("attn_101_cast_fp16")]; + tensor var_15600 = const()[name = string("op_15600"), val = tensor([1, 2048, 1, 1])]; + tensor input_437_cast_fp16 = reshape(shape = var_15600, x = attn_101_cast_fp16)[name = string("input_437_cast_fp16")]; + string obj_451_pad_type_0 = const()[name = string("obj_451_pad_type_0"), val = string("valid")]; + tensor obj_451_strides_0 = const()[name = string("obj_451_strides_0"), val = tensor([1, 1])]; + tensor obj_451_pad_0 = const()[name = string("obj_451_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_451_dilations_0 = const()[name = string("obj_451_dilations_0"), val = tensor([1, 1])]; + int32 obj_451_groups_0 = const()[name = string("obj_451_groups_0"), val = int32(1)]; + tensor obj_451_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_451_dilations_0, groups = obj_451_groups_0, pad = obj_451_pad_0, pad_type = obj_451_pad_type_0, strides = obj_451_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_437_cast_fp16)[name = string("obj_451_cast_fp16")]; + tensor inputs_425_cast_fp16 = add(x = inputs_419_cast_fp16, y = obj_451_cast_fp16)[name = string("inputs_425_cast_fp16")]; + tensor inputs_sq_425_cast_fp16 = mul(x = inputs_425_cast_fp16, y = inputs_425_cast_fp16)[name = string("inputs_sq_425_cast_fp16")]; + tensor variance_425_axes_0 = const()[name = string("variance_425_axes_0"), val = tensor([1])]; + bool variance_425_keep_dims_0 = const()[name = string("variance_425_keep_dims_0"), val = bool(true)]; + tensor variance_425_cast_fp16 = reduce_mean(axes = variance_425_axes_0, keep_dims = variance_425_keep_dims_0, x = inputs_sq_425_cast_fp16)[name = string("variance_425_cast_fp16")]; + fp16 var_15618_to_fp16 = const()[name = string("op_15618_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15619_cast_fp16 = add(x = variance_425_cast_fp16, y = var_15618_to_fp16)[name = string("op_15619_cast_fp16")]; + fp32 var_15620_epsilon_0 = const()[name = string("op_15620_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15620_cast_fp16 = rsqrt(epsilon = var_15620_epsilon_0, x = var_15619_cast_fp16)[name = string("op_15620_cast_fp16")]; + tensor hidden_states_525_cast_fp16 = mul(x = inputs_425_cast_fp16, y = var_15620_cast_fp16)[name = string("hidden_states_525_cast_fp16")]; + tensor input_439_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_525_cast_fp16)[name = string("input_439_cast_fp16")]; + string input_441_pad_type_0 = const()[name = string("input_441_pad_type_0"), val = string("valid")]; + tensor input_441_strides_0 = const()[name = string("input_441_strides_0"), val = tensor([1, 1])]; + tensor input_441_pad_0 = const()[name = string("input_441_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_441_dilations_0 = const()[name = string("input_441_dilations_0"), val = tensor([1, 1])]; + int32 input_441_groups_0 = const()[name = string("input_441_groups_0"), val = int32(1)]; + tensor input_441_cast_fp16 = conv(dilations = input_441_dilations_0, groups = input_441_groups_0, pad = input_441_pad_0, pad_type = input_441_pad_type_0, strides = input_441_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_439_cast_fp16)[name = string("input_441_cast_fp16")]; + tensor var_15634_cast_fp16 = silu(x = input_441_cast_fp16)[name = string("op_15634_cast_fp16")]; + string var_15640_pad_type_0 = const()[name = string("op_15640_pad_type_0"), val = string("valid")]; + tensor var_15640_strides_0 = const()[name = string("op_15640_strides_0"), val = tensor([1, 1])]; + tensor var_15640_pad_0 = const()[name = string("op_15640_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15640_dilations_0 = const()[name = string("op_15640_dilations_0"), val = tensor([1, 1])]; + int32 var_15640_groups_0 = const()[name = string("op_15640_groups_0"), val = int32(1)]; + tensor var_15640_cast_fp16 = conv(dilations = var_15640_dilations_0, groups = var_15640_groups_0, pad = var_15640_pad_0, pad_type = var_15640_pad_type_0, strides = var_15640_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_439_cast_fp16)[name = string("op_15640_cast_fp16")]; + tensor input_443_cast_fp16 = mul(x = var_15634_cast_fp16, y = var_15640_cast_fp16)[name = string("input_443_cast_fp16")]; + string hidden_states_527_pad_type_0 = const()[name = string("hidden_states_527_pad_type_0"), val = string("valid")]; + tensor hidden_states_527_strides_0 = const()[name = string("hidden_states_527_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_527_pad_0 = const()[name = string("hidden_states_527_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_527_dilations_0 = const()[name = string("hidden_states_527_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_527_groups_0 = const()[name = string("hidden_states_527_groups_0"), val = int32(1)]; + tensor hidden_states_527_cast_fp16 = conv(dilations = hidden_states_527_dilations_0, groups = hidden_states_527_groups_0, pad = hidden_states_527_pad_0, pad_type = hidden_states_527_pad_type_0, strides = hidden_states_527_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_443_cast_fp16)[name = string("hidden_states_527_cast_fp16")]; + tensor inputs_427_cast_fp16 = add(x = inputs_425_cast_fp16, y = hidden_states_527_cast_fp16)[name = string("inputs_427_cast_fp16")]; + tensor obj_455_begin_0 = const()[name = string("obj_455_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_455_end_0 = const()[name = string("obj_455_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_455_end_mask_0 = const()[name = string("obj_455_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_455_cast_fp16 = slice_by_index(begin = obj_455_begin_0, end = obj_455_end_0, end_mask = obj_455_end_mask_0, x = key_caches_21_cast_fp16)[name = string("obj_455_cast_fp16")]; + tensor obj_457_begin_0 = const()[name = string("obj_457_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_457_end_0 = const()[name = string("obj_457_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_457_end_mask_0 = const()[name = string("obj_457_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_457_cast_fp16 = slice_by_index(begin = obj_457_begin_0, end = obj_457_end_0, end_mask = obj_457_end_mask_0, x = value_caches_21_cast_fp16)[name = string("obj_457_cast_fp16")]; + int32 var_15688 = const()[name = string("op_15688"), val = int32(3)]; + int32 var_15698 = const()[name = string("op_15698"), val = int32(-2)]; + tensor inputs_sq_427_cast_fp16 = mul(x = inputs_427_cast_fp16, y = inputs_427_cast_fp16)[name = string("inputs_sq_427_cast_fp16")]; + tensor variance_427_axes_0 = const()[name = string("variance_427_axes_0"), val = tensor([1])]; + bool variance_427_keep_dims_0 = const()[name = string("variance_427_keep_dims_0"), val = bool(true)]; + tensor variance_427_cast_fp16 = reduce_mean(axes = variance_427_axes_0, keep_dims = variance_427_keep_dims_0, x = inputs_sq_427_cast_fp16)[name = string("variance_427_cast_fp16")]; + fp16 var_15712_to_fp16 = const()[name = string("op_15712_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15713_cast_fp16 = add(x = variance_427_cast_fp16, y = var_15712_to_fp16)[name = string("op_15713_cast_fp16")]; + fp32 var_15714_epsilon_0 = const()[name = string("op_15714_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15714_cast_fp16 = rsqrt(epsilon = var_15714_epsilon_0, x = var_15713_cast_fp16)[name = string("op_15714_cast_fp16")]; + tensor hidden_states_529_cast_fp16 = mul(x = inputs_427_cast_fp16, y = var_15714_cast_fp16)[name = string("hidden_states_529_cast_fp16")]; + tensor obj_453_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_529_cast_fp16)[name = string("obj_453_cast_fp16")]; + string query_307_pad_type_0 = const()[name = string("query_307_pad_type_0"), val = string("valid")]; + tensor query_307_strides_0 = const()[name = string("query_307_strides_0"), val = tensor([1, 1])]; + tensor query_307_pad_0 = const()[name = string("query_307_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_307_dilations_0 = const()[name = string("query_307_dilations_0"), val = tensor([1, 1])]; + int32 query_307_groups_0 = const()[name = string("query_307_groups_0"), val = int32(1)]; + tensor query_307_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_307_dilations_0, groups = query_307_groups_0, pad = query_307_pad_0, pad_type = query_307_pad_type_0, strides = query_307_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_453_cast_fp16)[name = string("query_307_cast_fp16")]; + string current_key_205_pad_type_0 = const()[name = string("current_key_205_pad_type_0"), val = string("valid")]; + tensor current_key_205_strides_0 = const()[name = string("current_key_205_strides_0"), val = tensor([1, 1])]; + tensor current_key_205_pad_0 = const()[name = string("current_key_205_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_205_dilations_0 = const()[name = string("current_key_205_dilations_0"), val = tensor([1, 1])]; + int32 current_key_205_groups_0 = const()[name = string("current_key_205_groups_0"), val = int32(1)]; + tensor current_key_205_cast_fp16 = conv(dilations = current_key_205_dilations_0, groups = current_key_205_groups_0, pad = current_key_205_pad_0, pad_type = current_key_205_pad_type_0, strides = current_key_205_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_453_cast_fp16)[name = string("current_key_205_cast_fp16")]; + string current_value_103_pad_type_0 = const()[name = string("current_value_103_pad_type_0"), val = string("valid")]; + tensor current_value_103_strides_0 = const()[name = string("current_value_103_strides_0"), val = tensor([1, 1])]; + tensor current_value_103_pad_0 = const()[name = string("current_value_103_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_103_dilations_0 = const()[name = string("current_value_103_dilations_0"), val = tensor([1, 1])]; + int32 current_value_103_groups_0 = const()[name = string("current_value_103_groups_0"), val = int32(1)]; + tensor current_value_103_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_103_dilations_0, groups = current_value_103_groups_0, pad = current_value_103_pad_0, pad_type = current_value_103_pad_type_0, strides = current_value_103_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_453_cast_fp16)[name = string("current_value_103_cast_fp16")]; + tensor var_15751 = const()[name = string("op_15751"), val = tensor([16, 128, 1, 1])]; + tensor inputs_429_cast_fp16 = reshape(shape = var_15751, x = query_307_cast_fp16)[name = string("inputs_429_cast_fp16")]; + tensor inputs_sq_429_cast_fp16 = mul(x = inputs_429_cast_fp16, y = inputs_429_cast_fp16)[name = string("inputs_sq_429_cast_fp16")]; + tensor variance_429_axes_0 = const()[name = string("variance_429_axes_0"), val = tensor([1])]; + bool variance_429_keep_dims_0 = const()[name = string("variance_429_keep_dims_0"), val = bool(true)]; + tensor variance_429_cast_fp16 = reduce_mean(axes = variance_429_axes_0, keep_dims = variance_429_keep_dims_0, x = inputs_sq_429_cast_fp16)[name = string("variance_429_cast_fp16")]; + fp16 var_15757_to_fp16 = const()[name = string("op_15757_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15758_cast_fp16 = add(x = variance_429_cast_fp16, y = var_15757_to_fp16)[name = string("op_15758_cast_fp16")]; + fp32 var_15759_epsilon_0 = const()[name = string("op_15759_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15759_cast_fp16 = rsqrt(epsilon = var_15759_epsilon_0, x = var_15758_cast_fp16)[name = string("op_15759_cast_fp16")]; + tensor hidden_states_531_cast_fp16 = mul(x = inputs_429_cast_fp16, y = var_15759_cast_fp16)[name = string("hidden_states_531_cast_fp16")]; + tensor query_normed_103_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_531_cast_fp16)[name = string("query_normed_103_cast_fp16")]; + tensor var_15767 = const()[name = string("op_15767"), val = tensor([8, 128, 1, 1])]; + tensor inputs_431_cast_fp16 = reshape(shape = var_15767, x = current_key_205_cast_fp16)[name = string("inputs_431_cast_fp16")]; + tensor inputs_sq_431_cast_fp16 = mul(x = inputs_431_cast_fp16, y = inputs_431_cast_fp16)[name = string("inputs_sq_431_cast_fp16")]; + tensor variance_431_axes_0 = const()[name = string("variance_431_axes_0"), val = tensor([1])]; + bool variance_431_keep_dims_0 = const()[name = string("variance_431_keep_dims_0"), val = bool(true)]; + tensor variance_431_cast_fp16 = reduce_mean(axes = variance_431_axes_0, keep_dims = variance_431_keep_dims_0, x = inputs_sq_431_cast_fp16)[name = string("variance_431_cast_fp16")]; + fp16 var_15773_to_fp16 = const()[name = string("op_15773_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15774_cast_fp16 = add(x = variance_431_cast_fp16, y = var_15773_to_fp16)[name = string("op_15774_cast_fp16")]; + fp32 var_15775_epsilon_0 = const()[name = string("op_15775_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15775_cast_fp16 = rsqrt(epsilon = var_15775_epsilon_0, x = var_15774_cast_fp16)[name = string("op_15775_cast_fp16")]; + tensor hidden_states_533_cast_fp16 = mul(x = inputs_431_cast_fp16, y = var_15775_cast_fp16)[name = string("hidden_states_533_cast_fp16")]; + tensor current_key_normed_103_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_533_cast_fp16)[name = string("current_key_normed_103_cast_fp16")]; + tensor var_15793 = const()[name = string("op_15793"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_409_cast_fp16 = reshape(shape = var_15793, x = query_normed_103_cast_fp16)[name = string("mh_q_409_cast_fp16")]; + tensor var_15795 = const()[name = string("op_15795"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_409_cast_fp16 = reshape(shape = var_15795, x = current_key_normed_103_cast_fp16)[name = string("mh_k_409_cast_fp16")]; + tensor var_15799_cast_fp16 = mul(x = mh_q_409_cast_fp16, y = cos_101_to_fp16)[name = string("op_15799_cast_fp16")]; + tensor var_15804_begin_0 = const()[name = string("op_15804_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15804_end_0 = const()[name = string("op_15804_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_15804_end_mask_0 = const()[name = string("op_15804_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_15804_cast_fp16 = slice_by_index(begin = var_15804_begin_0, end = var_15804_end_0, end_mask = var_15804_end_mask_0, x = mh_q_409_cast_fp16)[name = string("op_15804_cast_fp16")]; + tensor var_15810_begin_0 = const()[name = string("op_15810_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_15810_end_0 = const()[name = string("op_15810_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_15810_end_mask_0 = const()[name = string("op_15810_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_15810_cast_fp16 = slice_by_index(begin = var_15810_begin_0, end = var_15810_end_0, end_mask = var_15810_end_mask_0, x = mh_q_409_cast_fp16)[name = string("op_15810_cast_fp16")]; + fp16 const_1044_promoted_to_fp16 = const()[name = string("const_1044_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15812_cast_fp16 = mul(x = var_15810_cast_fp16, y = const_1044_promoted_to_fp16)[name = string("op_15812_cast_fp16")]; + bool var_15814_interleave_0 = const()[name = string("op_15814_interleave_0"), val = bool(false)]; + tensor var_15814_cast_fp16 = concat(axis = var_15698, interleave = var_15814_interleave_0, values = (var_15812_cast_fp16, var_15804_cast_fp16))[name = string("op_15814_cast_fp16")]; + tensor var_15815_cast_fp16 = mul(x = var_15814_cast_fp16, y = sin_101_to_fp16)[name = string("op_15815_cast_fp16")]; + tensor mh_q_411_cast_fp16 = add(x = var_15799_cast_fp16, y = var_15815_cast_fp16)[name = string("mh_q_411_cast_fp16")]; + tensor var_15817_cast_fp16 = mul(x = mh_k_409_cast_fp16, y = cos_101_to_fp16)[name = string("op_15817_cast_fp16")]; + tensor var_15822_begin_0 = const()[name = string("op_15822_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15822_end_0 = const()[name = string("op_15822_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_15822_end_mask_0 = const()[name = string("op_15822_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_15822_cast_fp16 = slice_by_index(begin = var_15822_begin_0, end = var_15822_end_0, end_mask = var_15822_end_mask_0, x = mh_k_409_cast_fp16)[name = string("op_15822_cast_fp16")]; + tensor var_15828_begin_0 = const()[name = string("op_15828_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_15828_end_0 = const()[name = string("op_15828_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_15828_end_mask_0 = const()[name = string("op_15828_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_15828_cast_fp16 = slice_by_index(begin = var_15828_begin_0, end = var_15828_end_0, end_mask = var_15828_end_mask_0, x = mh_k_409_cast_fp16)[name = string("op_15828_cast_fp16")]; + fp16 const_1047_promoted_to_fp16 = const()[name = string("const_1047_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_15830_cast_fp16 = mul(x = var_15828_cast_fp16, y = const_1047_promoted_to_fp16)[name = string("op_15830_cast_fp16")]; + bool var_15832_interleave_0 = const()[name = string("op_15832_interleave_0"), val = bool(false)]; + tensor var_15832_cast_fp16 = concat(axis = var_15698, interleave = var_15832_interleave_0, values = (var_15830_cast_fp16, var_15822_cast_fp16))[name = string("op_15832_cast_fp16")]; + tensor var_15833_cast_fp16 = mul(x = var_15832_cast_fp16, y = sin_101_to_fp16)[name = string("op_15833_cast_fp16")]; + tensor mh_k_411_cast_fp16 = add(x = var_15817_cast_fp16, y = var_15833_cast_fp16)[name = string("mh_k_411_cast_fp16")]; + tensor var_15837 = const()[name = string("op_15837"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_207_cast_fp16 = reshape(shape = var_15837, x = mh_k_411_cast_fp16)[name = string("current_key_207_cast_fp16")]; + tensor var_15843_to_fp16 = const()[name = string("op_15843_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199424)))]; + tensor var_15844_cast_fp16 = mul(x = obj_455_cast_fp16, y = var_15843_to_fp16)[name = string("op_15844_cast_fp16")]; + tensor var_15841_to_fp16 = const()[name = string("op_15841_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199552)))]; + tensor var_15845_cast_fp16 = mul(x = current_key_207_cast_fp16, y = var_15841_to_fp16)[name = string("op_15845_cast_fp16")]; + tensor key_207_cast_fp16 = add(x = var_15844_cast_fp16, y = var_15845_cast_fp16)[name = string("key_207_cast_fp16")]; + tensor var_15847_to_fp16 = const()[name = string("op_15847_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199424)))]; + tensor var_15848_cast_fp16 = mul(x = obj_457_cast_fp16, y = var_15847_to_fp16)[name = string("op_15848_cast_fp16")]; + tensor var_15849_cast_fp16 = mul(x = current_value_103_cast_fp16, y = var_15841_to_fp16)[name = string("op_15849_cast_fp16")]; + tensor value_103_cast_fp16 = add(x = var_15848_cast_fp16, y = var_15849_cast_fp16)[name = string("value_103_cast_fp16")]; + fp16 var_15856_to_fp16 = const()[name = string("op_15856_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_415_cast_fp16 = mul(x = mh_q_411_cast_fp16, y = var_15856_to_fp16)[name = string("mh_q_415_cast_fp16")]; + tensor var_15858 = const()[name = string("op_15858"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_413_cast_fp16 = reshape(shape = var_15858, x = key_207_cast_fp16)[name = string("mh_k_413_cast_fp16")]; + tensor var_15860 = const()[name = string("op_15860"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_205_cast_fp16 = reshape(shape = var_15860, x = value_103_cast_fp16)[name = string("mh_v_205_cast_fp16")]; + tensor transpose_204_perm_0 = const()[name = string("transpose_204_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_102_reps_0 = const()[name = string("tile_102_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_204_cast_fp16 = transpose(perm = transpose_204_perm_0, x = mh_k_413_cast_fp16)[name = string("transpose_173")]; + tensor tile_102_cast_fp16 = tile(reps = tile_102_reps_0, x = transpose_204_cast_fp16)[name = string("tile_102_cast_fp16")]; + tensor concat_257 = const()[name = string("concat_257"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_204_cast_fp16 = reshape(shape = concat_257, x = tile_102_cast_fp16)[name = string("reshape_204_cast_fp16")]; + tensor transpose_205_perm_0 = const()[name = string("transpose_205_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_258 = const()[name = string("concat_258"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_205_cast_fp16 = transpose(perm = transpose_205_perm_0, x = reshape_204_cast_fp16)[name = string("transpose_172")]; + tensor reshape_205_cast_fp16 = reshape(shape = concat_258, x = transpose_205_cast_fp16)[name = string("reshape_205_cast_fp16")]; + tensor transpose_206_perm_0 = const()[name = string("transpose_206_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_103_reps_0 = const()[name = string("tile_103_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_206_cast_fp16 = transpose(perm = transpose_206_perm_0, x = mh_v_205_cast_fp16)[name = string("transpose_171")]; + tensor tile_103_cast_fp16 = tile(reps = tile_103_reps_0, x = transpose_206_cast_fp16)[name = string("tile_103_cast_fp16")]; + tensor concat_259 = const()[name = string("concat_259"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_206_cast_fp16 = reshape(shape = concat_259, x = tile_103_cast_fp16)[name = string("reshape_206_cast_fp16")]; + tensor transpose_207_perm_0 = const()[name = string("transpose_207_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_260 = const()[name = string("concat_260"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_207_cast_fp16 = transpose(perm = transpose_207_perm_0, x = reshape_206_cast_fp16)[name = string("transpose_170")]; + tensor reshape_207_cast_fp16 = reshape(shape = concat_260, x = transpose_207_cast_fp16)[name = string("reshape_207_cast_fp16")]; + tensor transpose_521_perm_0 = const()[name = string("transpose_521_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_307_transpose_x_1 = const()[name = string("mh_w_307_transpose_x_1"), val = bool(true)]; + bool mh_w_307_transpose_y_1 = const()[name = string("mh_w_307_transpose_y_1"), val = bool(false)]; + tensor transpose_521_cast_fp16 = transpose(perm = transpose_521_perm_0, x = reshape_205_cast_fp16)[name = string("transpose_169")]; + tensor mh_w_307_cast_fp16 = matmul(transpose_x = mh_w_307_transpose_x_1, transpose_y = mh_w_307_transpose_y_1, x = mh_q_415_cast_fp16, y = transpose_521_cast_fp16)[name = string("mh_w_307_cast_fp16")]; + tensor var_15868_to_fp16 = const()[name = string("op_15868_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199680)))]; + tensor mh_w_309_cast_fp16 = add(x = mh_w_307_cast_fp16, y = var_15868_to_fp16)[name = string("mh_w_309_cast_fp16")]; + tensor mh_w_311_cast_fp16 = softmax(axis = var_15688, x = mh_w_309_cast_fp16)[name = string("mh_w_311_cast_fp16")]; + tensor transpose_522_perm_0 = const()[name = string("transpose_522_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_103_transpose_x_1 = const()[name = string("attn_103_transpose_x_1"), val = bool(false)]; + bool attn_103_transpose_y_1 = const()[name = string("attn_103_transpose_y_1"), val = bool(true)]; + tensor transpose_522_cast_fp16 = transpose(perm = transpose_522_perm_0, x = reshape_207_cast_fp16)[name = string("transpose_168")]; + tensor attn_103_cast_fp16 = matmul(transpose_x = attn_103_transpose_x_1, transpose_y = attn_103_transpose_y_1, x = transpose_522_cast_fp16, y = mh_w_311_cast_fp16)[name = string("attn_103_cast_fp16")]; + tensor var_15874 = const()[name = string("op_15874"), val = tensor([1, 2048, 1, 1])]; + tensor input_445_cast_fp16 = reshape(shape = var_15874, x = attn_103_cast_fp16)[name = string("input_445_cast_fp16")]; + string obj_459_pad_type_0 = const()[name = string("obj_459_pad_type_0"), val = string("valid")]; + tensor obj_459_strides_0 = const()[name = string("obj_459_strides_0"), val = tensor([1, 1])]; + tensor obj_459_pad_0 = const()[name = string("obj_459_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_459_dilations_0 = const()[name = string("obj_459_dilations_0"), val = tensor([1, 1])]; + int32 obj_459_groups_0 = const()[name = string("obj_459_groups_0"), val = int32(1)]; + tensor obj_459_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_459_dilations_0, groups = obj_459_groups_0, pad = obj_459_pad_0, pad_type = obj_459_pad_type_0, strides = obj_459_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_445_cast_fp16)[name = string("obj_459_cast_fp16")]; + tensor inputs_433_cast_fp16 = add(x = inputs_427_cast_fp16, y = obj_459_cast_fp16)[name = string("inputs_433_cast_fp16")]; + tensor inputs_sq_433_cast_fp16 = mul(x = inputs_433_cast_fp16, y = inputs_433_cast_fp16)[name = string("inputs_sq_433_cast_fp16")]; + tensor variance_433_axes_0 = const()[name = string("variance_433_axes_0"), val = tensor([1])]; + bool variance_433_keep_dims_0 = const()[name = string("variance_433_keep_dims_0"), val = bool(true)]; + tensor variance_433_cast_fp16 = reduce_mean(axes = variance_433_axes_0, keep_dims = variance_433_keep_dims_0, x = inputs_sq_433_cast_fp16)[name = string("variance_433_cast_fp16")]; + fp16 var_15892_to_fp16 = const()[name = string("op_15892_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15893_cast_fp16 = add(x = variance_433_cast_fp16, y = var_15892_to_fp16)[name = string("op_15893_cast_fp16")]; + fp32 var_15894_epsilon_0 = const()[name = string("op_15894_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15894_cast_fp16 = rsqrt(epsilon = var_15894_epsilon_0, x = var_15893_cast_fp16)[name = string("op_15894_cast_fp16")]; + tensor hidden_states_535_cast_fp16 = mul(x = inputs_433_cast_fp16, y = var_15894_cast_fp16)[name = string("hidden_states_535_cast_fp16")]; + tensor input_447_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_535_cast_fp16)[name = string("input_447_cast_fp16")]; + string input_449_pad_type_0 = const()[name = string("input_449_pad_type_0"), val = string("valid")]; + tensor input_449_strides_0 = const()[name = string("input_449_strides_0"), val = tensor([1, 1])]; + tensor input_449_pad_0 = const()[name = string("input_449_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_449_dilations_0 = const()[name = string("input_449_dilations_0"), val = tensor([1, 1])]; + int32 input_449_groups_0 = const()[name = string("input_449_groups_0"), val = int32(1)]; + tensor input_449_cast_fp16 = conv(dilations = input_449_dilations_0, groups = input_449_groups_0, pad = input_449_pad_0, pad_type = input_449_pad_type_0, strides = input_449_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_447_cast_fp16)[name = string("input_449_cast_fp16")]; + tensor var_15908_cast_fp16 = silu(x = input_449_cast_fp16)[name = string("op_15908_cast_fp16")]; + string var_15914_pad_type_0 = const()[name = string("op_15914_pad_type_0"), val = string("valid")]; + tensor var_15914_strides_0 = const()[name = string("op_15914_strides_0"), val = tensor([1, 1])]; + tensor var_15914_pad_0 = const()[name = string("op_15914_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_15914_dilations_0 = const()[name = string("op_15914_dilations_0"), val = tensor([1, 1])]; + int32 var_15914_groups_0 = const()[name = string("op_15914_groups_0"), val = int32(1)]; + tensor var_15914_cast_fp16 = conv(dilations = var_15914_dilations_0, groups = var_15914_groups_0, pad = var_15914_pad_0, pad_type = var_15914_pad_type_0, strides = var_15914_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_447_cast_fp16)[name = string("op_15914_cast_fp16")]; + tensor input_451_cast_fp16 = mul(x = var_15908_cast_fp16, y = var_15914_cast_fp16)[name = string("input_451_cast_fp16")]; + string hidden_states_537_pad_type_0 = const()[name = string("hidden_states_537_pad_type_0"), val = string("valid")]; + tensor hidden_states_537_strides_0 = const()[name = string("hidden_states_537_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_537_pad_0 = const()[name = string("hidden_states_537_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_537_dilations_0 = const()[name = string("hidden_states_537_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_537_groups_0 = const()[name = string("hidden_states_537_groups_0"), val = int32(1)]; + tensor hidden_states_537_cast_fp16 = conv(dilations = hidden_states_537_dilations_0, groups = hidden_states_537_groups_0, pad = hidden_states_537_pad_0, pad_type = hidden_states_537_pad_type_0, strides = hidden_states_537_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_451_cast_fp16)[name = string("hidden_states_537_cast_fp16")]; + tensor inputs_435_cast_fp16 = add(x = inputs_433_cast_fp16, y = hidden_states_537_cast_fp16)[name = string("inputs_435_cast_fp16")]; + tensor obj_463_begin_0 = const()[name = string("obj_463_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_463_end_0 = const()[name = string("obj_463_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_463_end_mask_0 = const()[name = string("obj_463_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_463_cast_fp16 = slice_by_index(begin = obj_463_begin_0, end = obj_463_end_0, end_mask = obj_463_end_mask_0, x = key_caches_21_cast_fp16)[name = string("obj_463_cast_fp16")]; + tensor obj_465_begin_0 = const()[name = string("obj_465_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_465_end_0 = const()[name = string("obj_465_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_465_end_mask_0 = const()[name = string("obj_465_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_465_cast_fp16 = slice_by_index(begin = obj_465_begin_0, end = obj_465_end_0, end_mask = obj_465_end_mask_0, x = value_caches_21_cast_fp16)[name = string("obj_465_cast_fp16")]; + int32 var_15962 = const()[name = string("op_15962"), val = int32(3)]; + int32 var_15972 = const()[name = string("op_15972"), val = int32(-2)]; + tensor inputs_sq_435_cast_fp16 = mul(x = inputs_435_cast_fp16, y = inputs_435_cast_fp16)[name = string("inputs_sq_435_cast_fp16")]; + tensor variance_435_axes_0 = const()[name = string("variance_435_axes_0"), val = tensor([1])]; + bool variance_435_keep_dims_0 = const()[name = string("variance_435_keep_dims_0"), val = bool(true)]; + tensor variance_435_cast_fp16 = reduce_mean(axes = variance_435_axes_0, keep_dims = variance_435_keep_dims_0, x = inputs_sq_435_cast_fp16)[name = string("variance_435_cast_fp16")]; + fp16 var_15986_to_fp16 = const()[name = string("op_15986_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_15987_cast_fp16 = add(x = variance_435_cast_fp16, y = var_15986_to_fp16)[name = string("op_15987_cast_fp16")]; + fp32 var_15988_epsilon_0 = const()[name = string("op_15988_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_15988_cast_fp16 = rsqrt(epsilon = var_15988_epsilon_0, x = var_15987_cast_fp16)[name = string("op_15988_cast_fp16")]; + tensor hidden_states_539_cast_fp16 = mul(x = inputs_435_cast_fp16, y = var_15988_cast_fp16)[name = string("hidden_states_539_cast_fp16")]; + tensor obj_461_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_539_cast_fp16)[name = string("obj_461_cast_fp16")]; + string query_313_pad_type_0 = const()[name = string("query_313_pad_type_0"), val = string("valid")]; + tensor query_313_strides_0 = const()[name = string("query_313_strides_0"), val = tensor([1, 1])]; + tensor query_313_pad_0 = const()[name = string("query_313_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_313_dilations_0 = const()[name = string("query_313_dilations_0"), val = tensor([1, 1])]; + int32 query_313_groups_0 = const()[name = string("query_313_groups_0"), val = int32(1)]; + tensor query_313_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_313_dilations_0, groups = query_313_groups_0, pad = query_313_pad_0, pad_type = query_313_pad_type_0, strides = query_313_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_461_cast_fp16)[name = string("query_313_cast_fp16")]; + string current_key_209_pad_type_0 = const()[name = string("current_key_209_pad_type_0"), val = string("valid")]; + tensor current_key_209_strides_0 = const()[name = string("current_key_209_strides_0"), val = tensor([1, 1])]; + tensor current_key_209_pad_0 = const()[name = string("current_key_209_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_209_dilations_0 = const()[name = string("current_key_209_dilations_0"), val = tensor([1, 1])]; + int32 current_key_209_groups_0 = const()[name = string("current_key_209_groups_0"), val = int32(1)]; + tensor current_key_209_cast_fp16 = conv(dilations = current_key_209_dilations_0, groups = current_key_209_groups_0, pad = current_key_209_pad_0, pad_type = current_key_209_pad_type_0, strides = current_key_209_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_461_cast_fp16)[name = string("current_key_209_cast_fp16")]; + string current_value_105_pad_type_0 = const()[name = string("current_value_105_pad_type_0"), val = string("valid")]; + tensor current_value_105_strides_0 = const()[name = string("current_value_105_strides_0"), val = tensor([1, 1])]; + tensor current_value_105_pad_0 = const()[name = string("current_value_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_105_dilations_0 = const()[name = string("current_value_105_dilations_0"), val = tensor([1, 1])]; + int32 current_value_105_groups_0 = const()[name = string("current_value_105_groups_0"), val = int32(1)]; + tensor current_value_105_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_105_dilations_0, groups = current_value_105_groups_0, pad = current_value_105_pad_0, pad_type = current_value_105_pad_type_0, strides = current_value_105_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_461_cast_fp16)[name = string("current_value_105_cast_fp16")]; + tensor var_16025 = const()[name = string("op_16025"), val = tensor([16, 128, 1, 1])]; + tensor inputs_437_cast_fp16 = reshape(shape = var_16025, x = query_313_cast_fp16)[name = string("inputs_437_cast_fp16")]; + tensor inputs_sq_437_cast_fp16 = mul(x = inputs_437_cast_fp16, y = inputs_437_cast_fp16)[name = string("inputs_sq_437_cast_fp16")]; + tensor variance_437_axes_0 = const()[name = string("variance_437_axes_0"), val = tensor([1])]; + bool variance_437_keep_dims_0 = const()[name = string("variance_437_keep_dims_0"), val = bool(true)]; + tensor variance_437_cast_fp16 = reduce_mean(axes = variance_437_axes_0, keep_dims = variance_437_keep_dims_0, x = inputs_sq_437_cast_fp16)[name = string("variance_437_cast_fp16")]; + fp16 var_16031_to_fp16 = const()[name = string("op_16031_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16032_cast_fp16 = add(x = variance_437_cast_fp16, y = var_16031_to_fp16)[name = string("op_16032_cast_fp16")]; + fp32 var_16033_epsilon_0 = const()[name = string("op_16033_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16033_cast_fp16 = rsqrt(epsilon = var_16033_epsilon_0, x = var_16032_cast_fp16)[name = string("op_16033_cast_fp16")]; + tensor hidden_states_541_cast_fp16 = mul(x = inputs_437_cast_fp16, y = var_16033_cast_fp16)[name = string("hidden_states_541_cast_fp16")]; + tensor query_normed_105_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_541_cast_fp16)[name = string("query_normed_105_cast_fp16")]; + tensor var_16041 = const()[name = string("op_16041"), val = tensor([8, 128, 1, 1])]; + tensor inputs_439_cast_fp16 = reshape(shape = var_16041, x = current_key_209_cast_fp16)[name = string("inputs_439_cast_fp16")]; + tensor inputs_sq_439_cast_fp16 = mul(x = inputs_439_cast_fp16, y = inputs_439_cast_fp16)[name = string("inputs_sq_439_cast_fp16")]; + tensor variance_439_axes_0 = const()[name = string("variance_439_axes_0"), val = tensor([1])]; + bool variance_439_keep_dims_0 = const()[name = string("variance_439_keep_dims_0"), val = bool(true)]; + tensor variance_439_cast_fp16 = reduce_mean(axes = variance_439_axes_0, keep_dims = variance_439_keep_dims_0, x = inputs_sq_439_cast_fp16)[name = string("variance_439_cast_fp16")]; + fp16 var_16047_to_fp16 = const()[name = string("op_16047_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16048_cast_fp16 = add(x = variance_439_cast_fp16, y = var_16047_to_fp16)[name = string("op_16048_cast_fp16")]; + fp32 var_16049_epsilon_0 = const()[name = string("op_16049_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16049_cast_fp16 = rsqrt(epsilon = var_16049_epsilon_0, x = var_16048_cast_fp16)[name = string("op_16049_cast_fp16")]; + tensor hidden_states_543_cast_fp16 = mul(x = inputs_439_cast_fp16, y = var_16049_cast_fp16)[name = string("hidden_states_543_cast_fp16")]; + tensor current_key_normed_105_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_543_cast_fp16)[name = string("current_key_normed_105_cast_fp16")]; + tensor var_16067 = const()[name = string("op_16067"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_417_cast_fp16 = reshape(shape = var_16067, x = query_normed_105_cast_fp16)[name = string("mh_q_417_cast_fp16")]; + tensor var_16069 = const()[name = string("op_16069"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_417_cast_fp16 = reshape(shape = var_16069, x = current_key_normed_105_cast_fp16)[name = string("mh_k_417_cast_fp16")]; + tensor var_16073_cast_fp16 = mul(x = mh_q_417_cast_fp16, y = cos_101_to_fp16)[name = string("op_16073_cast_fp16")]; + tensor var_16078_begin_0 = const()[name = string("op_16078_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16078_end_0 = const()[name = string("op_16078_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_16078_end_mask_0 = const()[name = string("op_16078_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_16078_cast_fp16 = slice_by_index(begin = var_16078_begin_0, end = var_16078_end_0, end_mask = var_16078_end_mask_0, x = mh_q_417_cast_fp16)[name = string("op_16078_cast_fp16")]; + tensor var_16084_begin_0 = const()[name = string("op_16084_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_16084_end_0 = const()[name = string("op_16084_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_16084_end_mask_0 = const()[name = string("op_16084_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_16084_cast_fp16 = slice_by_index(begin = var_16084_begin_0, end = var_16084_end_0, end_mask = var_16084_end_mask_0, x = mh_q_417_cast_fp16)[name = string("op_16084_cast_fp16")]; + fp16 const_1064_promoted_to_fp16 = const()[name = string("const_1064_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16086_cast_fp16 = mul(x = var_16084_cast_fp16, y = const_1064_promoted_to_fp16)[name = string("op_16086_cast_fp16")]; + bool var_16088_interleave_0 = const()[name = string("op_16088_interleave_0"), val = bool(false)]; + tensor var_16088_cast_fp16 = concat(axis = var_15972, interleave = var_16088_interleave_0, values = (var_16086_cast_fp16, var_16078_cast_fp16))[name = string("op_16088_cast_fp16")]; + tensor var_16089_cast_fp16 = mul(x = var_16088_cast_fp16, y = sin_101_to_fp16)[name = string("op_16089_cast_fp16")]; + tensor mh_q_419_cast_fp16 = add(x = var_16073_cast_fp16, y = var_16089_cast_fp16)[name = string("mh_q_419_cast_fp16")]; + tensor var_16091_cast_fp16 = mul(x = mh_k_417_cast_fp16, y = cos_101_to_fp16)[name = string("op_16091_cast_fp16")]; + tensor var_16096_begin_0 = const()[name = string("op_16096_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16096_end_0 = const()[name = string("op_16096_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_16096_end_mask_0 = const()[name = string("op_16096_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_16096_cast_fp16 = slice_by_index(begin = var_16096_begin_0, end = var_16096_end_0, end_mask = var_16096_end_mask_0, x = mh_k_417_cast_fp16)[name = string("op_16096_cast_fp16")]; + tensor var_16102_begin_0 = const()[name = string("op_16102_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_16102_end_0 = const()[name = string("op_16102_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_16102_end_mask_0 = const()[name = string("op_16102_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_16102_cast_fp16 = slice_by_index(begin = var_16102_begin_0, end = var_16102_end_0, end_mask = var_16102_end_mask_0, x = mh_k_417_cast_fp16)[name = string("op_16102_cast_fp16")]; + fp16 const_1067_promoted_to_fp16 = const()[name = string("const_1067_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16104_cast_fp16 = mul(x = var_16102_cast_fp16, y = const_1067_promoted_to_fp16)[name = string("op_16104_cast_fp16")]; + bool var_16106_interleave_0 = const()[name = string("op_16106_interleave_0"), val = bool(false)]; + tensor var_16106_cast_fp16 = concat(axis = var_15972, interleave = var_16106_interleave_0, values = (var_16104_cast_fp16, var_16096_cast_fp16))[name = string("op_16106_cast_fp16")]; + tensor var_16107_cast_fp16 = mul(x = var_16106_cast_fp16, y = sin_101_to_fp16)[name = string("op_16107_cast_fp16")]; + tensor mh_k_419_cast_fp16 = add(x = var_16091_cast_fp16, y = var_16107_cast_fp16)[name = string("mh_k_419_cast_fp16")]; + tensor var_16111 = const()[name = string("op_16111"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_211_cast_fp16 = reshape(shape = var_16111, x = mh_k_419_cast_fp16)[name = string("current_key_211_cast_fp16")]; + tensor var_16117_to_fp16 = const()[name = string("op_16117_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199424)))]; + tensor var_16118_cast_fp16 = mul(x = obj_463_cast_fp16, y = var_16117_to_fp16)[name = string("op_16118_cast_fp16")]; + tensor var_16115_to_fp16 = const()[name = string("op_16115_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199552)))]; + tensor var_16119_cast_fp16 = mul(x = current_key_211_cast_fp16, y = var_16115_to_fp16)[name = string("op_16119_cast_fp16")]; + tensor key_211_cast_fp16 = add(x = var_16118_cast_fp16, y = var_16119_cast_fp16)[name = string("key_211_cast_fp16")]; + tensor var_16121_to_fp16 = const()[name = string("op_16121_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199424)))]; + tensor var_16122_cast_fp16 = mul(x = obj_465_cast_fp16, y = var_16121_to_fp16)[name = string("op_16122_cast_fp16")]; + tensor var_16123_cast_fp16 = mul(x = current_value_105_cast_fp16, y = var_16115_to_fp16)[name = string("op_16123_cast_fp16")]; + tensor value_105_cast_fp16 = add(x = var_16122_cast_fp16, y = var_16123_cast_fp16)[name = string("value_105_cast_fp16")]; + fp16 var_16130_to_fp16 = const()[name = string("op_16130_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_423_cast_fp16 = mul(x = mh_q_419_cast_fp16, y = var_16130_to_fp16)[name = string("mh_q_423_cast_fp16")]; + tensor var_16132 = const()[name = string("op_16132"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_421_cast_fp16 = reshape(shape = var_16132, x = key_211_cast_fp16)[name = string("mh_k_421_cast_fp16")]; + tensor var_16134 = const()[name = string("op_16134"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_209_cast_fp16 = reshape(shape = var_16134, x = value_105_cast_fp16)[name = string("mh_v_209_cast_fp16")]; + tensor transpose_208_perm_0 = const()[name = string("transpose_208_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_104_reps_0 = const()[name = string("tile_104_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_208_cast_fp16 = transpose(perm = transpose_208_perm_0, x = mh_k_421_cast_fp16)[name = string("transpose_167")]; + tensor tile_104_cast_fp16 = tile(reps = tile_104_reps_0, x = transpose_208_cast_fp16)[name = string("tile_104_cast_fp16")]; + tensor concat_261 = const()[name = string("concat_261"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_208_cast_fp16 = reshape(shape = concat_261, x = tile_104_cast_fp16)[name = string("reshape_208_cast_fp16")]; + tensor transpose_209_perm_0 = const()[name = string("transpose_209_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_262 = const()[name = string("concat_262"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_209_cast_fp16 = transpose(perm = transpose_209_perm_0, x = reshape_208_cast_fp16)[name = string("transpose_166")]; + tensor reshape_209_cast_fp16 = reshape(shape = concat_262, x = transpose_209_cast_fp16)[name = string("reshape_209_cast_fp16")]; + tensor transpose_210_perm_0 = const()[name = string("transpose_210_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_105_reps_0 = const()[name = string("tile_105_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_210_cast_fp16 = transpose(perm = transpose_210_perm_0, x = mh_v_209_cast_fp16)[name = string("transpose_165")]; + tensor tile_105_cast_fp16 = tile(reps = tile_105_reps_0, x = transpose_210_cast_fp16)[name = string("tile_105_cast_fp16")]; + tensor concat_263 = const()[name = string("concat_263"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_210_cast_fp16 = reshape(shape = concat_263, x = tile_105_cast_fp16)[name = string("reshape_210_cast_fp16")]; + tensor transpose_211_perm_0 = const()[name = string("transpose_211_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_264 = const()[name = string("concat_264"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_211_cast_fp16 = transpose(perm = transpose_211_perm_0, x = reshape_210_cast_fp16)[name = string("transpose_164")]; + tensor reshape_211_cast_fp16 = reshape(shape = concat_264, x = transpose_211_cast_fp16)[name = string("reshape_211_cast_fp16")]; + tensor transpose_525_perm_0 = const()[name = string("transpose_525_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_313_transpose_x_1 = const()[name = string("mh_w_313_transpose_x_1"), val = bool(true)]; + bool mh_w_313_transpose_y_1 = const()[name = string("mh_w_313_transpose_y_1"), val = bool(false)]; + tensor transpose_525_cast_fp16 = transpose(perm = transpose_525_perm_0, x = reshape_209_cast_fp16)[name = string("transpose_163")]; + tensor mh_w_313_cast_fp16 = matmul(transpose_x = mh_w_313_transpose_x_1, transpose_y = mh_w_313_transpose_y_1, x = mh_q_423_cast_fp16, y = transpose_525_cast_fp16)[name = string("mh_w_313_cast_fp16")]; + tensor var_16142_to_fp16 = const()[name = string("op_16142_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199680)))]; + tensor mh_w_315_cast_fp16 = add(x = mh_w_313_cast_fp16, y = var_16142_to_fp16)[name = string("mh_w_315_cast_fp16")]; + tensor mh_w_317_cast_fp16 = softmax(axis = var_15962, x = mh_w_315_cast_fp16)[name = string("mh_w_317_cast_fp16")]; + tensor transpose_526_perm_0 = const()[name = string("transpose_526_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_105_transpose_x_1 = const()[name = string("attn_105_transpose_x_1"), val = bool(false)]; + bool attn_105_transpose_y_1 = const()[name = string("attn_105_transpose_y_1"), val = bool(true)]; + tensor transpose_526_cast_fp16 = transpose(perm = transpose_526_perm_0, x = reshape_211_cast_fp16)[name = string("transpose_162")]; + tensor attn_105_cast_fp16 = matmul(transpose_x = attn_105_transpose_x_1, transpose_y = attn_105_transpose_y_1, x = transpose_526_cast_fp16, y = mh_w_317_cast_fp16)[name = string("attn_105_cast_fp16")]; + tensor var_16148 = const()[name = string("op_16148"), val = tensor([1, 2048, 1, 1])]; + tensor input_453_cast_fp16 = reshape(shape = var_16148, x = attn_105_cast_fp16)[name = string("input_453_cast_fp16")]; + string obj_467_pad_type_0 = const()[name = string("obj_467_pad_type_0"), val = string("valid")]; + tensor obj_467_strides_0 = const()[name = string("obj_467_strides_0"), val = tensor([1, 1])]; + tensor obj_467_pad_0 = const()[name = string("obj_467_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_467_dilations_0 = const()[name = string("obj_467_dilations_0"), val = tensor([1, 1])]; + int32 obj_467_groups_0 = const()[name = string("obj_467_groups_0"), val = int32(1)]; + tensor obj_467_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_467_dilations_0, groups = obj_467_groups_0, pad = obj_467_pad_0, pad_type = obj_467_pad_type_0, strides = obj_467_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_453_cast_fp16)[name = string("obj_467_cast_fp16")]; + tensor inputs_441_cast_fp16 = add(x = inputs_435_cast_fp16, y = obj_467_cast_fp16)[name = string("inputs_441_cast_fp16")]; + tensor inputs_sq_441_cast_fp16 = mul(x = inputs_441_cast_fp16, y = inputs_441_cast_fp16)[name = string("inputs_sq_441_cast_fp16")]; + tensor variance_441_axes_0 = const()[name = string("variance_441_axes_0"), val = tensor([1])]; + bool variance_441_keep_dims_0 = const()[name = string("variance_441_keep_dims_0"), val = bool(true)]; + tensor variance_441_cast_fp16 = reduce_mean(axes = variance_441_axes_0, keep_dims = variance_441_keep_dims_0, x = inputs_sq_441_cast_fp16)[name = string("variance_441_cast_fp16")]; + fp16 var_16166_to_fp16 = const()[name = string("op_16166_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16167_cast_fp16 = add(x = variance_441_cast_fp16, y = var_16166_to_fp16)[name = string("op_16167_cast_fp16")]; + fp32 var_16168_epsilon_0 = const()[name = string("op_16168_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16168_cast_fp16 = rsqrt(epsilon = var_16168_epsilon_0, x = var_16167_cast_fp16)[name = string("op_16168_cast_fp16")]; + tensor hidden_states_545_cast_fp16 = mul(x = inputs_441_cast_fp16, y = var_16168_cast_fp16)[name = string("hidden_states_545_cast_fp16")]; + tensor input_455_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_545_cast_fp16)[name = string("input_455_cast_fp16")]; + string input_457_pad_type_0 = const()[name = string("input_457_pad_type_0"), val = string("valid")]; + tensor input_457_strides_0 = const()[name = string("input_457_strides_0"), val = tensor([1, 1])]; + tensor input_457_pad_0 = const()[name = string("input_457_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_457_dilations_0 = const()[name = string("input_457_dilations_0"), val = tensor([1, 1])]; + int32 input_457_groups_0 = const()[name = string("input_457_groups_0"), val = int32(1)]; + tensor input_457_cast_fp16 = conv(dilations = input_457_dilations_0, groups = input_457_groups_0, pad = input_457_pad_0, pad_type = input_457_pad_type_0, strides = input_457_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_455_cast_fp16)[name = string("input_457_cast_fp16")]; + tensor var_16182_cast_fp16 = silu(x = input_457_cast_fp16)[name = string("op_16182_cast_fp16")]; + string var_16188_pad_type_0 = const()[name = string("op_16188_pad_type_0"), val = string("valid")]; + tensor var_16188_strides_0 = const()[name = string("op_16188_strides_0"), val = tensor([1, 1])]; + tensor var_16188_pad_0 = const()[name = string("op_16188_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16188_dilations_0 = const()[name = string("op_16188_dilations_0"), val = tensor([1, 1])]; + int32 var_16188_groups_0 = const()[name = string("op_16188_groups_0"), val = int32(1)]; + tensor var_16188_cast_fp16 = conv(dilations = var_16188_dilations_0, groups = var_16188_groups_0, pad = var_16188_pad_0, pad_type = var_16188_pad_type_0, strides = var_16188_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_455_cast_fp16)[name = string("op_16188_cast_fp16")]; + tensor input_459_cast_fp16 = mul(x = var_16182_cast_fp16, y = var_16188_cast_fp16)[name = string("input_459_cast_fp16")]; + string hidden_states_547_pad_type_0 = const()[name = string("hidden_states_547_pad_type_0"), val = string("valid")]; + tensor hidden_states_547_strides_0 = const()[name = string("hidden_states_547_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_547_pad_0 = const()[name = string("hidden_states_547_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_547_dilations_0 = const()[name = string("hidden_states_547_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_547_groups_0 = const()[name = string("hidden_states_547_groups_0"), val = int32(1)]; + tensor hidden_states_547_cast_fp16 = conv(dilations = hidden_states_547_dilations_0, groups = hidden_states_547_groups_0, pad = hidden_states_547_pad_0, pad_type = hidden_states_547_pad_type_0, strides = hidden_states_547_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_459_cast_fp16)[name = string("hidden_states_547_cast_fp16")]; + tensor inputs_443_cast_fp16 = add(x = inputs_441_cast_fp16, y = hidden_states_547_cast_fp16)[name = string("inputs_443_cast_fp16")]; + tensor obj_471_begin_0 = const()[name = string("obj_471_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_471_end_0 = const()[name = string("obj_471_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_471_end_mask_0 = const()[name = string("obj_471_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_471_cast_fp16 = slice_by_index(begin = obj_471_begin_0, end = obj_471_end_0, end_mask = obj_471_end_mask_0, x = key_caches_21_cast_fp16)[name = string("obj_471_cast_fp16")]; + tensor obj_473_begin_0 = const()[name = string("obj_473_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_473_end_0 = const()[name = string("obj_473_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_473_end_mask_0 = const()[name = string("obj_473_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_473_cast_fp16 = slice_by_index(begin = obj_473_begin_0, end = obj_473_end_0, end_mask = obj_473_end_mask_0, x = value_caches_21_cast_fp16)[name = string("obj_473_cast_fp16")]; + int32 var_16236 = const()[name = string("op_16236"), val = int32(3)]; + int32 var_16246 = const()[name = string("op_16246"), val = int32(-2)]; + tensor inputs_sq_443_cast_fp16 = mul(x = inputs_443_cast_fp16, y = inputs_443_cast_fp16)[name = string("inputs_sq_443_cast_fp16")]; + tensor variance_443_axes_0 = const()[name = string("variance_443_axes_0"), val = tensor([1])]; + bool variance_443_keep_dims_0 = const()[name = string("variance_443_keep_dims_0"), val = bool(true)]; + tensor variance_443_cast_fp16 = reduce_mean(axes = variance_443_axes_0, keep_dims = variance_443_keep_dims_0, x = inputs_sq_443_cast_fp16)[name = string("variance_443_cast_fp16")]; + fp16 var_16260_to_fp16 = const()[name = string("op_16260_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16261_cast_fp16 = add(x = variance_443_cast_fp16, y = var_16260_to_fp16)[name = string("op_16261_cast_fp16")]; + fp32 var_16262_epsilon_0 = const()[name = string("op_16262_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16262_cast_fp16 = rsqrt(epsilon = var_16262_epsilon_0, x = var_16261_cast_fp16)[name = string("op_16262_cast_fp16")]; + tensor hidden_states_549_cast_fp16 = mul(x = inputs_443_cast_fp16, y = var_16262_cast_fp16)[name = string("hidden_states_549_cast_fp16")]; + tensor obj_469_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_549_cast_fp16)[name = string("obj_469_cast_fp16")]; + string query_319_pad_type_0 = const()[name = string("query_319_pad_type_0"), val = string("valid")]; + tensor query_319_strides_0 = const()[name = string("query_319_strides_0"), val = tensor([1, 1])]; + tensor query_319_pad_0 = const()[name = string("query_319_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_319_dilations_0 = const()[name = string("query_319_dilations_0"), val = tensor([1, 1])]; + int32 query_319_groups_0 = const()[name = string("query_319_groups_0"), val = int32(1)]; + tensor query_319_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_319_dilations_0, groups = query_319_groups_0, pad = query_319_pad_0, pad_type = query_319_pad_type_0, strides = query_319_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_469_cast_fp16)[name = string("query_319_cast_fp16")]; + string current_key_213_pad_type_0 = const()[name = string("current_key_213_pad_type_0"), val = string("valid")]; + tensor current_key_213_strides_0 = const()[name = string("current_key_213_strides_0"), val = tensor([1, 1])]; + tensor current_key_213_pad_0 = const()[name = string("current_key_213_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_213_dilations_0 = const()[name = string("current_key_213_dilations_0"), val = tensor([1, 1])]; + int32 current_key_213_groups_0 = const()[name = string("current_key_213_groups_0"), val = int32(1)]; + tensor current_key_213_cast_fp16 = conv(dilations = current_key_213_dilations_0, groups = current_key_213_groups_0, pad = current_key_213_pad_0, pad_type = current_key_213_pad_type_0, strides = current_key_213_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_469_cast_fp16)[name = string("current_key_213_cast_fp16")]; + string current_value_107_pad_type_0 = const()[name = string("current_value_107_pad_type_0"), val = string("valid")]; + tensor current_value_107_strides_0 = const()[name = string("current_value_107_strides_0"), val = tensor([1, 1])]; + tensor current_value_107_pad_0 = const()[name = string("current_value_107_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_107_dilations_0 = const()[name = string("current_value_107_dilations_0"), val = tensor([1, 1])]; + int32 current_value_107_groups_0 = const()[name = string("current_value_107_groups_0"), val = int32(1)]; + tensor current_value_107_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_107_dilations_0, groups = current_value_107_groups_0, pad = current_value_107_pad_0, pad_type = current_value_107_pad_type_0, strides = current_value_107_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_469_cast_fp16)[name = string("current_value_107_cast_fp16")]; + tensor var_16299 = const()[name = string("op_16299"), val = tensor([16, 128, 1, 1])]; + tensor inputs_445_cast_fp16 = reshape(shape = var_16299, x = query_319_cast_fp16)[name = string("inputs_445_cast_fp16")]; + tensor inputs_sq_445_cast_fp16 = mul(x = inputs_445_cast_fp16, y = inputs_445_cast_fp16)[name = string("inputs_sq_445_cast_fp16")]; + tensor variance_445_axes_0 = const()[name = string("variance_445_axes_0"), val = tensor([1])]; + bool variance_445_keep_dims_0 = const()[name = string("variance_445_keep_dims_0"), val = bool(true)]; + tensor variance_445_cast_fp16 = reduce_mean(axes = variance_445_axes_0, keep_dims = variance_445_keep_dims_0, x = inputs_sq_445_cast_fp16)[name = string("variance_445_cast_fp16")]; + fp16 var_16305_to_fp16 = const()[name = string("op_16305_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16306_cast_fp16 = add(x = variance_445_cast_fp16, y = var_16305_to_fp16)[name = string("op_16306_cast_fp16")]; + fp32 var_16307_epsilon_0 = const()[name = string("op_16307_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16307_cast_fp16 = rsqrt(epsilon = var_16307_epsilon_0, x = var_16306_cast_fp16)[name = string("op_16307_cast_fp16")]; + tensor hidden_states_551_cast_fp16 = mul(x = inputs_445_cast_fp16, y = var_16307_cast_fp16)[name = string("hidden_states_551_cast_fp16")]; + tensor query_normed_107_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_551_cast_fp16)[name = string("query_normed_107_cast_fp16")]; + tensor var_16315 = const()[name = string("op_16315"), val = tensor([8, 128, 1, 1])]; + tensor inputs_447_cast_fp16 = reshape(shape = var_16315, x = current_key_213_cast_fp16)[name = string("inputs_447_cast_fp16")]; + tensor inputs_sq_447_cast_fp16 = mul(x = inputs_447_cast_fp16, y = inputs_447_cast_fp16)[name = string("inputs_sq_447_cast_fp16")]; + tensor variance_447_axes_0 = const()[name = string("variance_447_axes_0"), val = tensor([1])]; + bool variance_447_keep_dims_0 = const()[name = string("variance_447_keep_dims_0"), val = bool(true)]; + tensor variance_447_cast_fp16 = reduce_mean(axes = variance_447_axes_0, keep_dims = variance_447_keep_dims_0, x = inputs_sq_447_cast_fp16)[name = string("variance_447_cast_fp16")]; + fp16 var_16321_to_fp16 = const()[name = string("op_16321_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16322_cast_fp16 = add(x = variance_447_cast_fp16, y = var_16321_to_fp16)[name = string("op_16322_cast_fp16")]; + fp32 var_16323_epsilon_0 = const()[name = string("op_16323_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16323_cast_fp16 = rsqrt(epsilon = var_16323_epsilon_0, x = var_16322_cast_fp16)[name = string("op_16323_cast_fp16")]; + tensor hidden_states_553_cast_fp16 = mul(x = inputs_447_cast_fp16, y = var_16323_cast_fp16)[name = string("hidden_states_553_cast_fp16")]; + tensor current_key_normed_107_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_553_cast_fp16)[name = string("current_key_normed_107_cast_fp16")]; + tensor var_16341 = const()[name = string("op_16341"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_425_cast_fp16 = reshape(shape = var_16341, x = query_normed_107_cast_fp16)[name = string("mh_q_425_cast_fp16")]; + tensor var_16343 = const()[name = string("op_16343"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_425_cast_fp16 = reshape(shape = var_16343, x = current_key_normed_107_cast_fp16)[name = string("mh_k_425_cast_fp16")]; + tensor var_16347_cast_fp16 = mul(x = mh_q_425_cast_fp16, y = cos_101_to_fp16)[name = string("op_16347_cast_fp16")]; + tensor var_16352_begin_0 = const()[name = string("op_16352_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16352_end_0 = const()[name = string("op_16352_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_16352_end_mask_0 = const()[name = string("op_16352_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_16352_cast_fp16 = slice_by_index(begin = var_16352_begin_0, end = var_16352_end_0, end_mask = var_16352_end_mask_0, x = mh_q_425_cast_fp16)[name = string("op_16352_cast_fp16")]; + tensor var_16358_begin_0 = const()[name = string("op_16358_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_16358_end_0 = const()[name = string("op_16358_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_16358_end_mask_0 = const()[name = string("op_16358_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_16358_cast_fp16 = slice_by_index(begin = var_16358_begin_0, end = var_16358_end_0, end_mask = var_16358_end_mask_0, x = mh_q_425_cast_fp16)[name = string("op_16358_cast_fp16")]; + fp16 const_1084_promoted_to_fp16 = const()[name = string("const_1084_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16360_cast_fp16 = mul(x = var_16358_cast_fp16, y = const_1084_promoted_to_fp16)[name = string("op_16360_cast_fp16")]; + bool var_16362_interleave_0 = const()[name = string("op_16362_interleave_0"), val = bool(false)]; + tensor var_16362_cast_fp16 = concat(axis = var_16246, interleave = var_16362_interleave_0, values = (var_16360_cast_fp16, var_16352_cast_fp16))[name = string("op_16362_cast_fp16")]; + tensor var_16363_cast_fp16 = mul(x = var_16362_cast_fp16, y = sin_101_to_fp16)[name = string("op_16363_cast_fp16")]; + tensor mh_q_427_cast_fp16 = add(x = var_16347_cast_fp16, y = var_16363_cast_fp16)[name = string("mh_q_427_cast_fp16")]; + tensor var_16365_cast_fp16 = mul(x = mh_k_425_cast_fp16, y = cos_101_to_fp16)[name = string("op_16365_cast_fp16")]; + tensor var_16370_begin_0 = const()[name = string("op_16370_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16370_end_0 = const()[name = string("op_16370_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_16370_end_mask_0 = const()[name = string("op_16370_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_16370_cast_fp16 = slice_by_index(begin = var_16370_begin_0, end = var_16370_end_0, end_mask = var_16370_end_mask_0, x = mh_k_425_cast_fp16)[name = string("op_16370_cast_fp16")]; + tensor var_16376_begin_0 = const()[name = string("op_16376_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_16376_end_0 = const()[name = string("op_16376_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_16376_end_mask_0 = const()[name = string("op_16376_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_16376_cast_fp16 = slice_by_index(begin = var_16376_begin_0, end = var_16376_end_0, end_mask = var_16376_end_mask_0, x = mh_k_425_cast_fp16)[name = string("op_16376_cast_fp16")]; + fp16 const_1087_promoted_to_fp16 = const()[name = string("const_1087_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16378_cast_fp16 = mul(x = var_16376_cast_fp16, y = const_1087_promoted_to_fp16)[name = string("op_16378_cast_fp16")]; + bool var_16380_interleave_0 = const()[name = string("op_16380_interleave_0"), val = bool(false)]; + tensor var_16380_cast_fp16 = concat(axis = var_16246, interleave = var_16380_interleave_0, values = (var_16378_cast_fp16, var_16370_cast_fp16))[name = string("op_16380_cast_fp16")]; + tensor var_16381_cast_fp16 = mul(x = var_16380_cast_fp16, y = sin_101_to_fp16)[name = string("op_16381_cast_fp16")]; + tensor mh_k_427_cast_fp16 = add(x = var_16365_cast_fp16, y = var_16381_cast_fp16)[name = string("mh_k_427_cast_fp16")]; + tensor var_16385 = const()[name = string("op_16385"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_215_cast_fp16 = reshape(shape = var_16385, x = mh_k_427_cast_fp16)[name = string("current_key_215_cast_fp16")]; + tensor var_16391_to_fp16 = const()[name = string("op_16391_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199424)))]; + tensor var_16392_cast_fp16 = mul(x = obj_471_cast_fp16, y = var_16391_to_fp16)[name = string("op_16392_cast_fp16")]; + tensor var_16389_to_fp16 = const()[name = string("op_16389_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199552)))]; + tensor var_16393_cast_fp16 = mul(x = current_key_215_cast_fp16, y = var_16389_to_fp16)[name = string("op_16393_cast_fp16")]; + tensor key_215_cast_fp16 = add(x = var_16392_cast_fp16, y = var_16393_cast_fp16)[name = string("key_215_cast_fp16")]; + tensor var_16395_to_fp16 = const()[name = string("op_16395_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199424)))]; + tensor var_16396_cast_fp16 = mul(x = obj_473_cast_fp16, y = var_16395_to_fp16)[name = string("op_16396_cast_fp16")]; + tensor var_16397_cast_fp16 = mul(x = current_value_107_cast_fp16, y = var_16389_to_fp16)[name = string("op_16397_cast_fp16")]; + tensor value_107_cast_fp16 = add(x = var_16396_cast_fp16, y = var_16397_cast_fp16)[name = string("value_107_cast_fp16")]; + fp16 var_16404_to_fp16 = const()[name = string("op_16404_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_431_cast_fp16 = mul(x = mh_q_427_cast_fp16, y = var_16404_to_fp16)[name = string("mh_q_431_cast_fp16")]; + tensor var_16406 = const()[name = string("op_16406"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_429_cast_fp16 = reshape(shape = var_16406, x = key_215_cast_fp16)[name = string("mh_k_429_cast_fp16")]; + tensor var_16408 = const()[name = string("op_16408"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_213_cast_fp16 = reshape(shape = var_16408, x = value_107_cast_fp16)[name = string("mh_v_213_cast_fp16")]; + tensor transpose_212_perm_0 = const()[name = string("transpose_212_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_106_reps_0 = const()[name = string("tile_106_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_212_cast_fp16 = transpose(perm = transpose_212_perm_0, x = mh_k_429_cast_fp16)[name = string("transpose_161")]; + tensor tile_106_cast_fp16 = tile(reps = tile_106_reps_0, x = transpose_212_cast_fp16)[name = string("tile_106_cast_fp16")]; + tensor concat_265 = const()[name = string("concat_265"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_212_cast_fp16 = reshape(shape = concat_265, x = tile_106_cast_fp16)[name = string("reshape_212_cast_fp16")]; + tensor transpose_213_perm_0 = const()[name = string("transpose_213_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_266 = const()[name = string("concat_266"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_213_cast_fp16 = transpose(perm = transpose_213_perm_0, x = reshape_212_cast_fp16)[name = string("transpose_160")]; + tensor reshape_213_cast_fp16 = reshape(shape = concat_266, x = transpose_213_cast_fp16)[name = string("reshape_213_cast_fp16")]; + tensor transpose_214_perm_0 = const()[name = string("transpose_214_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_107_reps_0 = const()[name = string("tile_107_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_214_cast_fp16 = transpose(perm = transpose_214_perm_0, x = mh_v_213_cast_fp16)[name = string("transpose_159")]; + tensor tile_107_cast_fp16 = tile(reps = tile_107_reps_0, x = transpose_214_cast_fp16)[name = string("tile_107_cast_fp16")]; + tensor concat_267 = const()[name = string("concat_267"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_214_cast_fp16 = reshape(shape = concat_267, x = tile_107_cast_fp16)[name = string("reshape_214_cast_fp16")]; + tensor transpose_215_perm_0 = const()[name = string("transpose_215_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_268 = const()[name = string("concat_268"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_215_cast_fp16 = transpose(perm = transpose_215_perm_0, x = reshape_214_cast_fp16)[name = string("transpose_158")]; + tensor reshape_215_cast_fp16 = reshape(shape = concat_268, x = transpose_215_cast_fp16)[name = string("reshape_215_cast_fp16")]; + tensor transpose_529_perm_0 = const()[name = string("transpose_529_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_319_transpose_x_1 = const()[name = string("mh_w_319_transpose_x_1"), val = bool(true)]; + bool mh_w_319_transpose_y_1 = const()[name = string("mh_w_319_transpose_y_1"), val = bool(false)]; + tensor transpose_529_cast_fp16 = transpose(perm = transpose_529_perm_0, x = reshape_213_cast_fp16)[name = string("transpose_157")]; + tensor mh_w_319_cast_fp16 = matmul(transpose_x = mh_w_319_transpose_x_1, transpose_y = mh_w_319_transpose_y_1, x = mh_q_431_cast_fp16, y = transpose_529_cast_fp16)[name = string("mh_w_319_cast_fp16")]; + tensor var_16416_to_fp16 = const()[name = string("op_16416_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199680)))]; + tensor mh_w_321_cast_fp16 = add(x = mh_w_319_cast_fp16, y = var_16416_to_fp16)[name = string("mh_w_321_cast_fp16")]; + tensor mh_w_323_cast_fp16 = softmax(axis = var_16236, x = mh_w_321_cast_fp16)[name = string("mh_w_323_cast_fp16")]; + tensor transpose_530_perm_0 = const()[name = string("transpose_530_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_107_transpose_x_1 = const()[name = string("attn_107_transpose_x_1"), val = bool(false)]; + bool attn_107_transpose_y_1 = const()[name = string("attn_107_transpose_y_1"), val = bool(true)]; + tensor transpose_530_cast_fp16 = transpose(perm = transpose_530_perm_0, x = reshape_215_cast_fp16)[name = string("transpose_156")]; + tensor attn_107_cast_fp16 = matmul(transpose_x = attn_107_transpose_x_1, transpose_y = attn_107_transpose_y_1, x = transpose_530_cast_fp16, y = mh_w_323_cast_fp16)[name = string("attn_107_cast_fp16")]; + tensor var_16422 = const()[name = string("op_16422"), val = tensor([1, 2048, 1, 1])]; + tensor input_461_cast_fp16 = reshape(shape = var_16422, x = attn_107_cast_fp16)[name = string("input_461_cast_fp16")]; + string obj_475_pad_type_0 = const()[name = string("obj_475_pad_type_0"), val = string("valid")]; + tensor obj_475_strides_0 = const()[name = string("obj_475_strides_0"), val = tensor([1, 1])]; + tensor obj_475_pad_0 = const()[name = string("obj_475_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_475_dilations_0 = const()[name = string("obj_475_dilations_0"), val = tensor([1, 1])]; + int32 obj_475_groups_0 = const()[name = string("obj_475_groups_0"), val = int32(1)]; + tensor obj_475_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_475_dilations_0, groups = obj_475_groups_0, pad = obj_475_pad_0, pad_type = obj_475_pad_type_0, strides = obj_475_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_461_cast_fp16)[name = string("obj_475_cast_fp16")]; + tensor inputs_449_cast_fp16 = add(x = inputs_443_cast_fp16, y = obj_475_cast_fp16)[name = string("inputs_449_cast_fp16")]; + tensor inputs_sq_449_cast_fp16 = mul(x = inputs_449_cast_fp16, y = inputs_449_cast_fp16)[name = string("inputs_sq_449_cast_fp16")]; + tensor variance_449_axes_0 = const()[name = string("variance_449_axes_0"), val = tensor([1])]; + bool variance_449_keep_dims_0 = const()[name = string("variance_449_keep_dims_0"), val = bool(true)]; + tensor variance_449_cast_fp16 = reduce_mean(axes = variance_449_axes_0, keep_dims = variance_449_keep_dims_0, x = inputs_sq_449_cast_fp16)[name = string("variance_449_cast_fp16")]; + fp16 var_16440_to_fp16 = const()[name = string("op_16440_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16441_cast_fp16 = add(x = variance_449_cast_fp16, y = var_16440_to_fp16)[name = string("op_16441_cast_fp16")]; + fp32 var_16442_epsilon_0 = const()[name = string("op_16442_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16442_cast_fp16 = rsqrt(epsilon = var_16442_epsilon_0, x = var_16441_cast_fp16)[name = string("op_16442_cast_fp16")]; + tensor hidden_states_555_cast_fp16 = mul(x = inputs_449_cast_fp16, y = var_16442_cast_fp16)[name = string("hidden_states_555_cast_fp16")]; + tensor input_463_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_555_cast_fp16)[name = string("input_463_cast_fp16")]; + string input_465_pad_type_0 = const()[name = string("input_465_pad_type_0"), val = string("valid")]; + tensor input_465_strides_0 = const()[name = string("input_465_strides_0"), val = tensor([1, 1])]; + tensor input_465_pad_0 = const()[name = string("input_465_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_465_dilations_0 = const()[name = string("input_465_dilations_0"), val = tensor([1, 1])]; + int32 input_465_groups_0 = const()[name = string("input_465_groups_0"), val = int32(1)]; + tensor input_465_cast_fp16 = conv(dilations = input_465_dilations_0, groups = input_465_groups_0, pad = input_465_pad_0, pad_type = input_465_pad_type_0, strides = input_465_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_463_cast_fp16)[name = string("input_465_cast_fp16")]; + tensor var_16456_cast_fp16 = silu(x = input_465_cast_fp16)[name = string("op_16456_cast_fp16")]; + string var_16462_pad_type_0 = const()[name = string("op_16462_pad_type_0"), val = string("valid")]; + tensor var_16462_strides_0 = const()[name = string("op_16462_strides_0"), val = tensor([1, 1])]; + tensor var_16462_pad_0 = const()[name = string("op_16462_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16462_dilations_0 = const()[name = string("op_16462_dilations_0"), val = tensor([1, 1])]; + int32 var_16462_groups_0 = const()[name = string("op_16462_groups_0"), val = int32(1)]; + tensor var_16462_cast_fp16 = conv(dilations = var_16462_dilations_0, groups = var_16462_groups_0, pad = var_16462_pad_0, pad_type = var_16462_pad_type_0, strides = var_16462_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_463_cast_fp16)[name = string("op_16462_cast_fp16")]; + tensor input_467_cast_fp16 = mul(x = var_16456_cast_fp16, y = var_16462_cast_fp16)[name = string("input_467_cast_fp16")]; + string hidden_states_557_pad_type_0 = const()[name = string("hidden_states_557_pad_type_0"), val = string("valid")]; + tensor hidden_states_557_strides_0 = const()[name = string("hidden_states_557_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_557_pad_0 = const()[name = string("hidden_states_557_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_557_dilations_0 = const()[name = string("hidden_states_557_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_557_groups_0 = const()[name = string("hidden_states_557_groups_0"), val = int32(1)]; + tensor hidden_states_557_cast_fp16 = conv(dilations = hidden_states_557_dilations_0, groups = hidden_states_557_groups_0, pad = hidden_states_557_pad_0, pad_type = hidden_states_557_pad_type_0, strides = hidden_states_557_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_467_cast_fp16)[name = string("hidden_states_557_cast_fp16")]; + tensor inputs_451_cast_fp16 = add(x = inputs_449_cast_fp16, y = hidden_states_557_cast_fp16)[name = string("inputs_451_cast_fp16")]; + tensor obj_479_begin_0 = const()[name = string("obj_479_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_479_end_0 = const()[name = string("obj_479_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_479_end_mask_0 = const()[name = string("obj_479_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_479_cast_fp16 = slice_by_index(begin = obj_479_begin_0, end = obj_479_end_0, end_mask = obj_479_end_mask_0, x = key_caches_21_cast_fp16)[name = string("obj_479_cast_fp16")]; + tensor obj_481_begin_0 = const()[name = string("obj_481_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_481_end_0 = const()[name = string("obj_481_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_481_end_mask_0 = const()[name = string("obj_481_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_481_cast_fp16 = slice_by_index(begin = obj_481_begin_0, end = obj_481_end_0, end_mask = obj_481_end_mask_0, x = value_caches_21_cast_fp16)[name = string("obj_481_cast_fp16")]; + int32 var_16510 = const()[name = string("op_16510"), val = int32(3)]; + int32 var_16520 = const()[name = string("op_16520"), val = int32(-2)]; + tensor inputs_sq_451_cast_fp16 = mul(x = inputs_451_cast_fp16, y = inputs_451_cast_fp16)[name = string("inputs_sq_451_cast_fp16")]; + tensor variance_451_axes_0 = const()[name = string("variance_451_axes_0"), val = tensor([1])]; + bool variance_451_keep_dims_0 = const()[name = string("variance_451_keep_dims_0"), val = bool(true)]; + tensor variance_451_cast_fp16 = reduce_mean(axes = variance_451_axes_0, keep_dims = variance_451_keep_dims_0, x = inputs_sq_451_cast_fp16)[name = string("variance_451_cast_fp16")]; + fp16 var_16534_to_fp16 = const()[name = string("op_16534_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16535_cast_fp16 = add(x = variance_451_cast_fp16, y = var_16534_to_fp16)[name = string("op_16535_cast_fp16")]; + fp32 var_16536_epsilon_0 = const()[name = string("op_16536_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16536_cast_fp16 = rsqrt(epsilon = var_16536_epsilon_0, x = var_16535_cast_fp16)[name = string("op_16536_cast_fp16")]; + tensor hidden_states_559_cast_fp16 = mul(x = inputs_451_cast_fp16, y = var_16536_cast_fp16)[name = string("hidden_states_559_cast_fp16")]; + tensor obj_477_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_559_cast_fp16)[name = string("obj_477_cast_fp16")]; + string query_325_pad_type_0 = const()[name = string("query_325_pad_type_0"), val = string("valid")]; + tensor query_325_strides_0 = const()[name = string("query_325_strides_0"), val = tensor([1, 1])]; + tensor query_325_pad_0 = const()[name = string("query_325_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_325_dilations_0 = const()[name = string("query_325_dilations_0"), val = tensor([1, 1])]; + int32 query_325_groups_0 = const()[name = string("query_325_groups_0"), val = int32(1)]; + tensor query_325_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_325_dilations_0, groups = query_325_groups_0, pad = query_325_pad_0, pad_type = query_325_pad_type_0, strides = query_325_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_477_cast_fp16)[name = string("query_325_cast_fp16")]; + string current_key_217_pad_type_0 = const()[name = string("current_key_217_pad_type_0"), val = string("valid")]; + tensor current_key_217_strides_0 = const()[name = string("current_key_217_strides_0"), val = tensor([1, 1])]; + tensor current_key_217_pad_0 = const()[name = string("current_key_217_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_217_dilations_0 = const()[name = string("current_key_217_dilations_0"), val = tensor([1, 1])]; + int32 current_key_217_groups_0 = const()[name = string("current_key_217_groups_0"), val = int32(1)]; + tensor current_key_217_cast_fp16 = conv(dilations = current_key_217_dilations_0, groups = current_key_217_groups_0, pad = current_key_217_pad_0, pad_type = current_key_217_pad_type_0, strides = current_key_217_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_477_cast_fp16)[name = string("current_key_217_cast_fp16")]; + string current_value_109_pad_type_0 = const()[name = string("current_value_109_pad_type_0"), val = string("valid")]; + tensor current_value_109_strides_0 = const()[name = string("current_value_109_strides_0"), val = tensor([1, 1])]; + tensor current_value_109_pad_0 = const()[name = string("current_value_109_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_109_dilations_0 = const()[name = string("current_value_109_dilations_0"), val = tensor([1, 1])]; + int32 current_value_109_groups_0 = const()[name = string("current_value_109_groups_0"), val = int32(1)]; + tensor current_value_109_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_109_dilations_0, groups = current_value_109_groups_0, pad = current_value_109_pad_0, pad_type = current_value_109_pad_type_0, strides = current_value_109_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_477_cast_fp16)[name = string("current_value_109_cast_fp16")]; + tensor var_16573 = const()[name = string("op_16573"), val = tensor([16, 128, 1, 1])]; + tensor inputs_453_cast_fp16 = reshape(shape = var_16573, x = query_325_cast_fp16)[name = string("inputs_453_cast_fp16")]; + tensor inputs_sq_453_cast_fp16 = mul(x = inputs_453_cast_fp16, y = inputs_453_cast_fp16)[name = string("inputs_sq_453_cast_fp16")]; + tensor variance_453_axes_0 = const()[name = string("variance_453_axes_0"), val = tensor([1])]; + bool variance_453_keep_dims_0 = const()[name = string("variance_453_keep_dims_0"), val = bool(true)]; + tensor variance_453_cast_fp16 = reduce_mean(axes = variance_453_axes_0, keep_dims = variance_453_keep_dims_0, x = inputs_sq_453_cast_fp16)[name = string("variance_453_cast_fp16")]; + fp16 var_16579_to_fp16 = const()[name = string("op_16579_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16580_cast_fp16 = add(x = variance_453_cast_fp16, y = var_16579_to_fp16)[name = string("op_16580_cast_fp16")]; + fp32 var_16581_epsilon_0 = const()[name = string("op_16581_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16581_cast_fp16 = rsqrt(epsilon = var_16581_epsilon_0, x = var_16580_cast_fp16)[name = string("op_16581_cast_fp16")]; + tensor hidden_states_561_cast_fp16 = mul(x = inputs_453_cast_fp16, y = var_16581_cast_fp16)[name = string("hidden_states_561_cast_fp16")]; + tensor query_normed_109_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_561_cast_fp16)[name = string("query_normed_109_cast_fp16")]; + tensor var_16589 = const()[name = string("op_16589"), val = tensor([8, 128, 1, 1])]; + tensor inputs_455_cast_fp16 = reshape(shape = var_16589, x = current_key_217_cast_fp16)[name = string("inputs_455_cast_fp16")]; + tensor inputs_sq_455_cast_fp16 = mul(x = inputs_455_cast_fp16, y = inputs_455_cast_fp16)[name = string("inputs_sq_455_cast_fp16")]; + tensor variance_455_axes_0 = const()[name = string("variance_455_axes_0"), val = tensor([1])]; + bool variance_455_keep_dims_0 = const()[name = string("variance_455_keep_dims_0"), val = bool(true)]; + tensor variance_455_cast_fp16 = reduce_mean(axes = variance_455_axes_0, keep_dims = variance_455_keep_dims_0, x = inputs_sq_455_cast_fp16)[name = string("variance_455_cast_fp16")]; + fp16 var_16595_to_fp16 = const()[name = string("op_16595_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16596_cast_fp16 = add(x = variance_455_cast_fp16, y = var_16595_to_fp16)[name = string("op_16596_cast_fp16")]; + fp32 var_16597_epsilon_0 = const()[name = string("op_16597_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16597_cast_fp16 = rsqrt(epsilon = var_16597_epsilon_0, x = var_16596_cast_fp16)[name = string("op_16597_cast_fp16")]; + tensor hidden_states_563_cast_fp16 = mul(x = inputs_455_cast_fp16, y = var_16597_cast_fp16)[name = string("hidden_states_563_cast_fp16")]; + tensor current_key_normed_109_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_563_cast_fp16)[name = string("current_key_normed_109_cast_fp16")]; + tensor var_16615 = const()[name = string("op_16615"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_433_cast_fp16 = reshape(shape = var_16615, x = query_normed_109_cast_fp16)[name = string("mh_q_433_cast_fp16")]; + tensor var_16617 = const()[name = string("op_16617"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_433_cast_fp16 = reshape(shape = var_16617, x = current_key_normed_109_cast_fp16)[name = string("mh_k_433_cast_fp16")]; + tensor var_16621_cast_fp16 = mul(x = mh_q_433_cast_fp16, y = cos_101_to_fp16)[name = string("op_16621_cast_fp16")]; + tensor var_16626_begin_0 = const()[name = string("op_16626_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16626_end_0 = const()[name = string("op_16626_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_16626_end_mask_0 = const()[name = string("op_16626_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_16626_cast_fp16 = slice_by_index(begin = var_16626_begin_0, end = var_16626_end_0, end_mask = var_16626_end_mask_0, x = mh_q_433_cast_fp16)[name = string("op_16626_cast_fp16")]; + tensor var_16632_begin_0 = const()[name = string("op_16632_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_16632_end_0 = const()[name = string("op_16632_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_16632_end_mask_0 = const()[name = string("op_16632_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_16632_cast_fp16 = slice_by_index(begin = var_16632_begin_0, end = var_16632_end_0, end_mask = var_16632_end_mask_0, x = mh_q_433_cast_fp16)[name = string("op_16632_cast_fp16")]; + fp16 const_1104_promoted_to_fp16 = const()[name = string("const_1104_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16634_cast_fp16 = mul(x = var_16632_cast_fp16, y = const_1104_promoted_to_fp16)[name = string("op_16634_cast_fp16")]; + bool var_16636_interleave_0 = const()[name = string("op_16636_interleave_0"), val = bool(false)]; + tensor var_16636_cast_fp16 = concat(axis = var_16520, interleave = var_16636_interleave_0, values = (var_16634_cast_fp16, var_16626_cast_fp16))[name = string("op_16636_cast_fp16")]; + tensor var_16637_cast_fp16 = mul(x = var_16636_cast_fp16, y = sin_101_to_fp16)[name = string("op_16637_cast_fp16")]; + tensor mh_q_435_cast_fp16 = add(x = var_16621_cast_fp16, y = var_16637_cast_fp16)[name = string("mh_q_435_cast_fp16")]; + tensor var_16639_cast_fp16 = mul(x = mh_k_433_cast_fp16, y = cos_101_to_fp16)[name = string("op_16639_cast_fp16")]; + tensor var_16644_begin_0 = const()[name = string("op_16644_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16644_end_0 = const()[name = string("op_16644_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_16644_end_mask_0 = const()[name = string("op_16644_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_16644_cast_fp16 = slice_by_index(begin = var_16644_begin_0, end = var_16644_end_0, end_mask = var_16644_end_mask_0, x = mh_k_433_cast_fp16)[name = string("op_16644_cast_fp16")]; + tensor var_16650_begin_0 = const()[name = string("op_16650_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_16650_end_0 = const()[name = string("op_16650_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_16650_end_mask_0 = const()[name = string("op_16650_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_16650_cast_fp16 = slice_by_index(begin = var_16650_begin_0, end = var_16650_end_0, end_mask = var_16650_end_mask_0, x = mh_k_433_cast_fp16)[name = string("op_16650_cast_fp16")]; + fp16 const_1107_promoted_to_fp16 = const()[name = string("const_1107_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_16652_cast_fp16 = mul(x = var_16650_cast_fp16, y = const_1107_promoted_to_fp16)[name = string("op_16652_cast_fp16")]; + bool var_16654_interleave_0 = const()[name = string("op_16654_interleave_0"), val = bool(false)]; + tensor var_16654_cast_fp16 = concat(axis = var_16520, interleave = var_16654_interleave_0, values = (var_16652_cast_fp16, var_16644_cast_fp16))[name = string("op_16654_cast_fp16")]; + tensor var_16655_cast_fp16 = mul(x = var_16654_cast_fp16, y = sin_101_to_fp16)[name = string("op_16655_cast_fp16")]; + tensor mh_k_435_cast_fp16 = add(x = var_16639_cast_fp16, y = var_16655_cast_fp16)[name = string("mh_k_435_cast_fp16")]; + tensor var_16659 = const()[name = string("op_16659"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_219_cast_fp16 = reshape(shape = var_16659, x = mh_k_435_cast_fp16)[name = string("current_key_219_cast_fp16")]; + tensor var_16665_to_fp16 = const()[name = string("op_16665_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199424)))]; + tensor var_16666_cast_fp16 = mul(x = obj_479_cast_fp16, y = var_16665_to_fp16)[name = string("op_16666_cast_fp16")]; + tensor var_16663_to_fp16 = const()[name = string("op_16663_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199552)))]; + tensor var_16667_cast_fp16 = mul(x = current_key_219_cast_fp16, y = var_16663_to_fp16)[name = string("op_16667_cast_fp16")]; + tensor key_219_cast_fp16 = add(x = var_16666_cast_fp16, y = var_16667_cast_fp16)[name = string("key_219_cast_fp16")]; + tensor var_16669_to_fp16 = const()[name = string("op_16669_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199424)))]; + tensor var_16670_cast_fp16 = mul(x = obj_481_cast_fp16, y = var_16669_to_fp16)[name = string("op_16670_cast_fp16")]; + tensor var_16671_cast_fp16 = mul(x = current_value_109_cast_fp16, y = var_16663_to_fp16)[name = string("op_16671_cast_fp16")]; + tensor value_109_cast_fp16 = add(x = var_16670_cast_fp16, y = var_16671_cast_fp16)[name = string("value_109_cast_fp16")]; + fp16 var_16678_to_fp16 = const()[name = string("op_16678_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_439_cast_fp16 = mul(x = mh_q_435_cast_fp16, y = var_16678_to_fp16)[name = string("mh_q_439_cast_fp16")]; + tensor var_16680 = const()[name = string("op_16680"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_437_cast_fp16 = reshape(shape = var_16680, x = key_219_cast_fp16)[name = string("mh_k_437_cast_fp16")]; + tensor var_16682 = const()[name = string("op_16682"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_217_cast_fp16 = reshape(shape = var_16682, x = value_109_cast_fp16)[name = string("mh_v_217_cast_fp16")]; + tensor transpose_216_perm_0 = const()[name = string("transpose_216_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_108_reps_0 = const()[name = string("tile_108_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_216_cast_fp16 = transpose(perm = transpose_216_perm_0, x = mh_k_437_cast_fp16)[name = string("transpose_155")]; + tensor tile_108_cast_fp16 = tile(reps = tile_108_reps_0, x = transpose_216_cast_fp16)[name = string("tile_108_cast_fp16")]; + tensor concat_269 = const()[name = string("concat_269"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_216_cast_fp16 = reshape(shape = concat_269, x = tile_108_cast_fp16)[name = string("reshape_216_cast_fp16")]; + tensor transpose_217_perm_0 = const()[name = string("transpose_217_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_270 = const()[name = string("concat_270"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_217_cast_fp16 = transpose(perm = transpose_217_perm_0, x = reshape_216_cast_fp16)[name = string("transpose_154")]; + tensor reshape_217_cast_fp16 = reshape(shape = concat_270, x = transpose_217_cast_fp16)[name = string("reshape_217_cast_fp16")]; + tensor transpose_218_perm_0 = const()[name = string("transpose_218_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_109_reps_0 = const()[name = string("tile_109_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_218_cast_fp16 = transpose(perm = transpose_218_perm_0, x = mh_v_217_cast_fp16)[name = string("transpose_153")]; + tensor tile_109_cast_fp16 = tile(reps = tile_109_reps_0, x = transpose_218_cast_fp16)[name = string("tile_109_cast_fp16")]; + tensor concat_271 = const()[name = string("concat_271"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_218_cast_fp16 = reshape(shape = concat_271, x = tile_109_cast_fp16)[name = string("reshape_218_cast_fp16")]; + tensor transpose_219_perm_0 = const()[name = string("transpose_219_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_272 = const()[name = string("concat_272"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_219_cast_fp16 = transpose(perm = transpose_219_perm_0, x = reshape_218_cast_fp16)[name = string("transpose_152")]; + tensor reshape_219_cast_fp16 = reshape(shape = concat_272, x = transpose_219_cast_fp16)[name = string("reshape_219_cast_fp16")]; + tensor transpose_533_perm_0 = const()[name = string("transpose_533_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_325_transpose_x_1 = const()[name = string("mh_w_325_transpose_x_1"), val = bool(true)]; + bool mh_w_325_transpose_y_1 = const()[name = string("mh_w_325_transpose_y_1"), val = bool(false)]; + tensor transpose_533_cast_fp16 = transpose(perm = transpose_533_perm_0, x = reshape_217_cast_fp16)[name = string("transpose_151")]; + tensor mh_w_325_cast_fp16 = matmul(transpose_x = mh_w_325_transpose_x_1, transpose_y = mh_w_325_transpose_y_1, x = mh_q_439_cast_fp16, y = transpose_533_cast_fp16)[name = string("mh_w_325_cast_fp16")]; + tensor var_16690_to_fp16 = const()[name = string("op_16690_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199680)))]; + tensor mh_w_327_cast_fp16 = add(x = mh_w_325_cast_fp16, y = var_16690_to_fp16)[name = string("mh_w_327_cast_fp16")]; + tensor mh_w_329_cast_fp16 = softmax(axis = var_16510, x = mh_w_327_cast_fp16)[name = string("mh_w_329_cast_fp16")]; + tensor transpose_534_perm_0 = const()[name = string("transpose_534_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_109_transpose_x_1 = const()[name = string("attn_109_transpose_x_1"), val = bool(false)]; + bool attn_109_transpose_y_1 = const()[name = string("attn_109_transpose_y_1"), val = bool(true)]; + tensor transpose_534_cast_fp16 = transpose(perm = transpose_534_perm_0, x = reshape_219_cast_fp16)[name = string("transpose_150")]; + tensor attn_109_cast_fp16 = matmul(transpose_x = attn_109_transpose_x_1, transpose_y = attn_109_transpose_y_1, x = transpose_534_cast_fp16, y = mh_w_329_cast_fp16)[name = string("attn_109_cast_fp16")]; + tensor var_16696 = const()[name = string("op_16696"), val = tensor([1, 2048, 1, 1])]; + tensor input_469_cast_fp16 = reshape(shape = var_16696, x = attn_109_cast_fp16)[name = string("input_469_cast_fp16")]; + string obj_483_pad_type_0 = const()[name = string("obj_483_pad_type_0"), val = string("valid")]; + tensor obj_483_strides_0 = const()[name = string("obj_483_strides_0"), val = tensor([1, 1])]; + tensor obj_483_pad_0 = const()[name = string("obj_483_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_483_dilations_0 = const()[name = string("obj_483_dilations_0"), val = tensor([1, 1])]; + int32 obj_483_groups_0 = const()[name = string("obj_483_groups_0"), val = int32(1)]; + tensor obj_483_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_483_dilations_0, groups = obj_483_groups_0, pad = obj_483_pad_0, pad_type = obj_483_pad_type_0, strides = obj_483_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_469_cast_fp16)[name = string("obj_483_cast_fp16")]; + tensor inputs_457_cast_fp16 = add(x = inputs_451_cast_fp16, y = obj_483_cast_fp16)[name = string("inputs_457_cast_fp16")]; + tensor inputs_sq_457_cast_fp16 = mul(x = inputs_457_cast_fp16, y = inputs_457_cast_fp16)[name = string("inputs_sq_457_cast_fp16")]; + tensor variance_457_axes_0 = const()[name = string("variance_457_axes_0"), val = tensor([1])]; + bool variance_457_keep_dims_0 = const()[name = string("variance_457_keep_dims_0"), val = bool(true)]; + tensor variance_457_cast_fp16 = reduce_mean(axes = variance_457_axes_0, keep_dims = variance_457_keep_dims_0, x = inputs_sq_457_cast_fp16)[name = string("variance_457_cast_fp16")]; + fp16 var_16714_to_fp16 = const()[name = string("op_16714_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16715_cast_fp16 = add(x = variance_457_cast_fp16, y = var_16714_to_fp16)[name = string("op_16715_cast_fp16")]; + fp32 var_16716_epsilon_0 = const()[name = string("op_16716_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16716_cast_fp16 = rsqrt(epsilon = var_16716_epsilon_0, x = var_16715_cast_fp16)[name = string("op_16716_cast_fp16")]; + tensor hidden_states_565_cast_fp16 = mul(x = inputs_457_cast_fp16, y = var_16716_cast_fp16)[name = string("hidden_states_565_cast_fp16")]; + tensor input_471_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_565_cast_fp16)[name = string("input_471_cast_fp16")]; + string input_473_pad_type_0 = const()[name = string("input_473_pad_type_0"), val = string("valid")]; + tensor input_473_strides_0 = const()[name = string("input_473_strides_0"), val = tensor([1, 1])]; + tensor input_473_pad_0 = const()[name = string("input_473_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_473_dilations_0 = const()[name = string("input_473_dilations_0"), val = tensor([1, 1])]; + int32 input_473_groups_0 = const()[name = string("input_473_groups_0"), val = int32(1)]; + tensor input_473_cast_fp16 = conv(dilations = input_473_dilations_0, groups = input_473_groups_0, pad = input_473_pad_0, pad_type = input_473_pad_type_0, strides = input_473_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_471_cast_fp16)[name = string("input_473_cast_fp16")]; + tensor var_16730_cast_fp16 = silu(x = input_473_cast_fp16)[name = string("op_16730_cast_fp16")]; + string var_16736_pad_type_0 = const()[name = string("op_16736_pad_type_0"), val = string("valid")]; + tensor var_16736_strides_0 = const()[name = string("op_16736_strides_0"), val = tensor([1, 1])]; + tensor var_16736_pad_0 = const()[name = string("op_16736_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_16736_dilations_0 = const()[name = string("op_16736_dilations_0"), val = tensor([1, 1])]; + int32 var_16736_groups_0 = const()[name = string("op_16736_groups_0"), val = int32(1)]; + tensor var_16736_cast_fp16 = conv(dilations = var_16736_dilations_0, groups = var_16736_groups_0, pad = var_16736_pad_0, pad_type = var_16736_pad_type_0, strides = var_16736_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_471_cast_fp16)[name = string("op_16736_cast_fp16")]; + tensor input_475_cast_fp16 = mul(x = var_16730_cast_fp16, y = var_16736_cast_fp16)[name = string("input_475_cast_fp16")]; + string hidden_states_567_pad_type_0 = const()[name = string("hidden_states_567_pad_type_0"), val = string("valid")]; + tensor hidden_states_567_strides_0 = const()[name = string("hidden_states_567_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_567_pad_0 = const()[name = string("hidden_states_567_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_567_dilations_0 = const()[name = string("hidden_states_567_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_567_groups_0 = const()[name = string("hidden_states_567_groups_0"), val = int32(1)]; + tensor hidden_states_567_cast_fp16 = conv(dilations = hidden_states_567_dilations_0, groups = hidden_states_567_groups_0, pad = hidden_states_567_pad_0, pad_type = hidden_states_567_pad_type_0, strides = hidden_states_567_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_475_cast_fp16)[name = string("hidden_states_567_cast_fp16")]; + tensor inputs_459_cast_fp16 = add(x = inputs_457_cast_fp16, y = hidden_states_567_cast_fp16)[name = string("inputs_459_cast_fp16")]; + int32 var_16764 = const()[name = string("op_16764"), val = int32(1)]; + bool key_caches_23_interleave_0 = const()[name = string("key_caches_23_interleave_0"), val = bool(false)]; + tensor key_caches_23_cast_fp16 = concat(axis = var_16764, interleave = key_caches_23_interleave_0, values = (key_203_cast_fp16, key_207_cast_fp16, key_211_cast_fp16, key_215_cast_fp16, key_219_cast_fp16))[name = string("key_caches_23_cast_fp16")]; + int32 var_16767 = const()[name = string("op_16767"), val = int32(1)]; + bool value_caches_23_interleave_0 = const()[name = string("value_caches_23_interleave_0"), val = bool(false)]; + tensor value_caches_23_cast_fp16 = concat(axis = var_16767, interleave = value_caches_23_interleave_0, values = (value_101_cast_fp16, value_103_cast_fp16, value_105_cast_fp16, value_107_cast_fp16, value_109_cast_fp16))[name = string("value_caches_23_cast_fp16")]; + tensor inputs_sq_459_cast_fp16 = mul(x = inputs_459_cast_fp16, y = inputs_459_cast_fp16)[name = string("inputs_sq_459_cast_fp16")]; + tensor variance_459_axes_0 = const()[name = string("variance_459_axes_0"), val = tensor([1])]; + bool variance_459_keep_dims_0 = const()[name = string("variance_459_keep_dims_0"), val = bool(true)]; + tensor variance_459_cast_fp16 = reduce_mean(axes = variance_459_axes_0, keep_dims = variance_459_keep_dims_0, x = inputs_sq_459_cast_fp16)[name = string("variance_459_cast_fp16")]; + fp16 var_16777_to_fp16 = const()[name = string("op_16777_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16778_cast_fp16 = add(x = variance_459_cast_fp16, y = var_16777_to_fp16)[name = string("op_16778_cast_fp16")]; + fp32 var_16779_epsilon_0 = const()[name = string("op_16779_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16779_cast_fp16 = rsqrt(epsilon = var_16779_epsilon_0, x = var_16778_cast_fp16)[name = string("op_16779_cast_fp16")]; + tensor hidden_states_569_cast_fp16 = mul(x = inputs_459_cast_fp16, y = var_16779_cast_fp16)[name = string("hidden_states_569_cast_fp16")]; + tensor input_477_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_569_cast_fp16)[name = string("input_477_cast_fp16")]; + string logits_37_pad_type_0 = const()[name = string("logits_37_pad_type_0"), val = string("valid")]; + tensor logits_37_strides_0 = const()[name = string("logits_37_strides_0"), val = tensor([1, 1])]; + tensor logits_37_pad_0 = const()[name = string("logits_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_37_dilations_0 = const()[name = string("logits_37_dilations_0"), val = tensor([1, 1])]; + int32 logits_37_groups_0 = const()[name = string("logits_37_groups_0"), val = int32(1)]; + tensor lm_heads_9_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(99686720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101783936))))[name = string("lm_heads_9_weight_to_fp16_palettized")]; + tensor logits_37_cast_fp16 = conv(dilations = logits_37_dilations_0, groups = logits_37_groups_0, pad = logits_37_pad_0, pad_type = logits_37_pad_type_0, strides = logits_37_strides_0, weight = lm_heads_9_weight_to_fp16_palettized, x = input_477_cast_fp16)[name = string("logits_37_cast_fp16")]; + tensor var_16797 = const()[name = string("op_16797"), val = tensor([1, 2048])]; + tensor logits_39_cast_fp16 = reshape(shape = var_16797, x = logits_37_cast_fp16)[name = string("logits_39_cast_fp16")]; + tensor scaled_logits_19_cast_fp16 = real_div(x = logits_39_cast_fp16, y = temperature)[name = string("scaled_logits_19_cast_fp16")]; + int32 var_16807 = const()[name = string("op_16807"), val = int32(100)]; + int32 top_values_19_axis_0 = const()[name = string("top_values_19_axis_0"), val = int32(1)]; + bool top_values_19_ascending_0 = const()[name = string("top_values_19_ascending_0"), val = bool(false)]; + bool top_values_19_sort_0 = const()[name = string("top_values_19_sort_0"), val = bool(true)]; + bool top_values_19_return_indices_0 = const()[name = string("top_values_19_return_indices_0"), val = bool(true)]; + string top_values_19_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_19_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_19_cast_fp16_cast_uint16_0, tensor top_values_19_cast_fp16_cast_uint16_1 = topk(ascending = top_values_19_ascending_0, axis = top_values_19_axis_0, k = var_16807, output_indices_dtype = top_values_19_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_19_return_indices_0, sort = top_values_19_sort_0, x = scaled_logits_19_cast_fp16)[name = string("top_values_19_cast_fp16_cast_uint16")]; + tensor var_16813_cast_fp16 = mul(x = top_values_19_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_16813_cast_fp16")]; + tensor var_16817_cast_fp16 = add(x = var_16813_cast_fp16, y = var_3001_cast_fp16)[name = string("op_16817_cast_fp16")]; + tensor reduce_min_9_axes_0 = const()[name = string("reduce_min_9_axes_0"), val = tensor([1])]; + bool reduce_min_9_keep_dims_0 = const()[name = string("reduce_min_9_keep_dims_0"), val = bool(true)]; + tensor reduce_min_9_cast_fp16 = reduce_min(axes = reduce_min_9_axes_0, keep_dims = reduce_min_9_keep_dims_0, x = var_16817_cast_fp16)[name = string("reduce_min_9_cast_fp16")]; + tensor var_16820_cast_fp16 = greater_equal(x = scaled_logits_19_cast_fp16, y = reduce_min_9_cast_fp16)[name = string("op_16820_cast_fp16")]; + fp16 var_16821_value_0_to_fp16 = const()[name = string("op_16821_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_16821_cast_fp16 = fill_like(ref_tensor = scaled_logits_19_cast_fp16, value = var_16821_value_0_to_fp16)[name = string("op_16821_cast_fp16")]; + tensor masked_logits_19_cast_fp16 = select(a = scaled_logits_19_cast_fp16, b = var_16821_cast_fp16, cond = var_16820_cast_fp16)[name = string("masked_logits_19_cast_fp16")]; + tensor var_16825_begin_0 = const()[name = string("op_16825_begin_0"), val = tensor([9, 0])]; + tensor var_16825_end_0 = const()[name = string("op_16825_end_0"), val = tensor([10, 2048])]; + tensor var_16825_end_mask_0 = const()[name = string("op_16825_end_mask_0"), val = tensor([false, true])]; + tensor var_16825_squeeze_mask_0 = const()[name = string("op_16825_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_16825_cast_fp16 = slice_by_index(begin = var_16825_begin_0, end = var_16825_end_0, end_mask = var_16825_end_mask_0, squeeze_mask = var_16825_squeeze_mask_0, x = gumbel)[name = string("op_16825_cast_fp16")]; + tensor var_16828 = const()[name = string("op_16828"), val = tensor([1, 2048])]; + tensor var_16829_cast_fp16 = reshape(shape = var_16828, x = var_16825_cast_fp16)[name = string("op_16829_cast_fp16")]; + tensor noisy_logits_19_cast_fp16 = add(x = masked_logits_19_cast_fp16, y = var_16829_cast_fp16)[name = string("noisy_logits_19_cast_fp16")]; + int32 code_19_axis_0 = const()[name = string("code_19_axis_0"), val = int32(1)]; + bool code_19_keep_dims_0 = const()[name = string("code_19_keep_dims_0"), val = bool(false)]; + string code_19_output_dtype_0 = const()[name = string("code_19_output_dtype_0"), val = string("int32")]; + tensor code_19_cast_fp16 = reduce_argmax(axis = code_19_axis_0, keep_dims = code_19_keep_dims_0, output_dtype = code_19_output_dtype_0, x = noisy_logits_19_cast_fp16)[name = string("code_19_cast_fp16")]; + int32 var_16840 = const()[name = string("op_16840"), val = int32(18432)]; + tensor input_479 = add(x = code_19_cast_fp16, y = var_16840)[name = string("input_479")]; + int32 code_embed_37_axis_0 = const()[name = string("code_embed_37_axis_0"), val = int32(0)]; + int32 code_embed_37_batch_dims_0 = const()[name = string("code_embed_37_batch_dims_0"), val = int32(0)]; + bool code_embed_37_validate_indices_0 = const()[name = string("code_embed_37_validate_indices_0"), val = bool(false)]; + string input_479_to_uint16_dtype_0 = const()[name = string("input_479_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_479_to_uint16 = cast(dtype = input_479_to_uint16_dtype_0, x = input_479)[name = string("cast_5")]; + tensor code_embed_37_cast_fp16_cast_uint16 = gather(axis = code_embed_37_axis_0, batch_dims = code_embed_37_batch_dims_0, indices = input_479_to_uint16, validate_indices = code_embed_37_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_37_cast_fp16_cast_uint16")]; + tensor var_16844 = const()[name = string("op_16844"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_39_cast_fp16 = reshape(shape = var_16844, x = code_embed_37_cast_fp16_cast_uint16)[name = string("code_embed_39_cast_fp16")]; + tensor embed_sum_21_cast_fp16 = add(x = embed_sum_19_cast_fp16, y = code_embed_39_cast_fp16)[name = string("embed_sum_21_cast_fp16")]; + string inputs_461_pad_type_0 = const()[name = string("inputs_461_pad_type_0"), val = string("valid")]; + tensor inputs_461_strides_0 = const()[name = string("inputs_461_strides_0"), val = tensor([1, 1])]; + tensor inputs_461_pad_0 = const()[name = string("inputs_461_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_461_dilations_0 = const()[name = string("inputs_461_dilations_0"), val = tensor([1, 1])]; + int32 inputs_461_groups_0 = const()[name = string("inputs_461_groups_0"), val = int32(1)]; + tensor inputs_461_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_461_dilations_0, groups = inputs_461_groups_0, pad = inputs_461_pad_0, pad_type = inputs_461_pad_type_0, strides = inputs_461_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_39_cast_fp16)[name = string("inputs_461_cast_fp16")]; + tensor obj_487_begin_0 = const()[name = string("obj_487_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_487_end_0 = const()[name = string("obj_487_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_487_end_mask_0 = const()[name = string("obj_487_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_487_cast_fp16 = slice_by_index(begin = obj_487_begin_0, end = obj_487_end_0, end_mask = obj_487_end_mask_0, x = key_caches_23_cast_fp16)[name = string("obj_487_cast_fp16")]; + tensor obj_489_begin_0 = const()[name = string("obj_489_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_489_end_0 = const()[name = string("obj_489_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_489_end_mask_0 = const()[name = string("obj_489_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_489_cast_fp16 = slice_by_index(begin = obj_489_begin_0, end = obj_489_end_0, end_mask = obj_489_end_mask_0, x = value_caches_23_cast_fp16)[name = string("obj_489_cast_fp16")]; + int32 var_16949 = const()[name = string("op_16949"), val = int32(3)]; + int32 var_16959 = const()[name = string("op_16959"), val = int32(-2)]; + tensor inputs_sq_461_cast_fp16 = mul(x = inputs_461_cast_fp16, y = inputs_461_cast_fp16)[name = string("inputs_sq_461_cast_fp16")]; + tensor variance_461_axes_0 = const()[name = string("variance_461_axes_0"), val = tensor([1])]; + bool variance_461_keep_dims_0 = const()[name = string("variance_461_keep_dims_0"), val = bool(true)]; + tensor variance_461_cast_fp16 = reduce_mean(axes = variance_461_axes_0, keep_dims = variance_461_keep_dims_0, x = inputs_sq_461_cast_fp16)[name = string("variance_461_cast_fp16")]; + fp16 var_16973_to_fp16 = const()[name = string("op_16973_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_16974_cast_fp16 = add(x = variance_461_cast_fp16, y = var_16973_to_fp16)[name = string("op_16974_cast_fp16")]; + fp32 var_16975_epsilon_0 = const()[name = string("op_16975_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_16975_cast_fp16 = rsqrt(epsilon = var_16975_epsilon_0, x = var_16974_cast_fp16)[name = string("op_16975_cast_fp16")]; + tensor hidden_states_571_cast_fp16 = mul(x = inputs_461_cast_fp16, y = var_16975_cast_fp16)[name = string("hidden_states_571_cast_fp16")]; + tensor obj_485_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_571_cast_fp16)[name = string("obj_485_cast_fp16")]; + string query_331_pad_type_0 = const()[name = string("query_331_pad_type_0"), val = string("valid")]; + tensor query_331_strides_0 = const()[name = string("query_331_strides_0"), val = tensor([1, 1])]; + tensor query_331_pad_0 = const()[name = string("query_331_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_331_dilations_0 = const()[name = string("query_331_dilations_0"), val = tensor([1, 1])]; + int32 query_331_groups_0 = const()[name = string("query_331_groups_0"), val = int32(1)]; + tensor query_331_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_331_dilations_0, groups = query_331_groups_0, pad = query_331_pad_0, pad_type = query_331_pad_type_0, strides = query_331_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_485_cast_fp16)[name = string("query_331_cast_fp16")]; + string current_key_221_pad_type_0 = const()[name = string("current_key_221_pad_type_0"), val = string("valid")]; + tensor current_key_221_strides_0 = const()[name = string("current_key_221_strides_0"), val = tensor([1, 1])]; + tensor current_key_221_pad_0 = const()[name = string("current_key_221_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_221_dilations_0 = const()[name = string("current_key_221_dilations_0"), val = tensor([1, 1])]; + int32 current_key_221_groups_0 = const()[name = string("current_key_221_groups_0"), val = int32(1)]; + tensor current_key_221_cast_fp16 = conv(dilations = current_key_221_dilations_0, groups = current_key_221_groups_0, pad = current_key_221_pad_0, pad_type = current_key_221_pad_type_0, strides = current_key_221_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_485_cast_fp16)[name = string("current_key_221_cast_fp16")]; + string current_value_111_pad_type_0 = const()[name = string("current_value_111_pad_type_0"), val = string("valid")]; + tensor current_value_111_strides_0 = const()[name = string("current_value_111_strides_0"), val = tensor([1, 1])]; + tensor current_value_111_pad_0 = const()[name = string("current_value_111_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_111_dilations_0 = const()[name = string("current_value_111_dilations_0"), val = tensor([1, 1])]; + int32 current_value_111_groups_0 = const()[name = string("current_value_111_groups_0"), val = int32(1)]; + tensor current_value_111_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_111_dilations_0, groups = current_value_111_groups_0, pad = current_value_111_pad_0, pad_type = current_value_111_pad_type_0, strides = current_value_111_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_485_cast_fp16)[name = string("current_value_111_cast_fp16")]; + tensor var_17012 = const()[name = string("op_17012"), val = tensor([16, 128, 1, 1])]; + tensor inputs_463_cast_fp16 = reshape(shape = var_17012, x = query_331_cast_fp16)[name = string("inputs_463_cast_fp16")]; + tensor inputs_sq_463_cast_fp16 = mul(x = inputs_463_cast_fp16, y = inputs_463_cast_fp16)[name = string("inputs_sq_463_cast_fp16")]; + tensor variance_463_axes_0 = const()[name = string("variance_463_axes_0"), val = tensor([1])]; + bool variance_463_keep_dims_0 = const()[name = string("variance_463_keep_dims_0"), val = bool(true)]; + tensor variance_463_cast_fp16 = reduce_mean(axes = variance_463_axes_0, keep_dims = variance_463_keep_dims_0, x = inputs_sq_463_cast_fp16)[name = string("variance_463_cast_fp16")]; + fp16 var_17018_to_fp16 = const()[name = string("op_17018_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17019_cast_fp16 = add(x = variance_463_cast_fp16, y = var_17018_to_fp16)[name = string("op_17019_cast_fp16")]; + fp32 var_17020_epsilon_0 = const()[name = string("op_17020_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17020_cast_fp16 = rsqrt(epsilon = var_17020_epsilon_0, x = var_17019_cast_fp16)[name = string("op_17020_cast_fp16")]; + tensor hidden_states_573_cast_fp16 = mul(x = inputs_463_cast_fp16, y = var_17020_cast_fp16)[name = string("hidden_states_573_cast_fp16")]; + tensor query_normed_111_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_573_cast_fp16)[name = string("query_normed_111_cast_fp16")]; + tensor var_17028 = const()[name = string("op_17028"), val = tensor([8, 128, 1, 1])]; + tensor inputs_465_cast_fp16 = reshape(shape = var_17028, x = current_key_221_cast_fp16)[name = string("inputs_465_cast_fp16")]; + tensor inputs_sq_465_cast_fp16 = mul(x = inputs_465_cast_fp16, y = inputs_465_cast_fp16)[name = string("inputs_sq_465_cast_fp16")]; + tensor variance_465_axes_0 = const()[name = string("variance_465_axes_0"), val = tensor([1])]; + bool variance_465_keep_dims_0 = const()[name = string("variance_465_keep_dims_0"), val = bool(true)]; + tensor variance_465_cast_fp16 = reduce_mean(axes = variance_465_axes_0, keep_dims = variance_465_keep_dims_0, x = inputs_sq_465_cast_fp16)[name = string("variance_465_cast_fp16")]; + fp16 var_17034_to_fp16 = const()[name = string("op_17034_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17035_cast_fp16 = add(x = variance_465_cast_fp16, y = var_17034_to_fp16)[name = string("op_17035_cast_fp16")]; + fp32 var_17036_epsilon_0 = const()[name = string("op_17036_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17036_cast_fp16 = rsqrt(epsilon = var_17036_epsilon_0, x = var_17035_cast_fp16)[name = string("op_17036_cast_fp16")]; + tensor hidden_states_575_cast_fp16 = mul(x = inputs_465_cast_fp16, y = var_17036_cast_fp16)[name = string("hidden_states_575_cast_fp16")]; + tensor current_key_normed_111_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_575_cast_fp16)[name = string("current_key_normed_111_cast_fp16")]; + tensor var_17054 = const()[name = string("op_17054"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_441_cast_fp16 = reshape(shape = var_17054, x = query_normed_111_cast_fp16)[name = string("mh_q_441_cast_fp16")]; + tensor var_17056 = const()[name = string("op_17056"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_441_cast_fp16 = reshape(shape = var_17056, x = current_key_normed_111_cast_fp16)[name = string("mh_k_441_cast_fp16")]; + tensor cos_111_to_fp16 = const()[name = string("cos_111_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175199808)))]; + tensor var_17060_cast_fp16 = mul(x = mh_q_441_cast_fp16, y = cos_111_to_fp16)[name = string("op_17060_cast_fp16")]; + tensor var_17065_begin_0 = const()[name = string("op_17065_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17065_end_0 = const()[name = string("op_17065_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_17065_end_mask_0 = const()[name = string("op_17065_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_17065_cast_fp16 = slice_by_index(begin = var_17065_begin_0, end = var_17065_end_0, end_mask = var_17065_end_mask_0, x = mh_q_441_cast_fp16)[name = string("op_17065_cast_fp16")]; + tensor var_17071_begin_0 = const()[name = string("op_17071_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_17071_end_0 = const()[name = string("op_17071_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_17071_end_mask_0 = const()[name = string("op_17071_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_17071_cast_fp16 = slice_by_index(begin = var_17071_begin_0, end = var_17071_end_0, end_mask = var_17071_end_mask_0, x = mh_q_441_cast_fp16)[name = string("op_17071_cast_fp16")]; + fp16 const_1125_promoted_to_fp16 = const()[name = string("const_1125_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17073_cast_fp16 = mul(x = var_17071_cast_fp16, y = const_1125_promoted_to_fp16)[name = string("op_17073_cast_fp16")]; + bool var_17075_interleave_0 = const()[name = string("op_17075_interleave_0"), val = bool(false)]; + tensor var_17075_cast_fp16 = concat(axis = var_16959, interleave = var_17075_interleave_0, values = (var_17073_cast_fp16, var_17065_cast_fp16))[name = string("op_17075_cast_fp16")]; + tensor sin_111_to_fp16 = const()[name = string("sin_111_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200128)))]; + tensor var_17076_cast_fp16 = mul(x = var_17075_cast_fp16, y = sin_111_to_fp16)[name = string("op_17076_cast_fp16")]; + tensor mh_q_443_cast_fp16 = add(x = var_17060_cast_fp16, y = var_17076_cast_fp16)[name = string("mh_q_443_cast_fp16")]; + tensor var_17078_cast_fp16 = mul(x = mh_k_441_cast_fp16, y = cos_111_to_fp16)[name = string("op_17078_cast_fp16")]; + tensor var_17083_begin_0 = const()[name = string("op_17083_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17083_end_0 = const()[name = string("op_17083_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_17083_end_mask_0 = const()[name = string("op_17083_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_17083_cast_fp16 = slice_by_index(begin = var_17083_begin_0, end = var_17083_end_0, end_mask = var_17083_end_mask_0, x = mh_k_441_cast_fp16)[name = string("op_17083_cast_fp16")]; + tensor var_17089_begin_0 = const()[name = string("op_17089_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_17089_end_0 = const()[name = string("op_17089_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_17089_end_mask_0 = const()[name = string("op_17089_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_17089_cast_fp16 = slice_by_index(begin = var_17089_begin_0, end = var_17089_end_0, end_mask = var_17089_end_mask_0, x = mh_k_441_cast_fp16)[name = string("op_17089_cast_fp16")]; + fp16 const_1128_promoted_to_fp16 = const()[name = string("const_1128_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17091_cast_fp16 = mul(x = var_17089_cast_fp16, y = const_1128_promoted_to_fp16)[name = string("op_17091_cast_fp16")]; + bool var_17093_interleave_0 = const()[name = string("op_17093_interleave_0"), val = bool(false)]; + tensor var_17093_cast_fp16 = concat(axis = var_16959, interleave = var_17093_interleave_0, values = (var_17091_cast_fp16, var_17083_cast_fp16))[name = string("op_17093_cast_fp16")]; + tensor var_17094_cast_fp16 = mul(x = var_17093_cast_fp16, y = sin_111_to_fp16)[name = string("op_17094_cast_fp16")]; + tensor mh_k_443_cast_fp16 = add(x = var_17078_cast_fp16, y = var_17094_cast_fp16)[name = string("mh_k_443_cast_fp16")]; + tensor var_17098 = const()[name = string("op_17098"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_223_cast_fp16 = reshape(shape = var_17098, x = mh_k_443_cast_fp16)[name = string("current_key_223_cast_fp16")]; + tensor var_17104_to_fp16 = const()[name = string("op_17104_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200448)))]; + tensor var_17105_cast_fp16 = mul(x = obj_487_cast_fp16, y = var_17104_to_fp16)[name = string("op_17105_cast_fp16")]; + tensor var_17102_to_fp16 = const()[name = string("op_17102_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200576)))]; + tensor var_17106_cast_fp16 = mul(x = current_key_223_cast_fp16, y = var_17102_to_fp16)[name = string("op_17106_cast_fp16")]; + tensor key_223_cast_fp16 = add(x = var_17105_cast_fp16, y = var_17106_cast_fp16)[name = string("key_223_cast_fp16")]; + tensor var_17108_to_fp16 = const()[name = string("op_17108_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200448)))]; + tensor var_17109_cast_fp16 = mul(x = obj_489_cast_fp16, y = var_17108_to_fp16)[name = string("op_17109_cast_fp16")]; + tensor var_17110_cast_fp16 = mul(x = current_value_111_cast_fp16, y = var_17102_to_fp16)[name = string("op_17110_cast_fp16")]; + tensor value_111_cast_fp16 = add(x = var_17109_cast_fp16, y = var_17110_cast_fp16)[name = string("value_111_cast_fp16")]; + fp16 var_17117_to_fp16 = const()[name = string("op_17117_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_447_cast_fp16 = mul(x = mh_q_443_cast_fp16, y = var_17117_to_fp16)[name = string("mh_q_447_cast_fp16")]; + tensor var_17119 = const()[name = string("op_17119"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_445_cast_fp16 = reshape(shape = var_17119, x = key_223_cast_fp16)[name = string("mh_k_445_cast_fp16")]; + tensor var_17121 = const()[name = string("op_17121"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_221_cast_fp16 = reshape(shape = var_17121, x = value_111_cast_fp16)[name = string("mh_v_221_cast_fp16")]; + tensor transpose_220_perm_0 = const()[name = string("transpose_220_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_110_reps_0 = const()[name = string("tile_110_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_220_cast_fp16 = transpose(perm = transpose_220_perm_0, x = mh_k_445_cast_fp16)[name = string("transpose_149")]; + tensor tile_110_cast_fp16 = tile(reps = tile_110_reps_0, x = transpose_220_cast_fp16)[name = string("tile_110_cast_fp16")]; + tensor concat_278 = const()[name = string("concat_278"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_220_cast_fp16 = reshape(shape = concat_278, x = tile_110_cast_fp16)[name = string("reshape_220_cast_fp16")]; + tensor transpose_221_perm_0 = const()[name = string("transpose_221_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_279 = const()[name = string("concat_279"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_221_cast_fp16 = transpose(perm = transpose_221_perm_0, x = reshape_220_cast_fp16)[name = string("transpose_148")]; + tensor reshape_221_cast_fp16 = reshape(shape = concat_279, x = transpose_221_cast_fp16)[name = string("reshape_221_cast_fp16")]; + tensor transpose_222_perm_0 = const()[name = string("transpose_222_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_111_reps_0 = const()[name = string("tile_111_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_222_cast_fp16 = transpose(perm = transpose_222_perm_0, x = mh_v_221_cast_fp16)[name = string("transpose_147")]; + tensor tile_111_cast_fp16 = tile(reps = tile_111_reps_0, x = transpose_222_cast_fp16)[name = string("tile_111_cast_fp16")]; + tensor concat_280 = const()[name = string("concat_280"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_222_cast_fp16 = reshape(shape = concat_280, x = tile_111_cast_fp16)[name = string("reshape_222_cast_fp16")]; + tensor transpose_223_perm_0 = const()[name = string("transpose_223_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_281 = const()[name = string("concat_281"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_223_cast_fp16 = transpose(perm = transpose_223_perm_0, x = reshape_222_cast_fp16)[name = string("transpose_146")]; + tensor reshape_223_cast_fp16 = reshape(shape = concat_281, x = transpose_223_cast_fp16)[name = string("reshape_223_cast_fp16")]; + tensor transpose_537_perm_0 = const()[name = string("transpose_537_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_331_transpose_x_1 = const()[name = string("mh_w_331_transpose_x_1"), val = bool(true)]; + bool mh_w_331_transpose_y_1 = const()[name = string("mh_w_331_transpose_y_1"), val = bool(false)]; + tensor transpose_537_cast_fp16 = transpose(perm = transpose_537_perm_0, x = reshape_221_cast_fp16)[name = string("transpose_145")]; + tensor mh_w_331_cast_fp16 = matmul(transpose_x = mh_w_331_transpose_x_1, transpose_y = mh_w_331_transpose_y_1, x = mh_q_447_cast_fp16, y = transpose_537_cast_fp16)[name = string("mh_w_331_cast_fp16")]; + tensor var_17129_to_fp16 = const()[name = string("op_17129_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200704)))]; + tensor mh_w_333_cast_fp16 = add(x = mh_w_331_cast_fp16, y = var_17129_to_fp16)[name = string("mh_w_333_cast_fp16")]; + tensor mh_w_335_cast_fp16 = softmax(axis = var_16949, x = mh_w_333_cast_fp16)[name = string("mh_w_335_cast_fp16")]; + tensor transpose_538_perm_0 = const()[name = string("transpose_538_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_111_transpose_x_1 = const()[name = string("attn_111_transpose_x_1"), val = bool(false)]; + bool attn_111_transpose_y_1 = const()[name = string("attn_111_transpose_y_1"), val = bool(true)]; + tensor transpose_538_cast_fp16 = transpose(perm = transpose_538_perm_0, x = reshape_223_cast_fp16)[name = string("transpose_144")]; + tensor attn_111_cast_fp16 = matmul(transpose_x = attn_111_transpose_x_1, transpose_y = attn_111_transpose_y_1, x = transpose_538_cast_fp16, y = mh_w_335_cast_fp16)[name = string("attn_111_cast_fp16")]; + tensor var_17135 = const()[name = string("op_17135"), val = tensor([1, 2048, 1, 1])]; + tensor input_481_cast_fp16 = reshape(shape = var_17135, x = attn_111_cast_fp16)[name = string("input_481_cast_fp16")]; + string obj_495_pad_type_0 = const()[name = string("obj_495_pad_type_0"), val = string("valid")]; + tensor obj_495_strides_0 = const()[name = string("obj_495_strides_0"), val = tensor([1, 1])]; + tensor obj_495_pad_0 = const()[name = string("obj_495_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_495_dilations_0 = const()[name = string("obj_495_dilations_0"), val = tensor([1, 1])]; + int32 obj_495_groups_0 = const()[name = string("obj_495_groups_0"), val = int32(1)]; + tensor obj_495_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_495_dilations_0, groups = obj_495_groups_0, pad = obj_495_pad_0, pad_type = obj_495_pad_type_0, strides = obj_495_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_481_cast_fp16)[name = string("obj_495_cast_fp16")]; + tensor inputs_467_cast_fp16 = add(x = inputs_461_cast_fp16, y = obj_495_cast_fp16)[name = string("inputs_467_cast_fp16")]; + tensor inputs_sq_467_cast_fp16 = mul(x = inputs_467_cast_fp16, y = inputs_467_cast_fp16)[name = string("inputs_sq_467_cast_fp16")]; + tensor variance_467_axes_0 = const()[name = string("variance_467_axes_0"), val = tensor([1])]; + bool variance_467_keep_dims_0 = const()[name = string("variance_467_keep_dims_0"), val = bool(true)]; + tensor variance_467_cast_fp16 = reduce_mean(axes = variance_467_axes_0, keep_dims = variance_467_keep_dims_0, x = inputs_sq_467_cast_fp16)[name = string("variance_467_cast_fp16")]; + fp16 var_17153_to_fp16 = const()[name = string("op_17153_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17154_cast_fp16 = add(x = variance_467_cast_fp16, y = var_17153_to_fp16)[name = string("op_17154_cast_fp16")]; + fp32 var_17155_epsilon_0 = const()[name = string("op_17155_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17155_cast_fp16 = rsqrt(epsilon = var_17155_epsilon_0, x = var_17154_cast_fp16)[name = string("op_17155_cast_fp16")]; + tensor hidden_states_577_cast_fp16 = mul(x = inputs_467_cast_fp16, y = var_17155_cast_fp16)[name = string("hidden_states_577_cast_fp16")]; + tensor input_483_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_577_cast_fp16)[name = string("input_483_cast_fp16")]; + string input_485_pad_type_0 = const()[name = string("input_485_pad_type_0"), val = string("valid")]; + tensor input_485_strides_0 = const()[name = string("input_485_strides_0"), val = tensor([1, 1])]; + tensor input_485_pad_0 = const()[name = string("input_485_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_485_dilations_0 = const()[name = string("input_485_dilations_0"), val = tensor([1, 1])]; + int32 input_485_groups_0 = const()[name = string("input_485_groups_0"), val = int32(1)]; + tensor input_485_cast_fp16 = conv(dilations = input_485_dilations_0, groups = input_485_groups_0, pad = input_485_pad_0, pad_type = input_485_pad_type_0, strides = input_485_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_483_cast_fp16)[name = string("input_485_cast_fp16")]; + tensor var_17169_cast_fp16 = silu(x = input_485_cast_fp16)[name = string("op_17169_cast_fp16")]; + string var_17175_pad_type_0 = const()[name = string("op_17175_pad_type_0"), val = string("valid")]; + tensor var_17175_strides_0 = const()[name = string("op_17175_strides_0"), val = tensor([1, 1])]; + tensor var_17175_pad_0 = const()[name = string("op_17175_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17175_dilations_0 = const()[name = string("op_17175_dilations_0"), val = tensor([1, 1])]; + int32 var_17175_groups_0 = const()[name = string("op_17175_groups_0"), val = int32(1)]; + tensor var_17175_cast_fp16 = conv(dilations = var_17175_dilations_0, groups = var_17175_groups_0, pad = var_17175_pad_0, pad_type = var_17175_pad_type_0, strides = var_17175_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_483_cast_fp16)[name = string("op_17175_cast_fp16")]; + tensor input_487_cast_fp16 = mul(x = var_17169_cast_fp16, y = var_17175_cast_fp16)[name = string("input_487_cast_fp16")]; + string hidden_states_579_pad_type_0 = const()[name = string("hidden_states_579_pad_type_0"), val = string("valid")]; + tensor hidden_states_579_strides_0 = const()[name = string("hidden_states_579_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_579_pad_0 = const()[name = string("hidden_states_579_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_579_dilations_0 = const()[name = string("hidden_states_579_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_579_groups_0 = const()[name = string("hidden_states_579_groups_0"), val = int32(1)]; + tensor hidden_states_579_cast_fp16 = conv(dilations = hidden_states_579_dilations_0, groups = hidden_states_579_groups_0, pad = hidden_states_579_pad_0, pad_type = hidden_states_579_pad_type_0, strides = hidden_states_579_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_487_cast_fp16)[name = string("hidden_states_579_cast_fp16")]; + tensor inputs_469_cast_fp16 = add(x = inputs_467_cast_fp16, y = hidden_states_579_cast_fp16)[name = string("inputs_469_cast_fp16")]; + tensor obj_499_begin_0 = const()[name = string("obj_499_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_499_end_0 = const()[name = string("obj_499_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_499_end_mask_0 = const()[name = string("obj_499_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_499_cast_fp16 = slice_by_index(begin = obj_499_begin_0, end = obj_499_end_0, end_mask = obj_499_end_mask_0, x = key_caches_23_cast_fp16)[name = string("obj_499_cast_fp16")]; + tensor obj_501_begin_0 = const()[name = string("obj_501_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_501_end_0 = const()[name = string("obj_501_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_501_end_mask_0 = const()[name = string("obj_501_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_501_cast_fp16 = slice_by_index(begin = obj_501_begin_0, end = obj_501_end_0, end_mask = obj_501_end_mask_0, x = value_caches_23_cast_fp16)[name = string("obj_501_cast_fp16")]; + int32 var_17223 = const()[name = string("op_17223"), val = int32(3)]; + int32 var_17233 = const()[name = string("op_17233"), val = int32(-2)]; + tensor inputs_sq_469_cast_fp16 = mul(x = inputs_469_cast_fp16, y = inputs_469_cast_fp16)[name = string("inputs_sq_469_cast_fp16")]; + tensor variance_469_axes_0 = const()[name = string("variance_469_axes_0"), val = tensor([1])]; + bool variance_469_keep_dims_0 = const()[name = string("variance_469_keep_dims_0"), val = bool(true)]; + tensor variance_469_cast_fp16 = reduce_mean(axes = variance_469_axes_0, keep_dims = variance_469_keep_dims_0, x = inputs_sq_469_cast_fp16)[name = string("variance_469_cast_fp16")]; + fp16 var_17247_to_fp16 = const()[name = string("op_17247_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17248_cast_fp16 = add(x = variance_469_cast_fp16, y = var_17247_to_fp16)[name = string("op_17248_cast_fp16")]; + fp32 var_17249_epsilon_0 = const()[name = string("op_17249_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17249_cast_fp16 = rsqrt(epsilon = var_17249_epsilon_0, x = var_17248_cast_fp16)[name = string("op_17249_cast_fp16")]; + tensor hidden_states_581_cast_fp16 = mul(x = inputs_469_cast_fp16, y = var_17249_cast_fp16)[name = string("hidden_states_581_cast_fp16")]; + tensor obj_497_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_581_cast_fp16)[name = string("obj_497_cast_fp16")]; + string query_337_pad_type_0 = const()[name = string("query_337_pad_type_0"), val = string("valid")]; + tensor query_337_strides_0 = const()[name = string("query_337_strides_0"), val = tensor([1, 1])]; + tensor query_337_pad_0 = const()[name = string("query_337_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_337_dilations_0 = const()[name = string("query_337_dilations_0"), val = tensor([1, 1])]; + int32 query_337_groups_0 = const()[name = string("query_337_groups_0"), val = int32(1)]; + tensor query_337_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_337_dilations_0, groups = query_337_groups_0, pad = query_337_pad_0, pad_type = query_337_pad_type_0, strides = query_337_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_497_cast_fp16)[name = string("query_337_cast_fp16")]; + string current_key_225_pad_type_0 = const()[name = string("current_key_225_pad_type_0"), val = string("valid")]; + tensor current_key_225_strides_0 = const()[name = string("current_key_225_strides_0"), val = tensor([1, 1])]; + tensor current_key_225_pad_0 = const()[name = string("current_key_225_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_225_dilations_0 = const()[name = string("current_key_225_dilations_0"), val = tensor([1, 1])]; + int32 current_key_225_groups_0 = const()[name = string("current_key_225_groups_0"), val = int32(1)]; + tensor current_key_225_cast_fp16 = conv(dilations = current_key_225_dilations_0, groups = current_key_225_groups_0, pad = current_key_225_pad_0, pad_type = current_key_225_pad_type_0, strides = current_key_225_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_497_cast_fp16)[name = string("current_key_225_cast_fp16")]; + string current_value_113_pad_type_0 = const()[name = string("current_value_113_pad_type_0"), val = string("valid")]; + tensor current_value_113_strides_0 = const()[name = string("current_value_113_strides_0"), val = tensor([1, 1])]; + tensor current_value_113_pad_0 = const()[name = string("current_value_113_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_113_dilations_0 = const()[name = string("current_value_113_dilations_0"), val = tensor([1, 1])]; + int32 current_value_113_groups_0 = const()[name = string("current_value_113_groups_0"), val = int32(1)]; + tensor current_value_113_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_113_dilations_0, groups = current_value_113_groups_0, pad = current_value_113_pad_0, pad_type = current_value_113_pad_type_0, strides = current_value_113_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_497_cast_fp16)[name = string("current_value_113_cast_fp16")]; + tensor var_17286 = const()[name = string("op_17286"), val = tensor([16, 128, 1, 1])]; + tensor inputs_471_cast_fp16 = reshape(shape = var_17286, x = query_337_cast_fp16)[name = string("inputs_471_cast_fp16")]; + tensor inputs_sq_471_cast_fp16 = mul(x = inputs_471_cast_fp16, y = inputs_471_cast_fp16)[name = string("inputs_sq_471_cast_fp16")]; + tensor variance_471_axes_0 = const()[name = string("variance_471_axes_0"), val = tensor([1])]; + bool variance_471_keep_dims_0 = const()[name = string("variance_471_keep_dims_0"), val = bool(true)]; + tensor variance_471_cast_fp16 = reduce_mean(axes = variance_471_axes_0, keep_dims = variance_471_keep_dims_0, x = inputs_sq_471_cast_fp16)[name = string("variance_471_cast_fp16")]; + fp16 var_17292_to_fp16 = const()[name = string("op_17292_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17293_cast_fp16 = add(x = variance_471_cast_fp16, y = var_17292_to_fp16)[name = string("op_17293_cast_fp16")]; + fp32 var_17294_epsilon_0 = const()[name = string("op_17294_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17294_cast_fp16 = rsqrt(epsilon = var_17294_epsilon_0, x = var_17293_cast_fp16)[name = string("op_17294_cast_fp16")]; + tensor hidden_states_583_cast_fp16 = mul(x = inputs_471_cast_fp16, y = var_17294_cast_fp16)[name = string("hidden_states_583_cast_fp16")]; + tensor query_normed_113_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_583_cast_fp16)[name = string("query_normed_113_cast_fp16")]; + tensor var_17302 = const()[name = string("op_17302"), val = tensor([8, 128, 1, 1])]; + tensor inputs_473_cast_fp16 = reshape(shape = var_17302, x = current_key_225_cast_fp16)[name = string("inputs_473_cast_fp16")]; + tensor inputs_sq_473_cast_fp16 = mul(x = inputs_473_cast_fp16, y = inputs_473_cast_fp16)[name = string("inputs_sq_473_cast_fp16")]; + tensor variance_473_axes_0 = const()[name = string("variance_473_axes_0"), val = tensor([1])]; + bool variance_473_keep_dims_0 = const()[name = string("variance_473_keep_dims_0"), val = bool(true)]; + tensor variance_473_cast_fp16 = reduce_mean(axes = variance_473_axes_0, keep_dims = variance_473_keep_dims_0, x = inputs_sq_473_cast_fp16)[name = string("variance_473_cast_fp16")]; + fp16 var_17308_to_fp16 = const()[name = string("op_17308_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17309_cast_fp16 = add(x = variance_473_cast_fp16, y = var_17308_to_fp16)[name = string("op_17309_cast_fp16")]; + fp32 var_17310_epsilon_0 = const()[name = string("op_17310_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17310_cast_fp16 = rsqrt(epsilon = var_17310_epsilon_0, x = var_17309_cast_fp16)[name = string("op_17310_cast_fp16")]; + tensor hidden_states_585_cast_fp16 = mul(x = inputs_473_cast_fp16, y = var_17310_cast_fp16)[name = string("hidden_states_585_cast_fp16")]; + tensor current_key_normed_113_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_585_cast_fp16)[name = string("current_key_normed_113_cast_fp16")]; + tensor var_17328 = const()[name = string("op_17328"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_449_cast_fp16 = reshape(shape = var_17328, x = query_normed_113_cast_fp16)[name = string("mh_q_449_cast_fp16")]; + tensor var_17330 = const()[name = string("op_17330"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_449_cast_fp16 = reshape(shape = var_17330, x = current_key_normed_113_cast_fp16)[name = string("mh_k_449_cast_fp16")]; + tensor var_17334_cast_fp16 = mul(x = mh_q_449_cast_fp16, y = cos_111_to_fp16)[name = string("op_17334_cast_fp16")]; + tensor var_17339_begin_0 = const()[name = string("op_17339_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17339_end_0 = const()[name = string("op_17339_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_17339_end_mask_0 = const()[name = string("op_17339_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_17339_cast_fp16 = slice_by_index(begin = var_17339_begin_0, end = var_17339_end_0, end_mask = var_17339_end_mask_0, x = mh_q_449_cast_fp16)[name = string("op_17339_cast_fp16")]; + tensor var_17345_begin_0 = const()[name = string("op_17345_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_17345_end_0 = const()[name = string("op_17345_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_17345_end_mask_0 = const()[name = string("op_17345_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_17345_cast_fp16 = slice_by_index(begin = var_17345_begin_0, end = var_17345_end_0, end_mask = var_17345_end_mask_0, x = mh_q_449_cast_fp16)[name = string("op_17345_cast_fp16")]; + fp16 const_1145_promoted_to_fp16 = const()[name = string("const_1145_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17347_cast_fp16 = mul(x = var_17345_cast_fp16, y = const_1145_promoted_to_fp16)[name = string("op_17347_cast_fp16")]; + bool var_17349_interleave_0 = const()[name = string("op_17349_interleave_0"), val = bool(false)]; + tensor var_17349_cast_fp16 = concat(axis = var_17233, interleave = var_17349_interleave_0, values = (var_17347_cast_fp16, var_17339_cast_fp16))[name = string("op_17349_cast_fp16")]; + tensor var_17350_cast_fp16 = mul(x = var_17349_cast_fp16, y = sin_111_to_fp16)[name = string("op_17350_cast_fp16")]; + tensor mh_q_451_cast_fp16 = add(x = var_17334_cast_fp16, y = var_17350_cast_fp16)[name = string("mh_q_451_cast_fp16")]; + tensor var_17352_cast_fp16 = mul(x = mh_k_449_cast_fp16, y = cos_111_to_fp16)[name = string("op_17352_cast_fp16")]; + tensor var_17357_begin_0 = const()[name = string("op_17357_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17357_end_0 = const()[name = string("op_17357_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_17357_end_mask_0 = const()[name = string("op_17357_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_17357_cast_fp16 = slice_by_index(begin = var_17357_begin_0, end = var_17357_end_0, end_mask = var_17357_end_mask_0, x = mh_k_449_cast_fp16)[name = string("op_17357_cast_fp16")]; + tensor var_17363_begin_0 = const()[name = string("op_17363_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_17363_end_0 = const()[name = string("op_17363_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_17363_end_mask_0 = const()[name = string("op_17363_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_17363_cast_fp16 = slice_by_index(begin = var_17363_begin_0, end = var_17363_end_0, end_mask = var_17363_end_mask_0, x = mh_k_449_cast_fp16)[name = string("op_17363_cast_fp16")]; + fp16 const_1148_promoted_to_fp16 = const()[name = string("const_1148_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17365_cast_fp16 = mul(x = var_17363_cast_fp16, y = const_1148_promoted_to_fp16)[name = string("op_17365_cast_fp16")]; + bool var_17367_interleave_0 = const()[name = string("op_17367_interleave_0"), val = bool(false)]; + tensor var_17367_cast_fp16 = concat(axis = var_17233, interleave = var_17367_interleave_0, values = (var_17365_cast_fp16, var_17357_cast_fp16))[name = string("op_17367_cast_fp16")]; + tensor var_17368_cast_fp16 = mul(x = var_17367_cast_fp16, y = sin_111_to_fp16)[name = string("op_17368_cast_fp16")]; + tensor mh_k_451_cast_fp16 = add(x = var_17352_cast_fp16, y = var_17368_cast_fp16)[name = string("mh_k_451_cast_fp16")]; + tensor var_17372 = const()[name = string("op_17372"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_227_cast_fp16 = reshape(shape = var_17372, x = mh_k_451_cast_fp16)[name = string("current_key_227_cast_fp16")]; + tensor var_17378_to_fp16 = const()[name = string("op_17378_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200448)))]; + tensor var_17379_cast_fp16 = mul(x = obj_499_cast_fp16, y = var_17378_to_fp16)[name = string("op_17379_cast_fp16")]; + tensor var_17376_to_fp16 = const()[name = string("op_17376_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200576)))]; + tensor var_17380_cast_fp16 = mul(x = current_key_227_cast_fp16, y = var_17376_to_fp16)[name = string("op_17380_cast_fp16")]; + tensor key_227_cast_fp16 = add(x = var_17379_cast_fp16, y = var_17380_cast_fp16)[name = string("key_227_cast_fp16")]; + tensor var_17382_to_fp16 = const()[name = string("op_17382_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200448)))]; + tensor var_17383_cast_fp16 = mul(x = obj_501_cast_fp16, y = var_17382_to_fp16)[name = string("op_17383_cast_fp16")]; + tensor var_17384_cast_fp16 = mul(x = current_value_113_cast_fp16, y = var_17376_to_fp16)[name = string("op_17384_cast_fp16")]; + tensor value_113_cast_fp16 = add(x = var_17383_cast_fp16, y = var_17384_cast_fp16)[name = string("value_113_cast_fp16")]; + fp16 var_17391_to_fp16 = const()[name = string("op_17391_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_455_cast_fp16 = mul(x = mh_q_451_cast_fp16, y = var_17391_to_fp16)[name = string("mh_q_455_cast_fp16")]; + tensor var_17393 = const()[name = string("op_17393"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_453_cast_fp16 = reshape(shape = var_17393, x = key_227_cast_fp16)[name = string("mh_k_453_cast_fp16")]; + tensor var_17395 = const()[name = string("op_17395"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_225_cast_fp16 = reshape(shape = var_17395, x = value_113_cast_fp16)[name = string("mh_v_225_cast_fp16")]; + tensor transpose_224_perm_0 = const()[name = string("transpose_224_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_112_reps_0 = const()[name = string("tile_112_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_224_cast_fp16 = transpose(perm = transpose_224_perm_0, x = mh_k_453_cast_fp16)[name = string("transpose_143")]; + tensor tile_112_cast_fp16 = tile(reps = tile_112_reps_0, x = transpose_224_cast_fp16)[name = string("tile_112_cast_fp16")]; + tensor concat_282 = const()[name = string("concat_282"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_224_cast_fp16 = reshape(shape = concat_282, x = tile_112_cast_fp16)[name = string("reshape_224_cast_fp16")]; + tensor transpose_225_perm_0 = const()[name = string("transpose_225_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_283 = const()[name = string("concat_283"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_225_cast_fp16 = transpose(perm = transpose_225_perm_0, x = reshape_224_cast_fp16)[name = string("transpose_142")]; + tensor reshape_225_cast_fp16 = reshape(shape = concat_283, x = transpose_225_cast_fp16)[name = string("reshape_225_cast_fp16")]; + tensor transpose_226_perm_0 = const()[name = string("transpose_226_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_113_reps_0 = const()[name = string("tile_113_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_226_cast_fp16 = transpose(perm = transpose_226_perm_0, x = mh_v_225_cast_fp16)[name = string("transpose_141")]; + tensor tile_113_cast_fp16 = tile(reps = tile_113_reps_0, x = transpose_226_cast_fp16)[name = string("tile_113_cast_fp16")]; + tensor concat_284 = const()[name = string("concat_284"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_226_cast_fp16 = reshape(shape = concat_284, x = tile_113_cast_fp16)[name = string("reshape_226_cast_fp16")]; + tensor transpose_227_perm_0 = const()[name = string("transpose_227_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_285 = const()[name = string("concat_285"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_227_cast_fp16 = transpose(perm = transpose_227_perm_0, x = reshape_226_cast_fp16)[name = string("transpose_140")]; + tensor reshape_227_cast_fp16 = reshape(shape = concat_285, x = transpose_227_cast_fp16)[name = string("reshape_227_cast_fp16")]; + tensor transpose_541_perm_0 = const()[name = string("transpose_541_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_337_transpose_x_1 = const()[name = string("mh_w_337_transpose_x_1"), val = bool(true)]; + bool mh_w_337_transpose_y_1 = const()[name = string("mh_w_337_transpose_y_1"), val = bool(false)]; + tensor transpose_541_cast_fp16 = transpose(perm = transpose_541_perm_0, x = reshape_225_cast_fp16)[name = string("transpose_139")]; + tensor mh_w_337_cast_fp16 = matmul(transpose_x = mh_w_337_transpose_x_1, transpose_y = mh_w_337_transpose_y_1, x = mh_q_455_cast_fp16, y = transpose_541_cast_fp16)[name = string("mh_w_337_cast_fp16")]; + tensor var_17403_to_fp16 = const()[name = string("op_17403_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200704)))]; + tensor mh_w_339_cast_fp16 = add(x = mh_w_337_cast_fp16, y = var_17403_to_fp16)[name = string("mh_w_339_cast_fp16")]; + tensor mh_w_341_cast_fp16 = softmax(axis = var_17223, x = mh_w_339_cast_fp16)[name = string("mh_w_341_cast_fp16")]; + tensor transpose_542_perm_0 = const()[name = string("transpose_542_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_113_transpose_x_1 = const()[name = string("attn_113_transpose_x_1"), val = bool(false)]; + bool attn_113_transpose_y_1 = const()[name = string("attn_113_transpose_y_1"), val = bool(true)]; + tensor transpose_542_cast_fp16 = transpose(perm = transpose_542_perm_0, x = reshape_227_cast_fp16)[name = string("transpose_138")]; + tensor attn_113_cast_fp16 = matmul(transpose_x = attn_113_transpose_x_1, transpose_y = attn_113_transpose_y_1, x = transpose_542_cast_fp16, y = mh_w_341_cast_fp16)[name = string("attn_113_cast_fp16")]; + tensor var_17409 = const()[name = string("op_17409"), val = tensor([1, 2048, 1, 1])]; + tensor input_489_cast_fp16 = reshape(shape = var_17409, x = attn_113_cast_fp16)[name = string("input_489_cast_fp16")]; + string obj_503_pad_type_0 = const()[name = string("obj_503_pad_type_0"), val = string("valid")]; + tensor obj_503_strides_0 = const()[name = string("obj_503_strides_0"), val = tensor([1, 1])]; + tensor obj_503_pad_0 = const()[name = string("obj_503_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_503_dilations_0 = const()[name = string("obj_503_dilations_0"), val = tensor([1, 1])]; + int32 obj_503_groups_0 = const()[name = string("obj_503_groups_0"), val = int32(1)]; + tensor obj_503_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_503_dilations_0, groups = obj_503_groups_0, pad = obj_503_pad_0, pad_type = obj_503_pad_type_0, strides = obj_503_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_489_cast_fp16)[name = string("obj_503_cast_fp16")]; + tensor inputs_475_cast_fp16 = add(x = inputs_469_cast_fp16, y = obj_503_cast_fp16)[name = string("inputs_475_cast_fp16")]; + tensor inputs_sq_475_cast_fp16 = mul(x = inputs_475_cast_fp16, y = inputs_475_cast_fp16)[name = string("inputs_sq_475_cast_fp16")]; + tensor variance_475_axes_0 = const()[name = string("variance_475_axes_0"), val = tensor([1])]; + bool variance_475_keep_dims_0 = const()[name = string("variance_475_keep_dims_0"), val = bool(true)]; + tensor variance_475_cast_fp16 = reduce_mean(axes = variance_475_axes_0, keep_dims = variance_475_keep_dims_0, x = inputs_sq_475_cast_fp16)[name = string("variance_475_cast_fp16")]; + fp16 var_17427_to_fp16 = const()[name = string("op_17427_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17428_cast_fp16 = add(x = variance_475_cast_fp16, y = var_17427_to_fp16)[name = string("op_17428_cast_fp16")]; + fp32 var_17429_epsilon_0 = const()[name = string("op_17429_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17429_cast_fp16 = rsqrt(epsilon = var_17429_epsilon_0, x = var_17428_cast_fp16)[name = string("op_17429_cast_fp16")]; + tensor hidden_states_587_cast_fp16 = mul(x = inputs_475_cast_fp16, y = var_17429_cast_fp16)[name = string("hidden_states_587_cast_fp16")]; + tensor input_491_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_587_cast_fp16)[name = string("input_491_cast_fp16")]; + string input_493_pad_type_0 = const()[name = string("input_493_pad_type_0"), val = string("valid")]; + tensor input_493_strides_0 = const()[name = string("input_493_strides_0"), val = tensor([1, 1])]; + tensor input_493_pad_0 = const()[name = string("input_493_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_493_dilations_0 = const()[name = string("input_493_dilations_0"), val = tensor([1, 1])]; + int32 input_493_groups_0 = const()[name = string("input_493_groups_0"), val = int32(1)]; + tensor input_493_cast_fp16 = conv(dilations = input_493_dilations_0, groups = input_493_groups_0, pad = input_493_pad_0, pad_type = input_493_pad_type_0, strides = input_493_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_491_cast_fp16)[name = string("input_493_cast_fp16")]; + tensor var_17443_cast_fp16 = silu(x = input_493_cast_fp16)[name = string("op_17443_cast_fp16")]; + string var_17449_pad_type_0 = const()[name = string("op_17449_pad_type_0"), val = string("valid")]; + tensor var_17449_strides_0 = const()[name = string("op_17449_strides_0"), val = tensor([1, 1])]; + tensor var_17449_pad_0 = const()[name = string("op_17449_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17449_dilations_0 = const()[name = string("op_17449_dilations_0"), val = tensor([1, 1])]; + int32 var_17449_groups_0 = const()[name = string("op_17449_groups_0"), val = int32(1)]; + tensor var_17449_cast_fp16 = conv(dilations = var_17449_dilations_0, groups = var_17449_groups_0, pad = var_17449_pad_0, pad_type = var_17449_pad_type_0, strides = var_17449_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_491_cast_fp16)[name = string("op_17449_cast_fp16")]; + tensor input_495_cast_fp16 = mul(x = var_17443_cast_fp16, y = var_17449_cast_fp16)[name = string("input_495_cast_fp16")]; + string hidden_states_589_pad_type_0 = const()[name = string("hidden_states_589_pad_type_0"), val = string("valid")]; + tensor hidden_states_589_strides_0 = const()[name = string("hidden_states_589_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_589_pad_0 = const()[name = string("hidden_states_589_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_589_dilations_0 = const()[name = string("hidden_states_589_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_589_groups_0 = const()[name = string("hidden_states_589_groups_0"), val = int32(1)]; + tensor hidden_states_589_cast_fp16 = conv(dilations = hidden_states_589_dilations_0, groups = hidden_states_589_groups_0, pad = hidden_states_589_pad_0, pad_type = hidden_states_589_pad_type_0, strides = hidden_states_589_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_495_cast_fp16)[name = string("hidden_states_589_cast_fp16")]; + tensor inputs_477_cast_fp16 = add(x = inputs_475_cast_fp16, y = hidden_states_589_cast_fp16)[name = string("inputs_477_cast_fp16")]; + tensor obj_507_begin_0 = const()[name = string("obj_507_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_507_end_0 = const()[name = string("obj_507_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_507_end_mask_0 = const()[name = string("obj_507_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_507_cast_fp16 = slice_by_index(begin = obj_507_begin_0, end = obj_507_end_0, end_mask = obj_507_end_mask_0, x = key_caches_23_cast_fp16)[name = string("obj_507_cast_fp16")]; + tensor obj_509_begin_0 = const()[name = string("obj_509_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_509_end_0 = const()[name = string("obj_509_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_509_end_mask_0 = const()[name = string("obj_509_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_509_cast_fp16 = slice_by_index(begin = obj_509_begin_0, end = obj_509_end_0, end_mask = obj_509_end_mask_0, x = value_caches_23_cast_fp16)[name = string("obj_509_cast_fp16")]; + int32 var_17497 = const()[name = string("op_17497"), val = int32(3)]; + int32 var_17507 = const()[name = string("op_17507"), val = int32(-2)]; + tensor inputs_sq_477_cast_fp16 = mul(x = inputs_477_cast_fp16, y = inputs_477_cast_fp16)[name = string("inputs_sq_477_cast_fp16")]; + tensor variance_477_axes_0 = const()[name = string("variance_477_axes_0"), val = tensor([1])]; + bool variance_477_keep_dims_0 = const()[name = string("variance_477_keep_dims_0"), val = bool(true)]; + tensor variance_477_cast_fp16 = reduce_mean(axes = variance_477_axes_0, keep_dims = variance_477_keep_dims_0, x = inputs_sq_477_cast_fp16)[name = string("variance_477_cast_fp16")]; + fp16 var_17521_to_fp16 = const()[name = string("op_17521_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17522_cast_fp16 = add(x = variance_477_cast_fp16, y = var_17521_to_fp16)[name = string("op_17522_cast_fp16")]; + fp32 var_17523_epsilon_0 = const()[name = string("op_17523_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17523_cast_fp16 = rsqrt(epsilon = var_17523_epsilon_0, x = var_17522_cast_fp16)[name = string("op_17523_cast_fp16")]; + tensor hidden_states_591_cast_fp16 = mul(x = inputs_477_cast_fp16, y = var_17523_cast_fp16)[name = string("hidden_states_591_cast_fp16")]; + tensor obj_505_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_591_cast_fp16)[name = string("obj_505_cast_fp16")]; + string query_343_pad_type_0 = const()[name = string("query_343_pad_type_0"), val = string("valid")]; + tensor query_343_strides_0 = const()[name = string("query_343_strides_0"), val = tensor([1, 1])]; + tensor query_343_pad_0 = const()[name = string("query_343_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_343_dilations_0 = const()[name = string("query_343_dilations_0"), val = tensor([1, 1])]; + int32 query_343_groups_0 = const()[name = string("query_343_groups_0"), val = int32(1)]; + tensor query_343_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_343_dilations_0, groups = query_343_groups_0, pad = query_343_pad_0, pad_type = query_343_pad_type_0, strides = query_343_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_505_cast_fp16)[name = string("query_343_cast_fp16")]; + string current_key_229_pad_type_0 = const()[name = string("current_key_229_pad_type_0"), val = string("valid")]; + tensor current_key_229_strides_0 = const()[name = string("current_key_229_strides_0"), val = tensor([1, 1])]; + tensor current_key_229_pad_0 = const()[name = string("current_key_229_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_229_dilations_0 = const()[name = string("current_key_229_dilations_0"), val = tensor([1, 1])]; + int32 current_key_229_groups_0 = const()[name = string("current_key_229_groups_0"), val = int32(1)]; + tensor current_key_229_cast_fp16 = conv(dilations = current_key_229_dilations_0, groups = current_key_229_groups_0, pad = current_key_229_pad_0, pad_type = current_key_229_pad_type_0, strides = current_key_229_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_505_cast_fp16)[name = string("current_key_229_cast_fp16")]; + string current_value_115_pad_type_0 = const()[name = string("current_value_115_pad_type_0"), val = string("valid")]; + tensor current_value_115_strides_0 = const()[name = string("current_value_115_strides_0"), val = tensor([1, 1])]; + tensor current_value_115_pad_0 = const()[name = string("current_value_115_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_115_dilations_0 = const()[name = string("current_value_115_dilations_0"), val = tensor([1, 1])]; + int32 current_value_115_groups_0 = const()[name = string("current_value_115_groups_0"), val = int32(1)]; + tensor current_value_115_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_115_dilations_0, groups = current_value_115_groups_0, pad = current_value_115_pad_0, pad_type = current_value_115_pad_type_0, strides = current_value_115_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_505_cast_fp16)[name = string("current_value_115_cast_fp16")]; + tensor var_17560 = const()[name = string("op_17560"), val = tensor([16, 128, 1, 1])]; + tensor inputs_479_cast_fp16 = reshape(shape = var_17560, x = query_343_cast_fp16)[name = string("inputs_479_cast_fp16")]; + tensor inputs_sq_479_cast_fp16 = mul(x = inputs_479_cast_fp16, y = inputs_479_cast_fp16)[name = string("inputs_sq_479_cast_fp16")]; + tensor variance_479_axes_0 = const()[name = string("variance_479_axes_0"), val = tensor([1])]; + bool variance_479_keep_dims_0 = const()[name = string("variance_479_keep_dims_0"), val = bool(true)]; + tensor variance_479_cast_fp16 = reduce_mean(axes = variance_479_axes_0, keep_dims = variance_479_keep_dims_0, x = inputs_sq_479_cast_fp16)[name = string("variance_479_cast_fp16")]; + fp16 var_17566_to_fp16 = const()[name = string("op_17566_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17567_cast_fp16 = add(x = variance_479_cast_fp16, y = var_17566_to_fp16)[name = string("op_17567_cast_fp16")]; + fp32 var_17568_epsilon_0 = const()[name = string("op_17568_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17568_cast_fp16 = rsqrt(epsilon = var_17568_epsilon_0, x = var_17567_cast_fp16)[name = string("op_17568_cast_fp16")]; + tensor hidden_states_593_cast_fp16 = mul(x = inputs_479_cast_fp16, y = var_17568_cast_fp16)[name = string("hidden_states_593_cast_fp16")]; + tensor query_normed_115_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_593_cast_fp16)[name = string("query_normed_115_cast_fp16")]; + tensor var_17576 = const()[name = string("op_17576"), val = tensor([8, 128, 1, 1])]; + tensor inputs_481_cast_fp16 = reshape(shape = var_17576, x = current_key_229_cast_fp16)[name = string("inputs_481_cast_fp16")]; + tensor inputs_sq_481_cast_fp16 = mul(x = inputs_481_cast_fp16, y = inputs_481_cast_fp16)[name = string("inputs_sq_481_cast_fp16")]; + tensor variance_481_axes_0 = const()[name = string("variance_481_axes_0"), val = tensor([1])]; + bool variance_481_keep_dims_0 = const()[name = string("variance_481_keep_dims_0"), val = bool(true)]; + tensor variance_481_cast_fp16 = reduce_mean(axes = variance_481_axes_0, keep_dims = variance_481_keep_dims_0, x = inputs_sq_481_cast_fp16)[name = string("variance_481_cast_fp16")]; + fp16 var_17582_to_fp16 = const()[name = string("op_17582_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17583_cast_fp16 = add(x = variance_481_cast_fp16, y = var_17582_to_fp16)[name = string("op_17583_cast_fp16")]; + fp32 var_17584_epsilon_0 = const()[name = string("op_17584_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17584_cast_fp16 = rsqrt(epsilon = var_17584_epsilon_0, x = var_17583_cast_fp16)[name = string("op_17584_cast_fp16")]; + tensor hidden_states_595_cast_fp16 = mul(x = inputs_481_cast_fp16, y = var_17584_cast_fp16)[name = string("hidden_states_595_cast_fp16")]; + tensor current_key_normed_115_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_595_cast_fp16)[name = string("current_key_normed_115_cast_fp16")]; + tensor var_17602 = const()[name = string("op_17602"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_457_cast_fp16 = reshape(shape = var_17602, x = query_normed_115_cast_fp16)[name = string("mh_q_457_cast_fp16")]; + tensor var_17604 = const()[name = string("op_17604"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_457_cast_fp16 = reshape(shape = var_17604, x = current_key_normed_115_cast_fp16)[name = string("mh_k_457_cast_fp16")]; + tensor var_17608_cast_fp16 = mul(x = mh_q_457_cast_fp16, y = cos_111_to_fp16)[name = string("op_17608_cast_fp16")]; + tensor var_17613_begin_0 = const()[name = string("op_17613_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17613_end_0 = const()[name = string("op_17613_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_17613_end_mask_0 = const()[name = string("op_17613_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_17613_cast_fp16 = slice_by_index(begin = var_17613_begin_0, end = var_17613_end_0, end_mask = var_17613_end_mask_0, x = mh_q_457_cast_fp16)[name = string("op_17613_cast_fp16")]; + tensor var_17619_begin_0 = const()[name = string("op_17619_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_17619_end_0 = const()[name = string("op_17619_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_17619_end_mask_0 = const()[name = string("op_17619_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_17619_cast_fp16 = slice_by_index(begin = var_17619_begin_0, end = var_17619_end_0, end_mask = var_17619_end_mask_0, x = mh_q_457_cast_fp16)[name = string("op_17619_cast_fp16")]; + fp16 const_1165_promoted_to_fp16 = const()[name = string("const_1165_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17621_cast_fp16 = mul(x = var_17619_cast_fp16, y = const_1165_promoted_to_fp16)[name = string("op_17621_cast_fp16")]; + bool var_17623_interleave_0 = const()[name = string("op_17623_interleave_0"), val = bool(false)]; + tensor var_17623_cast_fp16 = concat(axis = var_17507, interleave = var_17623_interleave_0, values = (var_17621_cast_fp16, var_17613_cast_fp16))[name = string("op_17623_cast_fp16")]; + tensor var_17624_cast_fp16 = mul(x = var_17623_cast_fp16, y = sin_111_to_fp16)[name = string("op_17624_cast_fp16")]; + tensor mh_q_459_cast_fp16 = add(x = var_17608_cast_fp16, y = var_17624_cast_fp16)[name = string("mh_q_459_cast_fp16")]; + tensor var_17626_cast_fp16 = mul(x = mh_k_457_cast_fp16, y = cos_111_to_fp16)[name = string("op_17626_cast_fp16")]; + tensor var_17631_begin_0 = const()[name = string("op_17631_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17631_end_0 = const()[name = string("op_17631_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_17631_end_mask_0 = const()[name = string("op_17631_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_17631_cast_fp16 = slice_by_index(begin = var_17631_begin_0, end = var_17631_end_0, end_mask = var_17631_end_mask_0, x = mh_k_457_cast_fp16)[name = string("op_17631_cast_fp16")]; + tensor var_17637_begin_0 = const()[name = string("op_17637_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_17637_end_0 = const()[name = string("op_17637_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_17637_end_mask_0 = const()[name = string("op_17637_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_17637_cast_fp16 = slice_by_index(begin = var_17637_begin_0, end = var_17637_end_0, end_mask = var_17637_end_mask_0, x = mh_k_457_cast_fp16)[name = string("op_17637_cast_fp16")]; + fp16 const_1168_promoted_to_fp16 = const()[name = string("const_1168_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17639_cast_fp16 = mul(x = var_17637_cast_fp16, y = const_1168_promoted_to_fp16)[name = string("op_17639_cast_fp16")]; + bool var_17641_interleave_0 = const()[name = string("op_17641_interleave_0"), val = bool(false)]; + tensor var_17641_cast_fp16 = concat(axis = var_17507, interleave = var_17641_interleave_0, values = (var_17639_cast_fp16, var_17631_cast_fp16))[name = string("op_17641_cast_fp16")]; + tensor var_17642_cast_fp16 = mul(x = var_17641_cast_fp16, y = sin_111_to_fp16)[name = string("op_17642_cast_fp16")]; + tensor mh_k_459_cast_fp16 = add(x = var_17626_cast_fp16, y = var_17642_cast_fp16)[name = string("mh_k_459_cast_fp16")]; + tensor var_17646 = const()[name = string("op_17646"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_231_cast_fp16 = reshape(shape = var_17646, x = mh_k_459_cast_fp16)[name = string("current_key_231_cast_fp16")]; + tensor var_17652_to_fp16 = const()[name = string("op_17652_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200448)))]; + tensor var_17653_cast_fp16 = mul(x = obj_507_cast_fp16, y = var_17652_to_fp16)[name = string("op_17653_cast_fp16")]; + tensor var_17650_to_fp16 = const()[name = string("op_17650_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200576)))]; + tensor var_17654_cast_fp16 = mul(x = current_key_231_cast_fp16, y = var_17650_to_fp16)[name = string("op_17654_cast_fp16")]; + tensor key_231_cast_fp16 = add(x = var_17653_cast_fp16, y = var_17654_cast_fp16)[name = string("key_231_cast_fp16")]; + tensor var_17656_to_fp16 = const()[name = string("op_17656_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200448)))]; + tensor var_17657_cast_fp16 = mul(x = obj_509_cast_fp16, y = var_17656_to_fp16)[name = string("op_17657_cast_fp16")]; + tensor var_17658_cast_fp16 = mul(x = current_value_115_cast_fp16, y = var_17650_to_fp16)[name = string("op_17658_cast_fp16")]; + tensor value_115_cast_fp16 = add(x = var_17657_cast_fp16, y = var_17658_cast_fp16)[name = string("value_115_cast_fp16")]; + fp16 var_17665_to_fp16 = const()[name = string("op_17665_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_463_cast_fp16 = mul(x = mh_q_459_cast_fp16, y = var_17665_to_fp16)[name = string("mh_q_463_cast_fp16")]; + tensor var_17667 = const()[name = string("op_17667"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_461_cast_fp16 = reshape(shape = var_17667, x = key_231_cast_fp16)[name = string("mh_k_461_cast_fp16")]; + tensor var_17669 = const()[name = string("op_17669"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_229_cast_fp16 = reshape(shape = var_17669, x = value_115_cast_fp16)[name = string("mh_v_229_cast_fp16")]; + tensor transpose_228_perm_0 = const()[name = string("transpose_228_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_114_reps_0 = const()[name = string("tile_114_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_228_cast_fp16 = transpose(perm = transpose_228_perm_0, x = mh_k_461_cast_fp16)[name = string("transpose_137")]; + tensor tile_114_cast_fp16 = tile(reps = tile_114_reps_0, x = transpose_228_cast_fp16)[name = string("tile_114_cast_fp16")]; + tensor concat_286 = const()[name = string("concat_286"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_228_cast_fp16 = reshape(shape = concat_286, x = tile_114_cast_fp16)[name = string("reshape_228_cast_fp16")]; + tensor transpose_229_perm_0 = const()[name = string("transpose_229_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_287 = const()[name = string("concat_287"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_229_cast_fp16 = transpose(perm = transpose_229_perm_0, x = reshape_228_cast_fp16)[name = string("transpose_136")]; + tensor reshape_229_cast_fp16 = reshape(shape = concat_287, x = transpose_229_cast_fp16)[name = string("reshape_229_cast_fp16")]; + tensor transpose_230_perm_0 = const()[name = string("transpose_230_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_115_reps_0 = const()[name = string("tile_115_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_230_cast_fp16 = transpose(perm = transpose_230_perm_0, x = mh_v_229_cast_fp16)[name = string("transpose_135")]; + tensor tile_115_cast_fp16 = tile(reps = tile_115_reps_0, x = transpose_230_cast_fp16)[name = string("tile_115_cast_fp16")]; + tensor concat_288 = const()[name = string("concat_288"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_230_cast_fp16 = reshape(shape = concat_288, x = tile_115_cast_fp16)[name = string("reshape_230_cast_fp16")]; + tensor transpose_231_perm_0 = const()[name = string("transpose_231_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_289 = const()[name = string("concat_289"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_231_cast_fp16 = transpose(perm = transpose_231_perm_0, x = reshape_230_cast_fp16)[name = string("transpose_134")]; + tensor reshape_231_cast_fp16 = reshape(shape = concat_289, x = transpose_231_cast_fp16)[name = string("reshape_231_cast_fp16")]; + tensor transpose_545_perm_0 = const()[name = string("transpose_545_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_343_transpose_x_1 = const()[name = string("mh_w_343_transpose_x_1"), val = bool(true)]; + bool mh_w_343_transpose_y_1 = const()[name = string("mh_w_343_transpose_y_1"), val = bool(false)]; + tensor transpose_545_cast_fp16 = transpose(perm = transpose_545_perm_0, x = reshape_229_cast_fp16)[name = string("transpose_133")]; + tensor mh_w_343_cast_fp16 = matmul(transpose_x = mh_w_343_transpose_x_1, transpose_y = mh_w_343_transpose_y_1, x = mh_q_463_cast_fp16, y = transpose_545_cast_fp16)[name = string("mh_w_343_cast_fp16")]; + tensor var_17677_to_fp16 = const()[name = string("op_17677_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200704)))]; + tensor mh_w_345_cast_fp16 = add(x = mh_w_343_cast_fp16, y = var_17677_to_fp16)[name = string("mh_w_345_cast_fp16")]; + tensor mh_w_347_cast_fp16 = softmax(axis = var_17497, x = mh_w_345_cast_fp16)[name = string("mh_w_347_cast_fp16")]; + tensor transpose_546_perm_0 = const()[name = string("transpose_546_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_115_transpose_x_1 = const()[name = string("attn_115_transpose_x_1"), val = bool(false)]; + bool attn_115_transpose_y_1 = const()[name = string("attn_115_transpose_y_1"), val = bool(true)]; + tensor transpose_546_cast_fp16 = transpose(perm = transpose_546_perm_0, x = reshape_231_cast_fp16)[name = string("transpose_132")]; + tensor attn_115_cast_fp16 = matmul(transpose_x = attn_115_transpose_x_1, transpose_y = attn_115_transpose_y_1, x = transpose_546_cast_fp16, y = mh_w_347_cast_fp16)[name = string("attn_115_cast_fp16")]; + tensor var_17683 = const()[name = string("op_17683"), val = tensor([1, 2048, 1, 1])]; + tensor input_497_cast_fp16 = reshape(shape = var_17683, x = attn_115_cast_fp16)[name = string("input_497_cast_fp16")]; + string obj_511_pad_type_0 = const()[name = string("obj_511_pad_type_0"), val = string("valid")]; + tensor obj_511_strides_0 = const()[name = string("obj_511_strides_0"), val = tensor([1, 1])]; + tensor obj_511_pad_0 = const()[name = string("obj_511_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_511_dilations_0 = const()[name = string("obj_511_dilations_0"), val = tensor([1, 1])]; + int32 obj_511_groups_0 = const()[name = string("obj_511_groups_0"), val = int32(1)]; + tensor obj_511_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_511_dilations_0, groups = obj_511_groups_0, pad = obj_511_pad_0, pad_type = obj_511_pad_type_0, strides = obj_511_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_497_cast_fp16)[name = string("obj_511_cast_fp16")]; + tensor inputs_483_cast_fp16 = add(x = inputs_477_cast_fp16, y = obj_511_cast_fp16)[name = string("inputs_483_cast_fp16")]; + tensor inputs_sq_483_cast_fp16 = mul(x = inputs_483_cast_fp16, y = inputs_483_cast_fp16)[name = string("inputs_sq_483_cast_fp16")]; + tensor variance_483_axes_0 = const()[name = string("variance_483_axes_0"), val = tensor([1])]; + bool variance_483_keep_dims_0 = const()[name = string("variance_483_keep_dims_0"), val = bool(true)]; + tensor variance_483_cast_fp16 = reduce_mean(axes = variance_483_axes_0, keep_dims = variance_483_keep_dims_0, x = inputs_sq_483_cast_fp16)[name = string("variance_483_cast_fp16")]; + fp16 var_17701_to_fp16 = const()[name = string("op_17701_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17702_cast_fp16 = add(x = variance_483_cast_fp16, y = var_17701_to_fp16)[name = string("op_17702_cast_fp16")]; + fp32 var_17703_epsilon_0 = const()[name = string("op_17703_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17703_cast_fp16 = rsqrt(epsilon = var_17703_epsilon_0, x = var_17702_cast_fp16)[name = string("op_17703_cast_fp16")]; + tensor hidden_states_597_cast_fp16 = mul(x = inputs_483_cast_fp16, y = var_17703_cast_fp16)[name = string("hidden_states_597_cast_fp16")]; + tensor input_499_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_597_cast_fp16)[name = string("input_499_cast_fp16")]; + string input_501_pad_type_0 = const()[name = string("input_501_pad_type_0"), val = string("valid")]; + tensor input_501_strides_0 = const()[name = string("input_501_strides_0"), val = tensor([1, 1])]; + tensor input_501_pad_0 = const()[name = string("input_501_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_501_dilations_0 = const()[name = string("input_501_dilations_0"), val = tensor([1, 1])]; + int32 input_501_groups_0 = const()[name = string("input_501_groups_0"), val = int32(1)]; + tensor input_501_cast_fp16 = conv(dilations = input_501_dilations_0, groups = input_501_groups_0, pad = input_501_pad_0, pad_type = input_501_pad_type_0, strides = input_501_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_499_cast_fp16)[name = string("input_501_cast_fp16")]; + tensor var_17717_cast_fp16 = silu(x = input_501_cast_fp16)[name = string("op_17717_cast_fp16")]; + string var_17723_pad_type_0 = const()[name = string("op_17723_pad_type_0"), val = string("valid")]; + tensor var_17723_strides_0 = const()[name = string("op_17723_strides_0"), val = tensor([1, 1])]; + tensor var_17723_pad_0 = const()[name = string("op_17723_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17723_dilations_0 = const()[name = string("op_17723_dilations_0"), val = tensor([1, 1])]; + int32 var_17723_groups_0 = const()[name = string("op_17723_groups_0"), val = int32(1)]; + tensor var_17723_cast_fp16 = conv(dilations = var_17723_dilations_0, groups = var_17723_groups_0, pad = var_17723_pad_0, pad_type = var_17723_pad_type_0, strides = var_17723_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_499_cast_fp16)[name = string("op_17723_cast_fp16")]; + tensor input_503_cast_fp16 = mul(x = var_17717_cast_fp16, y = var_17723_cast_fp16)[name = string("input_503_cast_fp16")]; + string hidden_states_599_pad_type_0 = const()[name = string("hidden_states_599_pad_type_0"), val = string("valid")]; + tensor hidden_states_599_strides_0 = const()[name = string("hidden_states_599_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_599_pad_0 = const()[name = string("hidden_states_599_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_599_dilations_0 = const()[name = string("hidden_states_599_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_599_groups_0 = const()[name = string("hidden_states_599_groups_0"), val = int32(1)]; + tensor hidden_states_599_cast_fp16 = conv(dilations = hidden_states_599_dilations_0, groups = hidden_states_599_groups_0, pad = hidden_states_599_pad_0, pad_type = hidden_states_599_pad_type_0, strides = hidden_states_599_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_503_cast_fp16)[name = string("hidden_states_599_cast_fp16")]; + tensor inputs_485_cast_fp16 = add(x = inputs_483_cast_fp16, y = hidden_states_599_cast_fp16)[name = string("inputs_485_cast_fp16")]; + tensor obj_515_begin_0 = const()[name = string("obj_515_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_515_end_0 = const()[name = string("obj_515_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_515_end_mask_0 = const()[name = string("obj_515_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_515_cast_fp16 = slice_by_index(begin = obj_515_begin_0, end = obj_515_end_0, end_mask = obj_515_end_mask_0, x = key_caches_23_cast_fp16)[name = string("obj_515_cast_fp16")]; + tensor obj_517_begin_0 = const()[name = string("obj_517_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_517_end_0 = const()[name = string("obj_517_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_517_end_mask_0 = const()[name = string("obj_517_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_517_cast_fp16 = slice_by_index(begin = obj_517_begin_0, end = obj_517_end_0, end_mask = obj_517_end_mask_0, x = value_caches_23_cast_fp16)[name = string("obj_517_cast_fp16")]; + int32 var_17771 = const()[name = string("op_17771"), val = int32(3)]; + int32 var_17781 = const()[name = string("op_17781"), val = int32(-2)]; + tensor inputs_sq_485_cast_fp16 = mul(x = inputs_485_cast_fp16, y = inputs_485_cast_fp16)[name = string("inputs_sq_485_cast_fp16")]; + tensor variance_485_axes_0 = const()[name = string("variance_485_axes_0"), val = tensor([1])]; + bool variance_485_keep_dims_0 = const()[name = string("variance_485_keep_dims_0"), val = bool(true)]; + tensor variance_485_cast_fp16 = reduce_mean(axes = variance_485_axes_0, keep_dims = variance_485_keep_dims_0, x = inputs_sq_485_cast_fp16)[name = string("variance_485_cast_fp16")]; + fp16 var_17795_to_fp16 = const()[name = string("op_17795_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17796_cast_fp16 = add(x = variance_485_cast_fp16, y = var_17795_to_fp16)[name = string("op_17796_cast_fp16")]; + fp32 var_17797_epsilon_0 = const()[name = string("op_17797_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17797_cast_fp16 = rsqrt(epsilon = var_17797_epsilon_0, x = var_17796_cast_fp16)[name = string("op_17797_cast_fp16")]; + tensor hidden_states_601_cast_fp16 = mul(x = inputs_485_cast_fp16, y = var_17797_cast_fp16)[name = string("hidden_states_601_cast_fp16")]; + tensor obj_513_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_601_cast_fp16)[name = string("obj_513_cast_fp16")]; + string query_349_pad_type_0 = const()[name = string("query_349_pad_type_0"), val = string("valid")]; + tensor query_349_strides_0 = const()[name = string("query_349_strides_0"), val = tensor([1, 1])]; + tensor query_349_pad_0 = const()[name = string("query_349_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_349_dilations_0 = const()[name = string("query_349_dilations_0"), val = tensor([1, 1])]; + int32 query_349_groups_0 = const()[name = string("query_349_groups_0"), val = int32(1)]; + tensor query_349_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_349_dilations_0, groups = query_349_groups_0, pad = query_349_pad_0, pad_type = query_349_pad_type_0, strides = query_349_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_513_cast_fp16)[name = string("query_349_cast_fp16")]; + string current_key_233_pad_type_0 = const()[name = string("current_key_233_pad_type_0"), val = string("valid")]; + tensor current_key_233_strides_0 = const()[name = string("current_key_233_strides_0"), val = tensor([1, 1])]; + tensor current_key_233_pad_0 = const()[name = string("current_key_233_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_233_dilations_0 = const()[name = string("current_key_233_dilations_0"), val = tensor([1, 1])]; + int32 current_key_233_groups_0 = const()[name = string("current_key_233_groups_0"), val = int32(1)]; + tensor current_key_233_cast_fp16 = conv(dilations = current_key_233_dilations_0, groups = current_key_233_groups_0, pad = current_key_233_pad_0, pad_type = current_key_233_pad_type_0, strides = current_key_233_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_513_cast_fp16)[name = string("current_key_233_cast_fp16")]; + string current_value_117_pad_type_0 = const()[name = string("current_value_117_pad_type_0"), val = string("valid")]; + tensor current_value_117_strides_0 = const()[name = string("current_value_117_strides_0"), val = tensor([1, 1])]; + tensor current_value_117_pad_0 = const()[name = string("current_value_117_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_117_dilations_0 = const()[name = string("current_value_117_dilations_0"), val = tensor([1, 1])]; + int32 current_value_117_groups_0 = const()[name = string("current_value_117_groups_0"), val = int32(1)]; + tensor current_value_117_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_117_dilations_0, groups = current_value_117_groups_0, pad = current_value_117_pad_0, pad_type = current_value_117_pad_type_0, strides = current_value_117_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_513_cast_fp16)[name = string("current_value_117_cast_fp16")]; + tensor var_17834 = const()[name = string("op_17834"), val = tensor([16, 128, 1, 1])]; + tensor inputs_487_cast_fp16 = reshape(shape = var_17834, x = query_349_cast_fp16)[name = string("inputs_487_cast_fp16")]; + tensor inputs_sq_487_cast_fp16 = mul(x = inputs_487_cast_fp16, y = inputs_487_cast_fp16)[name = string("inputs_sq_487_cast_fp16")]; + tensor variance_487_axes_0 = const()[name = string("variance_487_axes_0"), val = tensor([1])]; + bool variance_487_keep_dims_0 = const()[name = string("variance_487_keep_dims_0"), val = bool(true)]; + tensor variance_487_cast_fp16 = reduce_mean(axes = variance_487_axes_0, keep_dims = variance_487_keep_dims_0, x = inputs_sq_487_cast_fp16)[name = string("variance_487_cast_fp16")]; + fp16 var_17840_to_fp16 = const()[name = string("op_17840_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17841_cast_fp16 = add(x = variance_487_cast_fp16, y = var_17840_to_fp16)[name = string("op_17841_cast_fp16")]; + fp32 var_17842_epsilon_0 = const()[name = string("op_17842_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17842_cast_fp16 = rsqrt(epsilon = var_17842_epsilon_0, x = var_17841_cast_fp16)[name = string("op_17842_cast_fp16")]; + tensor hidden_states_603_cast_fp16 = mul(x = inputs_487_cast_fp16, y = var_17842_cast_fp16)[name = string("hidden_states_603_cast_fp16")]; + tensor query_normed_117_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_603_cast_fp16)[name = string("query_normed_117_cast_fp16")]; + tensor var_17850 = const()[name = string("op_17850"), val = tensor([8, 128, 1, 1])]; + tensor inputs_489_cast_fp16 = reshape(shape = var_17850, x = current_key_233_cast_fp16)[name = string("inputs_489_cast_fp16")]; + tensor inputs_sq_489_cast_fp16 = mul(x = inputs_489_cast_fp16, y = inputs_489_cast_fp16)[name = string("inputs_sq_489_cast_fp16")]; + tensor variance_489_axes_0 = const()[name = string("variance_489_axes_0"), val = tensor([1])]; + bool variance_489_keep_dims_0 = const()[name = string("variance_489_keep_dims_0"), val = bool(true)]; + tensor variance_489_cast_fp16 = reduce_mean(axes = variance_489_axes_0, keep_dims = variance_489_keep_dims_0, x = inputs_sq_489_cast_fp16)[name = string("variance_489_cast_fp16")]; + fp16 var_17856_to_fp16 = const()[name = string("op_17856_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17857_cast_fp16 = add(x = variance_489_cast_fp16, y = var_17856_to_fp16)[name = string("op_17857_cast_fp16")]; + fp32 var_17858_epsilon_0 = const()[name = string("op_17858_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17858_cast_fp16 = rsqrt(epsilon = var_17858_epsilon_0, x = var_17857_cast_fp16)[name = string("op_17858_cast_fp16")]; + tensor hidden_states_605_cast_fp16 = mul(x = inputs_489_cast_fp16, y = var_17858_cast_fp16)[name = string("hidden_states_605_cast_fp16")]; + tensor current_key_normed_117_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_605_cast_fp16)[name = string("current_key_normed_117_cast_fp16")]; + tensor var_17876 = const()[name = string("op_17876"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_465_cast_fp16 = reshape(shape = var_17876, x = query_normed_117_cast_fp16)[name = string("mh_q_465_cast_fp16")]; + tensor var_17878 = const()[name = string("op_17878"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_465_cast_fp16 = reshape(shape = var_17878, x = current_key_normed_117_cast_fp16)[name = string("mh_k_465_cast_fp16")]; + tensor var_17882_cast_fp16 = mul(x = mh_q_465_cast_fp16, y = cos_111_to_fp16)[name = string("op_17882_cast_fp16")]; + tensor var_17887_begin_0 = const()[name = string("op_17887_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17887_end_0 = const()[name = string("op_17887_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_17887_end_mask_0 = const()[name = string("op_17887_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_17887_cast_fp16 = slice_by_index(begin = var_17887_begin_0, end = var_17887_end_0, end_mask = var_17887_end_mask_0, x = mh_q_465_cast_fp16)[name = string("op_17887_cast_fp16")]; + tensor var_17893_begin_0 = const()[name = string("op_17893_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_17893_end_0 = const()[name = string("op_17893_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_17893_end_mask_0 = const()[name = string("op_17893_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_17893_cast_fp16 = slice_by_index(begin = var_17893_begin_0, end = var_17893_end_0, end_mask = var_17893_end_mask_0, x = mh_q_465_cast_fp16)[name = string("op_17893_cast_fp16")]; + fp16 const_1185_promoted_to_fp16 = const()[name = string("const_1185_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17895_cast_fp16 = mul(x = var_17893_cast_fp16, y = const_1185_promoted_to_fp16)[name = string("op_17895_cast_fp16")]; + bool var_17897_interleave_0 = const()[name = string("op_17897_interleave_0"), val = bool(false)]; + tensor var_17897_cast_fp16 = concat(axis = var_17781, interleave = var_17897_interleave_0, values = (var_17895_cast_fp16, var_17887_cast_fp16))[name = string("op_17897_cast_fp16")]; + tensor var_17898_cast_fp16 = mul(x = var_17897_cast_fp16, y = sin_111_to_fp16)[name = string("op_17898_cast_fp16")]; + tensor mh_q_467_cast_fp16 = add(x = var_17882_cast_fp16, y = var_17898_cast_fp16)[name = string("mh_q_467_cast_fp16")]; + tensor var_17900_cast_fp16 = mul(x = mh_k_465_cast_fp16, y = cos_111_to_fp16)[name = string("op_17900_cast_fp16")]; + tensor var_17905_begin_0 = const()[name = string("op_17905_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17905_end_0 = const()[name = string("op_17905_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_17905_end_mask_0 = const()[name = string("op_17905_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_17905_cast_fp16 = slice_by_index(begin = var_17905_begin_0, end = var_17905_end_0, end_mask = var_17905_end_mask_0, x = mh_k_465_cast_fp16)[name = string("op_17905_cast_fp16")]; + tensor var_17911_begin_0 = const()[name = string("op_17911_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_17911_end_0 = const()[name = string("op_17911_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_17911_end_mask_0 = const()[name = string("op_17911_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_17911_cast_fp16 = slice_by_index(begin = var_17911_begin_0, end = var_17911_end_0, end_mask = var_17911_end_mask_0, x = mh_k_465_cast_fp16)[name = string("op_17911_cast_fp16")]; + fp16 const_1188_promoted_to_fp16 = const()[name = string("const_1188_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_17913_cast_fp16 = mul(x = var_17911_cast_fp16, y = const_1188_promoted_to_fp16)[name = string("op_17913_cast_fp16")]; + bool var_17915_interleave_0 = const()[name = string("op_17915_interleave_0"), val = bool(false)]; + tensor var_17915_cast_fp16 = concat(axis = var_17781, interleave = var_17915_interleave_0, values = (var_17913_cast_fp16, var_17905_cast_fp16))[name = string("op_17915_cast_fp16")]; + tensor var_17916_cast_fp16 = mul(x = var_17915_cast_fp16, y = sin_111_to_fp16)[name = string("op_17916_cast_fp16")]; + tensor mh_k_467_cast_fp16 = add(x = var_17900_cast_fp16, y = var_17916_cast_fp16)[name = string("mh_k_467_cast_fp16")]; + tensor var_17920 = const()[name = string("op_17920"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_235_cast_fp16 = reshape(shape = var_17920, x = mh_k_467_cast_fp16)[name = string("current_key_235_cast_fp16")]; + tensor var_17926_to_fp16 = const()[name = string("op_17926_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200448)))]; + tensor var_17927_cast_fp16 = mul(x = obj_515_cast_fp16, y = var_17926_to_fp16)[name = string("op_17927_cast_fp16")]; + tensor var_17924_to_fp16 = const()[name = string("op_17924_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200576)))]; + tensor var_17928_cast_fp16 = mul(x = current_key_235_cast_fp16, y = var_17924_to_fp16)[name = string("op_17928_cast_fp16")]; + tensor key_235_cast_fp16 = add(x = var_17927_cast_fp16, y = var_17928_cast_fp16)[name = string("key_235_cast_fp16")]; + tensor var_17930_to_fp16 = const()[name = string("op_17930_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200448)))]; + tensor var_17931_cast_fp16 = mul(x = obj_517_cast_fp16, y = var_17930_to_fp16)[name = string("op_17931_cast_fp16")]; + tensor var_17932_cast_fp16 = mul(x = current_value_117_cast_fp16, y = var_17924_to_fp16)[name = string("op_17932_cast_fp16")]; + tensor value_117_cast_fp16 = add(x = var_17931_cast_fp16, y = var_17932_cast_fp16)[name = string("value_117_cast_fp16")]; + fp16 var_17939_to_fp16 = const()[name = string("op_17939_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_471_cast_fp16 = mul(x = mh_q_467_cast_fp16, y = var_17939_to_fp16)[name = string("mh_q_471_cast_fp16")]; + tensor var_17941 = const()[name = string("op_17941"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_469_cast_fp16 = reshape(shape = var_17941, x = key_235_cast_fp16)[name = string("mh_k_469_cast_fp16")]; + tensor var_17943 = const()[name = string("op_17943"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_233_cast_fp16 = reshape(shape = var_17943, x = value_117_cast_fp16)[name = string("mh_v_233_cast_fp16")]; + tensor transpose_232_perm_0 = const()[name = string("transpose_232_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_116_reps_0 = const()[name = string("tile_116_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_232_cast_fp16 = transpose(perm = transpose_232_perm_0, x = mh_k_469_cast_fp16)[name = string("transpose_131")]; + tensor tile_116_cast_fp16 = tile(reps = tile_116_reps_0, x = transpose_232_cast_fp16)[name = string("tile_116_cast_fp16")]; + tensor concat_290 = const()[name = string("concat_290"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_232_cast_fp16 = reshape(shape = concat_290, x = tile_116_cast_fp16)[name = string("reshape_232_cast_fp16")]; + tensor transpose_233_perm_0 = const()[name = string("transpose_233_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_291 = const()[name = string("concat_291"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_233_cast_fp16 = transpose(perm = transpose_233_perm_0, x = reshape_232_cast_fp16)[name = string("transpose_130")]; + tensor reshape_233_cast_fp16 = reshape(shape = concat_291, x = transpose_233_cast_fp16)[name = string("reshape_233_cast_fp16")]; + tensor transpose_234_perm_0 = const()[name = string("transpose_234_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_117_reps_0 = const()[name = string("tile_117_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_234_cast_fp16 = transpose(perm = transpose_234_perm_0, x = mh_v_233_cast_fp16)[name = string("transpose_129")]; + tensor tile_117_cast_fp16 = tile(reps = tile_117_reps_0, x = transpose_234_cast_fp16)[name = string("tile_117_cast_fp16")]; + tensor concat_292 = const()[name = string("concat_292"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_234_cast_fp16 = reshape(shape = concat_292, x = tile_117_cast_fp16)[name = string("reshape_234_cast_fp16")]; + tensor transpose_235_perm_0 = const()[name = string("transpose_235_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_293 = const()[name = string("concat_293"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_235_cast_fp16 = transpose(perm = transpose_235_perm_0, x = reshape_234_cast_fp16)[name = string("transpose_128")]; + tensor reshape_235_cast_fp16 = reshape(shape = concat_293, x = transpose_235_cast_fp16)[name = string("reshape_235_cast_fp16")]; + tensor transpose_549_perm_0 = const()[name = string("transpose_549_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_349_transpose_x_1 = const()[name = string("mh_w_349_transpose_x_1"), val = bool(true)]; + bool mh_w_349_transpose_y_1 = const()[name = string("mh_w_349_transpose_y_1"), val = bool(false)]; + tensor transpose_549_cast_fp16 = transpose(perm = transpose_549_perm_0, x = reshape_233_cast_fp16)[name = string("transpose_127")]; + tensor mh_w_349_cast_fp16 = matmul(transpose_x = mh_w_349_transpose_x_1, transpose_y = mh_w_349_transpose_y_1, x = mh_q_471_cast_fp16, y = transpose_549_cast_fp16)[name = string("mh_w_349_cast_fp16")]; + tensor var_17951_to_fp16 = const()[name = string("op_17951_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200704)))]; + tensor mh_w_351_cast_fp16 = add(x = mh_w_349_cast_fp16, y = var_17951_to_fp16)[name = string("mh_w_351_cast_fp16")]; + tensor mh_w_353_cast_fp16 = softmax(axis = var_17771, x = mh_w_351_cast_fp16)[name = string("mh_w_353_cast_fp16")]; + tensor transpose_550_perm_0 = const()[name = string("transpose_550_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_117_transpose_x_1 = const()[name = string("attn_117_transpose_x_1"), val = bool(false)]; + bool attn_117_transpose_y_1 = const()[name = string("attn_117_transpose_y_1"), val = bool(true)]; + tensor transpose_550_cast_fp16 = transpose(perm = transpose_550_perm_0, x = reshape_235_cast_fp16)[name = string("transpose_126")]; + tensor attn_117_cast_fp16 = matmul(transpose_x = attn_117_transpose_x_1, transpose_y = attn_117_transpose_y_1, x = transpose_550_cast_fp16, y = mh_w_353_cast_fp16)[name = string("attn_117_cast_fp16")]; + tensor var_17957 = const()[name = string("op_17957"), val = tensor([1, 2048, 1, 1])]; + tensor input_505_cast_fp16 = reshape(shape = var_17957, x = attn_117_cast_fp16)[name = string("input_505_cast_fp16")]; + string obj_519_pad_type_0 = const()[name = string("obj_519_pad_type_0"), val = string("valid")]; + tensor obj_519_strides_0 = const()[name = string("obj_519_strides_0"), val = tensor([1, 1])]; + tensor obj_519_pad_0 = const()[name = string("obj_519_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_519_dilations_0 = const()[name = string("obj_519_dilations_0"), val = tensor([1, 1])]; + int32 obj_519_groups_0 = const()[name = string("obj_519_groups_0"), val = int32(1)]; + tensor obj_519_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_519_dilations_0, groups = obj_519_groups_0, pad = obj_519_pad_0, pad_type = obj_519_pad_type_0, strides = obj_519_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_505_cast_fp16)[name = string("obj_519_cast_fp16")]; + tensor inputs_491_cast_fp16 = add(x = inputs_485_cast_fp16, y = obj_519_cast_fp16)[name = string("inputs_491_cast_fp16")]; + tensor inputs_sq_491_cast_fp16 = mul(x = inputs_491_cast_fp16, y = inputs_491_cast_fp16)[name = string("inputs_sq_491_cast_fp16")]; + tensor variance_491_axes_0 = const()[name = string("variance_491_axes_0"), val = tensor([1])]; + bool variance_491_keep_dims_0 = const()[name = string("variance_491_keep_dims_0"), val = bool(true)]; + tensor variance_491_cast_fp16 = reduce_mean(axes = variance_491_axes_0, keep_dims = variance_491_keep_dims_0, x = inputs_sq_491_cast_fp16)[name = string("variance_491_cast_fp16")]; + fp16 var_17975_to_fp16 = const()[name = string("op_17975_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_17976_cast_fp16 = add(x = variance_491_cast_fp16, y = var_17975_to_fp16)[name = string("op_17976_cast_fp16")]; + fp32 var_17977_epsilon_0 = const()[name = string("op_17977_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_17977_cast_fp16 = rsqrt(epsilon = var_17977_epsilon_0, x = var_17976_cast_fp16)[name = string("op_17977_cast_fp16")]; + tensor hidden_states_607_cast_fp16 = mul(x = inputs_491_cast_fp16, y = var_17977_cast_fp16)[name = string("hidden_states_607_cast_fp16")]; + tensor input_507_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_607_cast_fp16)[name = string("input_507_cast_fp16")]; + string input_509_pad_type_0 = const()[name = string("input_509_pad_type_0"), val = string("valid")]; + tensor input_509_strides_0 = const()[name = string("input_509_strides_0"), val = tensor([1, 1])]; + tensor input_509_pad_0 = const()[name = string("input_509_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_509_dilations_0 = const()[name = string("input_509_dilations_0"), val = tensor([1, 1])]; + int32 input_509_groups_0 = const()[name = string("input_509_groups_0"), val = int32(1)]; + tensor input_509_cast_fp16 = conv(dilations = input_509_dilations_0, groups = input_509_groups_0, pad = input_509_pad_0, pad_type = input_509_pad_type_0, strides = input_509_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_507_cast_fp16)[name = string("input_509_cast_fp16")]; + tensor var_17991_cast_fp16 = silu(x = input_509_cast_fp16)[name = string("op_17991_cast_fp16")]; + string var_17997_pad_type_0 = const()[name = string("op_17997_pad_type_0"), val = string("valid")]; + tensor var_17997_strides_0 = const()[name = string("op_17997_strides_0"), val = tensor([1, 1])]; + tensor var_17997_pad_0 = const()[name = string("op_17997_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_17997_dilations_0 = const()[name = string("op_17997_dilations_0"), val = tensor([1, 1])]; + int32 var_17997_groups_0 = const()[name = string("op_17997_groups_0"), val = int32(1)]; + tensor var_17997_cast_fp16 = conv(dilations = var_17997_dilations_0, groups = var_17997_groups_0, pad = var_17997_pad_0, pad_type = var_17997_pad_type_0, strides = var_17997_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_507_cast_fp16)[name = string("op_17997_cast_fp16")]; + tensor input_511_cast_fp16 = mul(x = var_17991_cast_fp16, y = var_17997_cast_fp16)[name = string("input_511_cast_fp16")]; + string hidden_states_609_pad_type_0 = const()[name = string("hidden_states_609_pad_type_0"), val = string("valid")]; + tensor hidden_states_609_strides_0 = const()[name = string("hidden_states_609_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_609_pad_0 = const()[name = string("hidden_states_609_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_609_dilations_0 = const()[name = string("hidden_states_609_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_609_groups_0 = const()[name = string("hidden_states_609_groups_0"), val = int32(1)]; + tensor hidden_states_609_cast_fp16 = conv(dilations = hidden_states_609_dilations_0, groups = hidden_states_609_groups_0, pad = hidden_states_609_pad_0, pad_type = hidden_states_609_pad_type_0, strides = hidden_states_609_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_511_cast_fp16)[name = string("hidden_states_609_cast_fp16")]; + tensor inputs_493_cast_fp16 = add(x = inputs_491_cast_fp16, y = hidden_states_609_cast_fp16)[name = string("inputs_493_cast_fp16")]; + tensor obj_523_begin_0 = const()[name = string("obj_523_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_523_end_0 = const()[name = string("obj_523_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_523_end_mask_0 = const()[name = string("obj_523_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_523_cast_fp16 = slice_by_index(begin = obj_523_begin_0, end = obj_523_end_0, end_mask = obj_523_end_mask_0, x = key_caches_23_cast_fp16)[name = string("obj_523_cast_fp16")]; + tensor obj_525_begin_0 = const()[name = string("obj_525_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_525_end_0 = const()[name = string("obj_525_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_525_end_mask_0 = const()[name = string("obj_525_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_525_cast_fp16 = slice_by_index(begin = obj_525_begin_0, end = obj_525_end_0, end_mask = obj_525_end_mask_0, x = value_caches_23_cast_fp16)[name = string("obj_525_cast_fp16")]; + int32 var_18045 = const()[name = string("op_18045"), val = int32(3)]; + int32 var_18055 = const()[name = string("op_18055"), val = int32(-2)]; + tensor inputs_sq_493_cast_fp16 = mul(x = inputs_493_cast_fp16, y = inputs_493_cast_fp16)[name = string("inputs_sq_493_cast_fp16")]; + tensor variance_493_axes_0 = const()[name = string("variance_493_axes_0"), val = tensor([1])]; + bool variance_493_keep_dims_0 = const()[name = string("variance_493_keep_dims_0"), val = bool(true)]; + tensor variance_493_cast_fp16 = reduce_mean(axes = variance_493_axes_0, keep_dims = variance_493_keep_dims_0, x = inputs_sq_493_cast_fp16)[name = string("variance_493_cast_fp16")]; + fp16 var_18069_to_fp16 = const()[name = string("op_18069_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18070_cast_fp16 = add(x = variance_493_cast_fp16, y = var_18069_to_fp16)[name = string("op_18070_cast_fp16")]; + fp32 var_18071_epsilon_0 = const()[name = string("op_18071_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18071_cast_fp16 = rsqrt(epsilon = var_18071_epsilon_0, x = var_18070_cast_fp16)[name = string("op_18071_cast_fp16")]; + tensor hidden_states_611_cast_fp16 = mul(x = inputs_493_cast_fp16, y = var_18071_cast_fp16)[name = string("hidden_states_611_cast_fp16")]; + tensor obj_521_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_611_cast_fp16)[name = string("obj_521_cast_fp16")]; + string query_355_pad_type_0 = const()[name = string("query_355_pad_type_0"), val = string("valid")]; + tensor query_355_strides_0 = const()[name = string("query_355_strides_0"), val = tensor([1, 1])]; + tensor query_355_pad_0 = const()[name = string("query_355_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_355_dilations_0 = const()[name = string("query_355_dilations_0"), val = tensor([1, 1])]; + int32 query_355_groups_0 = const()[name = string("query_355_groups_0"), val = int32(1)]; + tensor query_355_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_355_dilations_0, groups = query_355_groups_0, pad = query_355_pad_0, pad_type = query_355_pad_type_0, strides = query_355_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_521_cast_fp16)[name = string("query_355_cast_fp16")]; + string current_key_237_pad_type_0 = const()[name = string("current_key_237_pad_type_0"), val = string("valid")]; + tensor current_key_237_strides_0 = const()[name = string("current_key_237_strides_0"), val = tensor([1, 1])]; + tensor current_key_237_pad_0 = const()[name = string("current_key_237_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_237_dilations_0 = const()[name = string("current_key_237_dilations_0"), val = tensor([1, 1])]; + int32 current_key_237_groups_0 = const()[name = string("current_key_237_groups_0"), val = int32(1)]; + tensor current_key_237_cast_fp16 = conv(dilations = current_key_237_dilations_0, groups = current_key_237_groups_0, pad = current_key_237_pad_0, pad_type = current_key_237_pad_type_0, strides = current_key_237_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_521_cast_fp16)[name = string("current_key_237_cast_fp16")]; + string current_value_119_pad_type_0 = const()[name = string("current_value_119_pad_type_0"), val = string("valid")]; + tensor current_value_119_strides_0 = const()[name = string("current_value_119_strides_0"), val = tensor([1, 1])]; + tensor current_value_119_pad_0 = const()[name = string("current_value_119_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_119_dilations_0 = const()[name = string("current_value_119_dilations_0"), val = tensor([1, 1])]; + int32 current_value_119_groups_0 = const()[name = string("current_value_119_groups_0"), val = int32(1)]; + tensor current_value_119_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_119_dilations_0, groups = current_value_119_groups_0, pad = current_value_119_pad_0, pad_type = current_value_119_pad_type_0, strides = current_value_119_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_521_cast_fp16)[name = string("current_value_119_cast_fp16")]; + tensor var_18108 = const()[name = string("op_18108"), val = tensor([16, 128, 1, 1])]; + tensor inputs_495_cast_fp16 = reshape(shape = var_18108, x = query_355_cast_fp16)[name = string("inputs_495_cast_fp16")]; + tensor inputs_sq_495_cast_fp16 = mul(x = inputs_495_cast_fp16, y = inputs_495_cast_fp16)[name = string("inputs_sq_495_cast_fp16")]; + tensor variance_495_axes_0 = const()[name = string("variance_495_axes_0"), val = tensor([1])]; + bool variance_495_keep_dims_0 = const()[name = string("variance_495_keep_dims_0"), val = bool(true)]; + tensor variance_495_cast_fp16 = reduce_mean(axes = variance_495_axes_0, keep_dims = variance_495_keep_dims_0, x = inputs_sq_495_cast_fp16)[name = string("variance_495_cast_fp16")]; + fp16 var_18114_to_fp16 = const()[name = string("op_18114_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18115_cast_fp16 = add(x = variance_495_cast_fp16, y = var_18114_to_fp16)[name = string("op_18115_cast_fp16")]; + fp32 var_18116_epsilon_0 = const()[name = string("op_18116_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18116_cast_fp16 = rsqrt(epsilon = var_18116_epsilon_0, x = var_18115_cast_fp16)[name = string("op_18116_cast_fp16")]; + tensor hidden_states_613_cast_fp16 = mul(x = inputs_495_cast_fp16, y = var_18116_cast_fp16)[name = string("hidden_states_613_cast_fp16")]; + tensor query_normed_119_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_613_cast_fp16)[name = string("query_normed_119_cast_fp16")]; + tensor var_18124 = const()[name = string("op_18124"), val = tensor([8, 128, 1, 1])]; + tensor inputs_497_cast_fp16 = reshape(shape = var_18124, x = current_key_237_cast_fp16)[name = string("inputs_497_cast_fp16")]; + tensor inputs_sq_497_cast_fp16 = mul(x = inputs_497_cast_fp16, y = inputs_497_cast_fp16)[name = string("inputs_sq_497_cast_fp16")]; + tensor variance_497_axes_0 = const()[name = string("variance_497_axes_0"), val = tensor([1])]; + bool variance_497_keep_dims_0 = const()[name = string("variance_497_keep_dims_0"), val = bool(true)]; + tensor variance_497_cast_fp16 = reduce_mean(axes = variance_497_axes_0, keep_dims = variance_497_keep_dims_0, x = inputs_sq_497_cast_fp16)[name = string("variance_497_cast_fp16")]; + fp16 var_18130_to_fp16 = const()[name = string("op_18130_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18131_cast_fp16 = add(x = variance_497_cast_fp16, y = var_18130_to_fp16)[name = string("op_18131_cast_fp16")]; + fp32 var_18132_epsilon_0 = const()[name = string("op_18132_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18132_cast_fp16 = rsqrt(epsilon = var_18132_epsilon_0, x = var_18131_cast_fp16)[name = string("op_18132_cast_fp16")]; + tensor hidden_states_615_cast_fp16 = mul(x = inputs_497_cast_fp16, y = var_18132_cast_fp16)[name = string("hidden_states_615_cast_fp16")]; + tensor current_key_normed_119_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_615_cast_fp16)[name = string("current_key_normed_119_cast_fp16")]; + tensor var_18150 = const()[name = string("op_18150"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_473_cast_fp16 = reshape(shape = var_18150, x = query_normed_119_cast_fp16)[name = string("mh_q_473_cast_fp16")]; + tensor var_18152 = const()[name = string("op_18152"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_473_cast_fp16 = reshape(shape = var_18152, x = current_key_normed_119_cast_fp16)[name = string("mh_k_473_cast_fp16")]; + tensor var_18156_cast_fp16 = mul(x = mh_q_473_cast_fp16, y = cos_111_to_fp16)[name = string("op_18156_cast_fp16")]; + tensor var_18161_begin_0 = const()[name = string("op_18161_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18161_end_0 = const()[name = string("op_18161_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_18161_end_mask_0 = const()[name = string("op_18161_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_18161_cast_fp16 = slice_by_index(begin = var_18161_begin_0, end = var_18161_end_0, end_mask = var_18161_end_mask_0, x = mh_q_473_cast_fp16)[name = string("op_18161_cast_fp16")]; + tensor var_18167_begin_0 = const()[name = string("op_18167_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_18167_end_0 = const()[name = string("op_18167_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_18167_end_mask_0 = const()[name = string("op_18167_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_18167_cast_fp16 = slice_by_index(begin = var_18167_begin_0, end = var_18167_end_0, end_mask = var_18167_end_mask_0, x = mh_q_473_cast_fp16)[name = string("op_18167_cast_fp16")]; + fp16 const_1205_promoted_to_fp16 = const()[name = string("const_1205_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18169_cast_fp16 = mul(x = var_18167_cast_fp16, y = const_1205_promoted_to_fp16)[name = string("op_18169_cast_fp16")]; + bool var_18171_interleave_0 = const()[name = string("op_18171_interleave_0"), val = bool(false)]; + tensor var_18171_cast_fp16 = concat(axis = var_18055, interleave = var_18171_interleave_0, values = (var_18169_cast_fp16, var_18161_cast_fp16))[name = string("op_18171_cast_fp16")]; + tensor var_18172_cast_fp16 = mul(x = var_18171_cast_fp16, y = sin_111_to_fp16)[name = string("op_18172_cast_fp16")]; + tensor mh_q_475_cast_fp16 = add(x = var_18156_cast_fp16, y = var_18172_cast_fp16)[name = string("mh_q_475_cast_fp16")]; + tensor var_18174_cast_fp16 = mul(x = mh_k_473_cast_fp16, y = cos_111_to_fp16)[name = string("op_18174_cast_fp16")]; + tensor var_18179_begin_0 = const()[name = string("op_18179_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18179_end_0 = const()[name = string("op_18179_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_18179_end_mask_0 = const()[name = string("op_18179_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_18179_cast_fp16 = slice_by_index(begin = var_18179_begin_0, end = var_18179_end_0, end_mask = var_18179_end_mask_0, x = mh_k_473_cast_fp16)[name = string("op_18179_cast_fp16")]; + tensor var_18185_begin_0 = const()[name = string("op_18185_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_18185_end_0 = const()[name = string("op_18185_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_18185_end_mask_0 = const()[name = string("op_18185_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_18185_cast_fp16 = slice_by_index(begin = var_18185_begin_0, end = var_18185_end_0, end_mask = var_18185_end_mask_0, x = mh_k_473_cast_fp16)[name = string("op_18185_cast_fp16")]; + fp16 const_1208_promoted_to_fp16 = const()[name = string("const_1208_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18187_cast_fp16 = mul(x = var_18185_cast_fp16, y = const_1208_promoted_to_fp16)[name = string("op_18187_cast_fp16")]; + bool var_18189_interleave_0 = const()[name = string("op_18189_interleave_0"), val = bool(false)]; + tensor var_18189_cast_fp16 = concat(axis = var_18055, interleave = var_18189_interleave_0, values = (var_18187_cast_fp16, var_18179_cast_fp16))[name = string("op_18189_cast_fp16")]; + tensor var_18190_cast_fp16 = mul(x = var_18189_cast_fp16, y = sin_111_to_fp16)[name = string("op_18190_cast_fp16")]; + tensor mh_k_475_cast_fp16 = add(x = var_18174_cast_fp16, y = var_18190_cast_fp16)[name = string("mh_k_475_cast_fp16")]; + tensor var_18194 = const()[name = string("op_18194"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_239_cast_fp16 = reshape(shape = var_18194, x = mh_k_475_cast_fp16)[name = string("current_key_239_cast_fp16")]; + tensor var_18200_to_fp16 = const()[name = string("op_18200_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200448)))]; + tensor var_18201_cast_fp16 = mul(x = obj_523_cast_fp16, y = var_18200_to_fp16)[name = string("op_18201_cast_fp16")]; + tensor var_18198_to_fp16 = const()[name = string("op_18198_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200576)))]; + tensor var_18202_cast_fp16 = mul(x = current_key_239_cast_fp16, y = var_18198_to_fp16)[name = string("op_18202_cast_fp16")]; + tensor key_239_cast_fp16 = add(x = var_18201_cast_fp16, y = var_18202_cast_fp16)[name = string("key_239_cast_fp16")]; + tensor var_18204_to_fp16 = const()[name = string("op_18204_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200448)))]; + tensor var_18205_cast_fp16 = mul(x = obj_525_cast_fp16, y = var_18204_to_fp16)[name = string("op_18205_cast_fp16")]; + tensor var_18206_cast_fp16 = mul(x = current_value_119_cast_fp16, y = var_18198_to_fp16)[name = string("op_18206_cast_fp16")]; + tensor value_119_cast_fp16 = add(x = var_18205_cast_fp16, y = var_18206_cast_fp16)[name = string("value_119_cast_fp16")]; + fp16 var_18213_to_fp16 = const()[name = string("op_18213_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_479_cast_fp16 = mul(x = mh_q_475_cast_fp16, y = var_18213_to_fp16)[name = string("mh_q_479_cast_fp16")]; + tensor var_18215 = const()[name = string("op_18215"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_477_cast_fp16 = reshape(shape = var_18215, x = key_239_cast_fp16)[name = string("mh_k_477_cast_fp16")]; + tensor var_18217 = const()[name = string("op_18217"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_237_cast_fp16 = reshape(shape = var_18217, x = value_119_cast_fp16)[name = string("mh_v_237_cast_fp16")]; + tensor transpose_236_perm_0 = const()[name = string("transpose_236_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_118_reps_0 = const()[name = string("tile_118_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_236_cast_fp16 = transpose(perm = transpose_236_perm_0, x = mh_k_477_cast_fp16)[name = string("transpose_125")]; + tensor tile_118_cast_fp16 = tile(reps = tile_118_reps_0, x = transpose_236_cast_fp16)[name = string("tile_118_cast_fp16")]; + tensor concat_294 = const()[name = string("concat_294"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_236_cast_fp16 = reshape(shape = concat_294, x = tile_118_cast_fp16)[name = string("reshape_236_cast_fp16")]; + tensor transpose_237_perm_0 = const()[name = string("transpose_237_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_295 = const()[name = string("concat_295"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_237_cast_fp16 = transpose(perm = transpose_237_perm_0, x = reshape_236_cast_fp16)[name = string("transpose_124")]; + tensor reshape_237_cast_fp16 = reshape(shape = concat_295, x = transpose_237_cast_fp16)[name = string("reshape_237_cast_fp16")]; + tensor transpose_238_perm_0 = const()[name = string("transpose_238_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_119_reps_0 = const()[name = string("tile_119_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_238_cast_fp16 = transpose(perm = transpose_238_perm_0, x = mh_v_237_cast_fp16)[name = string("transpose_123")]; + tensor tile_119_cast_fp16 = tile(reps = tile_119_reps_0, x = transpose_238_cast_fp16)[name = string("tile_119_cast_fp16")]; + tensor concat_296 = const()[name = string("concat_296"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_238_cast_fp16 = reshape(shape = concat_296, x = tile_119_cast_fp16)[name = string("reshape_238_cast_fp16")]; + tensor transpose_239_perm_0 = const()[name = string("transpose_239_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_297 = const()[name = string("concat_297"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_239_cast_fp16 = transpose(perm = transpose_239_perm_0, x = reshape_238_cast_fp16)[name = string("transpose_122")]; + tensor reshape_239_cast_fp16 = reshape(shape = concat_297, x = transpose_239_cast_fp16)[name = string("reshape_239_cast_fp16")]; + tensor transpose_553_perm_0 = const()[name = string("transpose_553_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_355_transpose_x_1 = const()[name = string("mh_w_355_transpose_x_1"), val = bool(true)]; + bool mh_w_355_transpose_y_1 = const()[name = string("mh_w_355_transpose_y_1"), val = bool(false)]; + tensor transpose_553_cast_fp16 = transpose(perm = transpose_553_perm_0, x = reshape_237_cast_fp16)[name = string("transpose_121")]; + tensor mh_w_355_cast_fp16 = matmul(transpose_x = mh_w_355_transpose_x_1, transpose_y = mh_w_355_transpose_y_1, x = mh_q_479_cast_fp16, y = transpose_553_cast_fp16)[name = string("mh_w_355_cast_fp16")]; + tensor var_18225_to_fp16 = const()[name = string("op_18225_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200704)))]; + tensor mh_w_357_cast_fp16 = add(x = mh_w_355_cast_fp16, y = var_18225_to_fp16)[name = string("mh_w_357_cast_fp16")]; + tensor mh_w_359_cast_fp16 = softmax(axis = var_18045, x = mh_w_357_cast_fp16)[name = string("mh_w_359_cast_fp16")]; + tensor transpose_554_perm_0 = const()[name = string("transpose_554_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_119_transpose_x_1 = const()[name = string("attn_119_transpose_x_1"), val = bool(false)]; + bool attn_119_transpose_y_1 = const()[name = string("attn_119_transpose_y_1"), val = bool(true)]; + tensor transpose_554_cast_fp16 = transpose(perm = transpose_554_perm_0, x = reshape_239_cast_fp16)[name = string("transpose_120")]; + tensor attn_119_cast_fp16 = matmul(transpose_x = attn_119_transpose_x_1, transpose_y = attn_119_transpose_y_1, x = transpose_554_cast_fp16, y = mh_w_359_cast_fp16)[name = string("attn_119_cast_fp16")]; + tensor var_18231 = const()[name = string("op_18231"), val = tensor([1, 2048, 1, 1])]; + tensor input_513_cast_fp16 = reshape(shape = var_18231, x = attn_119_cast_fp16)[name = string("input_513_cast_fp16")]; + string obj_527_pad_type_0 = const()[name = string("obj_527_pad_type_0"), val = string("valid")]; + tensor obj_527_strides_0 = const()[name = string("obj_527_strides_0"), val = tensor([1, 1])]; + tensor obj_527_pad_0 = const()[name = string("obj_527_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_527_dilations_0 = const()[name = string("obj_527_dilations_0"), val = tensor([1, 1])]; + int32 obj_527_groups_0 = const()[name = string("obj_527_groups_0"), val = int32(1)]; + tensor obj_527_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_527_dilations_0, groups = obj_527_groups_0, pad = obj_527_pad_0, pad_type = obj_527_pad_type_0, strides = obj_527_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_513_cast_fp16)[name = string("obj_527_cast_fp16")]; + tensor inputs_499_cast_fp16 = add(x = inputs_493_cast_fp16, y = obj_527_cast_fp16)[name = string("inputs_499_cast_fp16")]; + tensor inputs_sq_499_cast_fp16 = mul(x = inputs_499_cast_fp16, y = inputs_499_cast_fp16)[name = string("inputs_sq_499_cast_fp16")]; + tensor variance_499_axes_0 = const()[name = string("variance_499_axes_0"), val = tensor([1])]; + bool variance_499_keep_dims_0 = const()[name = string("variance_499_keep_dims_0"), val = bool(true)]; + tensor variance_499_cast_fp16 = reduce_mean(axes = variance_499_axes_0, keep_dims = variance_499_keep_dims_0, x = inputs_sq_499_cast_fp16)[name = string("variance_499_cast_fp16")]; + fp16 var_18249_to_fp16 = const()[name = string("op_18249_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18250_cast_fp16 = add(x = variance_499_cast_fp16, y = var_18249_to_fp16)[name = string("op_18250_cast_fp16")]; + fp32 var_18251_epsilon_0 = const()[name = string("op_18251_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18251_cast_fp16 = rsqrt(epsilon = var_18251_epsilon_0, x = var_18250_cast_fp16)[name = string("op_18251_cast_fp16")]; + tensor hidden_states_617_cast_fp16 = mul(x = inputs_499_cast_fp16, y = var_18251_cast_fp16)[name = string("hidden_states_617_cast_fp16")]; + tensor input_515_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_617_cast_fp16)[name = string("input_515_cast_fp16")]; + string input_517_pad_type_0 = const()[name = string("input_517_pad_type_0"), val = string("valid")]; + tensor input_517_strides_0 = const()[name = string("input_517_strides_0"), val = tensor([1, 1])]; + tensor input_517_pad_0 = const()[name = string("input_517_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_517_dilations_0 = const()[name = string("input_517_dilations_0"), val = tensor([1, 1])]; + int32 input_517_groups_0 = const()[name = string("input_517_groups_0"), val = int32(1)]; + tensor input_517_cast_fp16 = conv(dilations = input_517_dilations_0, groups = input_517_groups_0, pad = input_517_pad_0, pad_type = input_517_pad_type_0, strides = input_517_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_515_cast_fp16)[name = string("input_517_cast_fp16")]; + tensor var_18265_cast_fp16 = silu(x = input_517_cast_fp16)[name = string("op_18265_cast_fp16")]; + string var_18271_pad_type_0 = const()[name = string("op_18271_pad_type_0"), val = string("valid")]; + tensor var_18271_strides_0 = const()[name = string("op_18271_strides_0"), val = tensor([1, 1])]; + tensor var_18271_pad_0 = const()[name = string("op_18271_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18271_dilations_0 = const()[name = string("op_18271_dilations_0"), val = tensor([1, 1])]; + int32 var_18271_groups_0 = const()[name = string("op_18271_groups_0"), val = int32(1)]; + tensor var_18271_cast_fp16 = conv(dilations = var_18271_dilations_0, groups = var_18271_groups_0, pad = var_18271_pad_0, pad_type = var_18271_pad_type_0, strides = var_18271_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_515_cast_fp16)[name = string("op_18271_cast_fp16")]; + tensor input_519_cast_fp16 = mul(x = var_18265_cast_fp16, y = var_18271_cast_fp16)[name = string("input_519_cast_fp16")]; + string hidden_states_619_pad_type_0 = const()[name = string("hidden_states_619_pad_type_0"), val = string("valid")]; + tensor hidden_states_619_strides_0 = const()[name = string("hidden_states_619_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_619_pad_0 = const()[name = string("hidden_states_619_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_619_dilations_0 = const()[name = string("hidden_states_619_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_619_groups_0 = const()[name = string("hidden_states_619_groups_0"), val = int32(1)]; + tensor hidden_states_619_cast_fp16 = conv(dilations = hidden_states_619_dilations_0, groups = hidden_states_619_groups_0, pad = hidden_states_619_pad_0, pad_type = hidden_states_619_pad_type_0, strides = hidden_states_619_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_519_cast_fp16)[name = string("hidden_states_619_cast_fp16")]; + tensor inputs_501_cast_fp16 = add(x = inputs_499_cast_fp16, y = hidden_states_619_cast_fp16)[name = string("inputs_501_cast_fp16")]; + int32 var_18299 = const()[name = string("op_18299"), val = int32(1)]; + bool key_caches_25_interleave_0 = const()[name = string("key_caches_25_interleave_0"), val = bool(false)]; + tensor key_caches_25_cast_fp16 = concat(axis = var_18299, interleave = key_caches_25_interleave_0, values = (key_223_cast_fp16, key_227_cast_fp16, key_231_cast_fp16, key_235_cast_fp16, key_239_cast_fp16))[name = string("key_caches_25_cast_fp16")]; + int32 var_18302 = const()[name = string("op_18302"), val = int32(1)]; + bool value_caches_25_interleave_0 = const()[name = string("value_caches_25_interleave_0"), val = bool(false)]; + tensor value_caches_25_cast_fp16 = concat(axis = var_18302, interleave = value_caches_25_interleave_0, values = (value_111_cast_fp16, value_113_cast_fp16, value_115_cast_fp16, value_117_cast_fp16, value_119_cast_fp16))[name = string("value_caches_25_cast_fp16")]; + tensor inputs_sq_501_cast_fp16 = mul(x = inputs_501_cast_fp16, y = inputs_501_cast_fp16)[name = string("inputs_sq_501_cast_fp16")]; + tensor variance_501_axes_0 = const()[name = string("variance_501_axes_0"), val = tensor([1])]; + bool variance_501_keep_dims_0 = const()[name = string("variance_501_keep_dims_0"), val = bool(true)]; + tensor variance_501_cast_fp16 = reduce_mean(axes = variance_501_axes_0, keep_dims = variance_501_keep_dims_0, x = inputs_sq_501_cast_fp16)[name = string("variance_501_cast_fp16")]; + fp16 var_18312_to_fp16 = const()[name = string("op_18312_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18313_cast_fp16 = add(x = variance_501_cast_fp16, y = var_18312_to_fp16)[name = string("op_18313_cast_fp16")]; + fp32 var_18314_epsilon_0 = const()[name = string("op_18314_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18314_cast_fp16 = rsqrt(epsilon = var_18314_epsilon_0, x = var_18313_cast_fp16)[name = string("op_18314_cast_fp16")]; + tensor hidden_states_621_cast_fp16 = mul(x = inputs_501_cast_fp16, y = var_18314_cast_fp16)[name = string("hidden_states_621_cast_fp16")]; + tensor input_521_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_621_cast_fp16)[name = string("input_521_cast_fp16")]; + string logits_41_pad_type_0 = const()[name = string("logits_41_pad_type_0"), val = string("valid")]; + tensor logits_41_strides_0 = const()[name = string("logits_41_strides_0"), val = tensor([1, 1])]; + tensor logits_41_pad_0 = const()[name = string("logits_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_41_dilations_0 = const()[name = string("logits_41_dilations_0"), val = tensor([1, 1])]; + int32 logits_41_groups_0 = const()[name = string("logits_41_groups_0"), val = int32(1)]; + tensor lm_heads_10_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101784512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103881728))))[name = string("lm_heads_10_weight_to_fp16_palettized")]; + tensor logits_41_cast_fp16 = conv(dilations = logits_41_dilations_0, groups = logits_41_groups_0, pad = logits_41_pad_0, pad_type = logits_41_pad_type_0, strides = logits_41_strides_0, weight = lm_heads_10_weight_to_fp16_palettized, x = input_521_cast_fp16)[name = string("logits_41_cast_fp16")]; + tensor var_18332 = const()[name = string("op_18332"), val = tensor([1, 2048])]; + tensor logits_43_cast_fp16 = reshape(shape = var_18332, x = logits_41_cast_fp16)[name = string("logits_43_cast_fp16")]; + tensor scaled_logits_21_cast_fp16 = real_div(x = logits_43_cast_fp16, y = temperature)[name = string("scaled_logits_21_cast_fp16")]; + int32 var_18342 = const()[name = string("op_18342"), val = int32(100)]; + int32 top_values_21_axis_0 = const()[name = string("top_values_21_axis_0"), val = int32(1)]; + bool top_values_21_ascending_0 = const()[name = string("top_values_21_ascending_0"), val = bool(false)]; + bool top_values_21_sort_0 = const()[name = string("top_values_21_sort_0"), val = bool(true)]; + bool top_values_21_return_indices_0 = const()[name = string("top_values_21_return_indices_0"), val = bool(true)]; + string top_values_21_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_21_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_21_cast_fp16_cast_uint16_0, tensor top_values_21_cast_fp16_cast_uint16_1 = topk(ascending = top_values_21_ascending_0, axis = top_values_21_axis_0, k = var_18342, output_indices_dtype = top_values_21_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_21_return_indices_0, sort = top_values_21_sort_0, x = scaled_logits_21_cast_fp16)[name = string("top_values_21_cast_fp16_cast_uint16")]; + tensor var_18348_cast_fp16 = mul(x = top_values_21_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_18348_cast_fp16")]; + tensor var_18352_cast_fp16 = add(x = var_18348_cast_fp16, y = var_3001_cast_fp16)[name = string("op_18352_cast_fp16")]; + tensor reduce_min_10_axes_0 = const()[name = string("reduce_min_10_axes_0"), val = tensor([1])]; + bool reduce_min_10_keep_dims_0 = const()[name = string("reduce_min_10_keep_dims_0"), val = bool(true)]; + tensor reduce_min_10_cast_fp16 = reduce_min(axes = reduce_min_10_axes_0, keep_dims = reduce_min_10_keep_dims_0, x = var_18352_cast_fp16)[name = string("reduce_min_10_cast_fp16")]; + tensor var_18355_cast_fp16 = greater_equal(x = scaled_logits_21_cast_fp16, y = reduce_min_10_cast_fp16)[name = string("op_18355_cast_fp16")]; + fp16 var_18356_value_0_to_fp16 = const()[name = string("op_18356_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_18356_cast_fp16 = fill_like(ref_tensor = scaled_logits_21_cast_fp16, value = var_18356_value_0_to_fp16)[name = string("op_18356_cast_fp16")]; + tensor masked_logits_21_cast_fp16 = select(a = scaled_logits_21_cast_fp16, b = var_18356_cast_fp16, cond = var_18355_cast_fp16)[name = string("masked_logits_21_cast_fp16")]; + tensor var_18360_begin_0 = const()[name = string("op_18360_begin_0"), val = tensor([10, 0])]; + tensor var_18360_end_0 = const()[name = string("op_18360_end_0"), val = tensor([11, 2048])]; + tensor var_18360_end_mask_0 = const()[name = string("op_18360_end_mask_0"), val = tensor([false, true])]; + tensor var_18360_squeeze_mask_0 = const()[name = string("op_18360_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_18360_cast_fp16 = slice_by_index(begin = var_18360_begin_0, end = var_18360_end_0, end_mask = var_18360_end_mask_0, squeeze_mask = var_18360_squeeze_mask_0, x = gumbel)[name = string("op_18360_cast_fp16")]; + tensor var_18363 = const()[name = string("op_18363"), val = tensor([1, 2048])]; + tensor var_18364_cast_fp16 = reshape(shape = var_18363, x = var_18360_cast_fp16)[name = string("op_18364_cast_fp16")]; + tensor noisy_logits_21_cast_fp16 = add(x = masked_logits_21_cast_fp16, y = var_18364_cast_fp16)[name = string("noisy_logits_21_cast_fp16")]; + int32 code_21_axis_0 = const()[name = string("code_21_axis_0"), val = int32(1)]; + bool code_21_keep_dims_0 = const()[name = string("code_21_keep_dims_0"), val = bool(false)]; + string code_21_output_dtype_0 = const()[name = string("code_21_output_dtype_0"), val = string("int32")]; + tensor code_21_cast_fp16 = reduce_argmax(axis = code_21_axis_0, keep_dims = code_21_keep_dims_0, output_dtype = code_21_output_dtype_0, x = noisy_logits_21_cast_fp16)[name = string("code_21_cast_fp16")]; + int32 var_18375 = const()[name = string("op_18375"), val = int32(20480)]; + tensor input_523 = add(x = code_21_cast_fp16, y = var_18375)[name = string("input_523")]; + int32 code_embed_41_axis_0 = const()[name = string("code_embed_41_axis_0"), val = int32(0)]; + int32 code_embed_41_batch_dims_0 = const()[name = string("code_embed_41_batch_dims_0"), val = int32(0)]; + bool code_embed_41_validate_indices_0 = const()[name = string("code_embed_41_validate_indices_0"), val = bool(false)]; + string input_523_to_uint16_dtype_0 = const()[name = string("input_523_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_523_to_uint16 = cast(dtype = input_523_to_uint16_dtype_0, x = input_523)[name = string("cast_4")]; + tensor code_embed_41_cast_fp16_cast_uint16 = gather(axis = code_embed_41_axis_0, batch_dims = code_embed_41_batch_dims_0, indices = input_523_to_uint16, validate_indices = code_embed_41_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_41_cast_fp16_cast_uint16")]; + tensor var_18379 = const()[name = string("op_18379"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_43_cast_fp16 = reshape(shape = var_18379, x = code_embed_41_cast_fp16_cast_uint16)[name = string("code_embed_43_cast_fp16")]; + tensor embed_sum_23_cast_fp16 = add(x = embed_sum_21_cast_fp16, y = code_embed_43_cast_fp16)[name = string("embed_sum_23_cast_fp16")]; + string inputs_503_pad_type_0 = const()[name = string("inputs_503_pad_type_0"), val = string("valid")]; + tensor inputs_503_strides_0 = const()[name = string("inputs_503_strides_0"), val = tensor([1, 1])]; + tensor inputs_503_pad_0 = const()[name = string("inputs_503_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_503_dilations_0 = const()[name = string("inputs_503_dilations_0"), val = tensor([1, 1])]; + int32 inputs_503_groups_0 = const()[name = string("inputs_503_groups_0"), val = int32(1)]; + tensor inputs_503_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_503_dilations_0, groups = inputs_503_groups_0, pad = inputs_503_pad_0, pad_type = inputs_503_pad_type_0, strides = inputs_503_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_43_cast_fp16)[name = string("inputs_503_cast_fp16")]; + tensor obj_531_begin_0 = const()[name = string("obj_531_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_531_end_0 = const()[name = string("obj_531_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_531_end_mask_0 = const()[name = string("obj_531_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_531_cast_fp16 = slice_by_index(begin = obj_531_begin_0, end = obj_531_end_0, end_mask = obj_531_end_mask_0, x = key_caches_25_cast_fp16)[name = string("obj_531_cast_fp16")]; + tensor obj_533_begin_0 = const()[name = string("obj_533_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_533_end_0 = const()[name = string("obj_533_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_533_end_mask_0 = const()[name = string("obj_533_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_533_cast_fp16 = slice_by_index(begin = obj_533_begin_0, end = obj_533_end_0, end_mask = obj_533_end_mask_0, x = value_caches_25_cast_fp16)[name = string("obj_533_cast_fp16")]; + int32 var_18484 = const()[name = string("op_18484"), val = int32(3)]; + int32 var_18494 = const()[name = string("op_18494"), val = int32(-2)]; + tensor inputs_sq_503_cast_fp16 = mul(x = inputs_503_cast_fp16, y = inputs_503_cast_fp16)[name = string("inputs_sq_503_cast_fp16")]; + tensor variance_503_axes_0 = const()[name = string("variance_503_axes_0"), val = tensor([1])]; + bool variance_503_keep_dims_0 = const()[name = string("variance_503_keep_dims_0"), val = bool(true)]; + tensor variance_503_cast_fp16 = reduce_mean(axes = variance_503_axes_0, keep_dims = variance_503_keep_dims_0, x = inputs_sq_503_cast_fp16)[name = string("variance_503_cast_fp16")]; + fp16 var_18508_to_fp16 = const()[name = string("op_18508_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18509_cast_fp16 = add(x = variance_503_cast_fp16, y = var_18508_to_fp16)[name = string("op_18509_cast_fp16")]; + fp32 var_18510_epsilon_0 = const()[name = string("op_18510_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18510_cast_fp16 = rsqrt(epsilon = var_18510_epsilon_0, x = var_18509_cast_fp16)[name = string("op_18510_cast_fp16")]; + tensor hidden_states_623_cast_fp16 = mul(x = inputs_503_cast_fp16, y = var_18510_cast_fp16)[name = string("hidden_states_623_cast_fp16")]; + tensor obj_529_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_623_cast_fp16)[name = string("obj_529_cast_fp16")]; + string query_361_pad_type_0 = const()[name = string("query_361_pad_type_0"), val = string("valid")]; + tensor query_361_strides_0 = const()[name = string("query_361_strides_0"), val = tensor([1, 1])]; + tensor query_361_pad_0 = const()[name = string("query_361_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_361_dilations_0 = const()[name = string("query_361_dilations_0"), val = tensor([1, 1])]; + int32 query_361_groups_0 = const()[name = string("query_361_groups_0"), val = int32(1)]; + tensor query_361_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_361_dilations_0, groups = query_361_groups_0, pad = query_361_pad_0, pad_type = query_361_pad_type_0, strides = query_361_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_529_cast_fp16)[name = string("query_361_cast_fp16")]; + string current_key_241_pad_type_0 = const()[name = string("current_key_241_pad_type_0"), val = string("valid")]; + tensor current_key_241_strides_0 = const()[name = string("current_key_241_strides_0"), val = tensor([1, 1])]; + tensor current_key_241_pad_0 = const()[name = string("current_key_241_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_241_dilations_0 = const()[name = string("current_key_241_dilations_0"), val = tensor([1, 1])]; + int32 current_key_241_groups_0 = const()[name = string("current_key_241_groups_0"), val = int32(1)]; + tensor current_key_241_cast_fp16 = conv(dilations = current_key_241_dilations_0, groups = current_key_241_groups_0, pad = current_key_241_pad_0, pad_type = current_key_241_pad_type_0, strides = current_key_241_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_529_cast_fp16)[name = string("current_key_241_cast_fp16")]; + string current_value_121_pad_type_0 = const()[name = string("current_value_121_pad_type_0"), val = string("valid")]; + tensor current_value_121_strides_0 = const()[name = string("current_value_121_strides_0"), val = tensor([1, 1])]; + tensor current_value_121_pad_0 = const()[name = string("current_value_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_121_dilations_0 = const()[name = string("current_value_121_dilations_0"), val = tensor([1, 1])]; + int32 current_value_121_groups_0 = const()[name = string("current_value_121_groups_0"), val = int32(1)]; + tensor current_value_121_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_121_dilations_0, groups = current_value_121_groups_0, pad = current_value_121_pad_0, pad_type = current_value_121_pad_type_0, strides = current_value_121_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_529_cast_fp16)[name = string("current_value_121_cast_fp16")]; + tensor var_18547 = const()[name = string("op_18547"), val = tensor([16, 128, 1, 1])]; + tensor inputs_505_cast_fp16 = reshape(shape = var_18547, x = query_361_cast_fp16)[name = string("inputs_505_cast_fp16")]; + tensor inputs_sq_505_cast_fp16 = mul(x = inputs_505_cast_fp16, y = inputs_505_cast_fp16)[name = string("inputs_sq_505_cast_fp16")]; + tensor variance_505_axes_0 = const()[name = string("variance_505_axes_0"), val = tensor([1])]; + bool variance_505_keep_dims_0 = const()[name = string("variance_505_keep_dims_0"), val = bool(true)]; + tensor variance_505_cast_fp16 = reduce_mean(axes = variance_505_axes_0, keep_dims = variance_505_keep_dims_0, x = inputs_sq_505_cast_fp16)[name = string("variance_505_cast_fp16")]; + fp16 var_18553_to_fp16 = const()[name = string("op_18553_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18554_cast_fp16 = add(x = variance_505_cast_fp16, y = var_18553_to_fp16)[name = string("op_18554_cast_fp16")]; + fp32 var_18555_epsilon_0 = const()[name = string("op_18555_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18555_cast_fp16 = rsqrt(epsilon = var_18555_epsilon_0, x = var_18554_cast_fp16)[name = string("op_18555_cast_fp16")]; + tensor hidden_states_625_cast_fp16 = mul(x = inputs_505_cast_fp16, y = var_18555_cast_fp16)[name = string("hidden_states_625_cast_fp16")]; + tensor query_normed_121_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_625_cast_fp16)[name = string("query_normed_121_cast_fp16")]; + tensor var_18563 = const()[name = string("op_18563"), val = tensor([8, 128, 1, 1])]; + tensor inputs_507_cast_fp16 = reshape(shape = var_18563, x = current_key_241_cast_fp16)[name = string("inputs_507_cast_fp16")]; + tensor inputs_sq_507_cast_fp16 = mul(x = inputs_507_cast_fp16, y = inputs_507_cast_fp16)[name = string("inputs_sq_507_cast_fp16")]; + tensor variance_507_axes_0 = const()[name = string("variance_507_axes_0"), val = tensor([1])]; + bool variance_507_keep_dims_0 = const()[name = string("variance_507_keep_dims_0"), val = bool(true)]; + tensor variance_507_cast_fp16 = reduce_mean(axes = variance_507_axes_0, keep_dims = variance_507_keep_dims_0, x = inputs_sq_507_cast_fp16)[name = string("variance_507_cast_fp16")]; + fp16 var_18569_to_fp16 = const()[name = string("op_18569_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18570_cast_fp16 = add(x = variance_507_cast_fp16, y = var_18569_to_fp16)[name = string("op_18570_cast_fp16")]; + fp32 var_18571_epsilon_0 = const()[name = string("op_18571_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18571_cast_fp16 = rsqrt(epsilon = var_18571_epsilon_0, x = var_18570_cast_fp16)[name = string("op_18571_cast_fp16")]; + tensor hidden_states_627_cast_fp16 = mul(x = inputs_507_cast_fp16, y = var_18571_cast_fp16)[name = string("hidden_states_627_cast_fp16")]; + tensor current_key_normed_121_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_627_cast_fp16)[name = string("current_key_normed_121_cast_fp16")]; + tensor var_18589 = const()[name = string("op_18589"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_481_cast_fp16 = reshape(shape = var_18589, x = query_normed_121_cast_fp16)[name = string("mh_q_481_cast_fp16")]; + tensor var_18591 = const()[name = string("op_18591"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_481_cast_fp16 = reshape(shape = var_18591, x = current_key_normed_121_cast_fp16)[name = string("mh_k_481_cast_fp16")]; + tensor cos_121_to_fp16 = const()[name = string("cos_121_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175200832)))]; + tensor var_18595_cast_fp16 = mul(x = mh_q_481_cast_fp16, y = cos_121_to_fp16)[name = string("op_18595_cast_fp16")]; + tensor var_18600_begin_0 = const()[name = string("op_18600_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18600_end_0 = const()[name = string("op_18600_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_18600_end_mask_0 = const()[name = string("op_18600_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_18600_cast_fp16 = slice_by_index(begin = var_18600_begin_0, end = var_18600_end_0, end_mask = var_18600_end_mask_0, x = mh_q_481_cast_fp16)[name = string("op_18600_cast_fp16")]; + tensor var_18606_begin_0 = const()[name = string("op_18606_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_18606_end_0 = const()[name = string("op_18606_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_18606_end_mask_0 = const()[name = string("op_18606_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_18606_cast_fp16 = slice_by_index(begin = var_18606_begin_0, end = var_18606_end_0, end_mask = var_18606_end_mask_0, x = mh_q_481_cast_fp16)[name = string("op_18606_cast_fp16")]; + fp16 const_1226_promoted_to_fp16 = const()[name = string("const_1226_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18608_cast_fp16 = mul(x = var_18606_cast_fp16, y = const_1226_promoted_to_fp16)[name = string("op_18608_cast_fp16")]; + bool var_18610_interleave_0 = const()[name = string("op_18610_interleave_0"), val = bool(false)]; + tensor var_18610_cast_fp16 = concat(axis = var_18494, interleave = var_18610_interleave_0, values = (var_18608_cast_fp16, var_18600_cast_fp16))[name = string("op_18610_cast_fp16")]; + tensor sin_121_to_fp16 = const()[name = string("sin_121_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201152)))]; + tensor var_18611_cast_fp16 = mul(x = var_18610_cast_fp16, y = sin_121_to_fp16)[name = string("op_18611_cast_fp16")]; + tensor mh_q_483_cast_fp16 = add(x = var_18595_cast_fp16, y = var_18611_cast_fp16)[name = string("mh_q_483_cast_fp16")]; + tensor var_18613_cast_fp16 = mul(x = mh_k_481_cast_fp16, y = cos_121_to_fp16)[name = string("op_18613_cast_fp16")]; + tensor var_18618_begin_0 = const()[name = string("op_18618_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18618_end_0 = const()[name = string("op_18618_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_18618_end_mask_0 = const()[name = string("op_18618_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_18618_cast_fp16 = slice_by_index(begin = var_18618_begin_0, end = var_18618_end_0, end_mask = var_18618_end_mask_0, x = mh_k_481_cast_fp16)[name = string("op_18618_cast_fp16")]; + tensor var_18624_begin_0 = const()[name = string("op_18624_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_18624_end_0 = const()[name = string("op_18624_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_18624_end_mask_0 = const()[name = string("op_18624_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_18624_cast_fp16 = slice_by_index(begin = var_18624_begin_0, end = var_18624_end_0, end_mask = var_18624_end_mask_0, x = mh_k_481_cast_fp16)[name = string("op_18624_cast_fp16")]; + fp16 const_1229_promoted_to_fp16 = const()[name = string("const_1229_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18626_cast_fp16 = mul(x = var_18624_cast_fp16, y = const_1229_promoted_to_fp16)[name = string("op_18626_cast_fp16")]; + bool var_18628_interleave_0 = const()[name = string("op_18628_interleave_0"), val = bool(false)]; + tensor var_18628_cast_fp16 = concat(axis = var_18494, interleave = var_18628_interleave_0, values = (var_18626_cast_fp16, var_18618_cast_fp16))[name = string("op_18628_cast_fp16")]; + tensor var_18629_cast_fp16 = mul(x = var_18628_cast_fp16, y = sin_121_to_fp16)[name = string("op_18629_cast_fp16")]; + tensor mh_k_483_cast_fp16 = add(x = var_18613_cast_fp16, y = var_18629_cast_fp16)[name = string("mh_k_483_cast_fp16")]; + tensor var_18633 = const()[name = string("op_18633"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_243_cast_fp16 = reshape(shape = var_18633, x = mh_k_483_cast_fp16)[name = string("current_key_243_cast_fp16")]; + tensor var_18639_to_fp16 = const()[name = string("op_18639_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201472)))]; + tensor var_18640_cast_fp16 = mul(x = obj_531_cast_fp16, y = var_18639_to_fp16)[name = string("op_18640_cast_fp16")]; + tensor var_18637_to_fp16 = const()[name = string("op_18637_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201600)))]; + tensor var_18641_cast_fp16 = mul(x = current_key_243_cast_fp16, y = var_18637_to_fp16)[name = string("op_18641_cast_fp16")]; + tensor key_243_cast_fp16 = add(x = var_18640_cast_fp16, y = var_18641_cast_fp16)[name = string("key_243_cast_fp16")]; + tensor var_18643_to_fp16 = const()[name = string("op_18643_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201472)))]; + tensor var_18644_cast_fp16 = mul(x = obj_533_cast_fp16, y = var_18643_to_fp16)[name = string("op_18644_cast_fp16")]; + tensor var_18645_cast_fp16 = mul(x = current_value_121_cast_fp16, y = var_18637_to_fp16)[name = string("op_18645_cast_fp16")]; + tensor value_121_cast_fp16 = add(x = var_18644_cast_fp16, y = var_18645_cast_fp16)[name = string("value_121_cast_fp16")]; + fp16 var_18652_to_fp16 = const()[name = string("op_18652_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_487_cast_fp16 = mul(x = mh_q_483_cast_fp16, y = var_18652_to_fp16)[name = string("mh_q_487_cast_fp16")]; + tensor var_18654 = const()[name = string("op_18654"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_485_cast_fp16 = reshape(shape = var_18654, x = key_243_cast_fp16)[name = string("mh_k_485_cast_fp16")]; + tensor var_18656 = const()[name = string("op_18656"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_241_cast_fp16 = reshape(shape = var_18656, x = value_121_cast_fp16)[name = string("mh_v_241_cast_fp16")]; + tensor transpose_240_perm_0 = const()[name = string("transpose_240_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_120_reps_0 = const()[name = string("tile_120_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_240_cast_fp16 = transpose(perm = transpose_240_perm_0, x = mh_k_485_cast_fp16)[name = string("transpose_119")]; + tensor tile_120_cast_fp16 = tile(reps = tile_120_reps_0, x = transpose_240_cast_fp16)[name = string("tile_120_cast_fp16")]; + tensor concat_303 = const()[name = string("concat_303"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_240_cast_fp16 = reshape(shape = concat_303, x = tile_120_cast_fp16)[name = string("reshape_240_cast_fp16")]; + tensor transpose_241_perm_0 = const()[name = string("transpose_241_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_304 = const()[name = string("concat_304"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_241_cast_fp16 = transpose(perm = transpose_241_perm_0, x = reshape_240_cast_fp16)[name = string("transpose_118")]; + tensor reshape_241_cast_fp16 = reshape(shape = concat_304, x = transpose_241_cast_fp16)[name = string("reshape_241_cast_fp16")]; + tensor transpose_242_perm_0 = const()[name = string("transpose_242_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_121_reps_0 = const()[name = string("tile_121_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_242_cast_fp16 = transpose(perm = transpose_242_perm_0, x = mh_v_241_cast_fp16)[name = string("transpose_117")]; + tensor tile_121_cast_fp16 = tile(reps = tile_121_reps_0, x = transpose_242_cast_fp16)[name = string("tile_121_cast_fp16")]; + tensor concat_305 = const()[name = string("concat_305"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_242_cast_fp16 = reshape(shape = concat_305, x = tile_121_cast_fp16)[name = string("reshape_242_cast_fp16")]; + tensor transpose_243_perm_0 = const()[name = string("transpose_243_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_306 = const()[name = string("concat_306"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_243_cast_fp16 = transpose(perm = transpose_243_perm_0, x = reshape_242_cast_fp16)[name = string("transpose_116")]; + tensor reshape_243_cast_fp16 = reshape(shape = concat_306, x = transpose_243_cast_fp16)[name = string("reshape_243_cast_fp16")]; + tensor transpose_557_perm_0 = const()[name = string("transpose_557_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_361_transpose_x_1 = const()[name = string("mh_w_361_transpose_x_1"), val = bool(true)]; + bool mh_w_361_transpose_y_1 = const()[name = string("mh_w_361_transpose_y_1"), val = bool(false)]; + tensor transpose_557_cast_fp16 = transpose(perm = transpose_557_perm_0, x = reshape_241_cast_fp16)[name = string("transpose_115")]; + tensor mh_w_361_cast_fp16 = matmul(transpose_x = mh_w_361_transpose_x_1, transpose_y = mh_w_361_transpose_y_1, x = mh_q_487_cast_fp16, y = transpose_557_cast_fp16)[name = string("mh_w_361_cast_fp16")]; + tensor var_18664_to_fp16 = const()[name = string("op_18664_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201728)))]; + tensor mh_w_363_cast_fp16 = add(x = mh_w_361_cast_fp16, y = var_18664_to_fp16)[name = string("mh_w_363_cast_fp16")]; + tensor mh_w_365_cast_fp16 = softmax(axis = var_18484, x = mh_w_363_cast_fp16)[name = string("mh_w_365_cast_fp16")]; + tensor transpose_558_perm_0 = const()[name = string("transpose_558_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_121_transpose_x_1 = const()[name = string("attn_121_transpose_x_1"), val = bool(false)]; + bool attn_121_transpose_y_1 = const()[name = string("attn_121_transpose_y_1"), val = bool(true)]; + tensor transpose_558_cast_fp16 = transpose(perm = transpose_558_perm_0, x = reshape_243_cast_fp16)[name = string("transpose_114")]; + tensor attn_121_cast_fp16 = matmul(transpose_x = attn_121_transpose_x_1, transpose_y = attn_121_transpose_y_1, x = transpose_558_cast_fp16, y = mh_w_365_cast_fp16)[name = string("attn_121_cast_fp16")]; + tensor var_18670 = const()[name = string("op_18670"), val = tensor([1, 2048, 1, 1])]; + tensor input_525_cast_fp16 = reshape(shape = var_18670, x = attn_121_cast_fp16)[name = string("input_525_cast_fp16")]; + string obj_539_pad_type_0 = const()[name = string("obj_539_pad_type_0"), val = string("valid")]; + tensor obj_539_strides_0 = const()[name = string("obj_539_strides_0"), val = tensor([1, 1])]; + tensor obj_539_pad_0 = const()[name = string("obj_539_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_539_dilations_0 = const()[name = string("obj_539_dilations_0"), val = tensor([1, 1])]; + int32 obj_539_groups_0 = const()[name = string("obj_539_groups_0"), val = int32(1)]; + tensor obj_539_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_539_dilations_0, groups = obj_539_groups_0, pad = obj_539_pad_0, pad_type = obj_539_pad_type_0, strides = obj_539_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_525_cast_fp16)[name = string("obj_539_cast_fp16")]; + tensor inputs_509_cast_fp16 = add(x = inputs_503_cast_fp16, y = obj_539_cast_fp16)[name = string("inputs_509_cast_fp16")]; + tensor inputs_sq_509_cast_fp16 = mul(x = inputs_509_cast_fp16, y = inputs_509_cast_fp16)[name = string("inputs_sq_509_cast_fp16")]; + tensor variance_509_axes_0 = const()[name = string("variance_509_axes_0"), val = tensor([1])]; + bool variance_509_keep_dims_0 = const()[name = string("variance_509_keep_dims_0"), val = bool(true)]; + tensor variance_509_cast_fp16 = reduce_mean(axes = variance_509_axes_0, keep_dims = variance_509_keep_dims_0, x = inputs_sq_509_cast_fp16)[name = string("variance_509_cast_fp16")]; + fp16 var_18688_to_fp16 = const()[name = string("op_18688_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18689_cast_fp16 = add(x = variance_509_cast_fp16, y = var_18688_to_fp16)[name = string("op_18689_cast_fp16")]; + fp32 var_18690_epsilon_0 = const()[name = string("op_18690_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18690_cast_fp16 = rsqrt(epsilon = var_18690_epsilon_0, x = var_18689_cast_fp16)[name = string("op_18690_cast_fp16")]; + tensor hidden_states_629_cast_fp16 = mul(x = inputs_509_cast_fp16, y = var_18690_cast_fp16)[name = string("hidden_states_629_cast_fp16")]; + tensor input_527_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_629_cast_fp16)[name = string("input_527_cast_fp16")]; + string input_529_pad_type_0 = const()[name = string("input_529_pad_type_0"), val = string("valid")]; + tensor input_529_strides_0 = const()[name = string("input_529_strides_0"), val = tensor([1, 1])]; + tensor input_529_pad_0 = const()[name = string("input_529_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_529_dilations_0 = const()[name = string("input_529_dilations_0"), val = tensor([1, 1])]; + int32 input_529_groups_0 = const()[name = string("input_529_groups_0"), val = int32(1)]; + tensor input_529_cast_fp16 = conv(dilations = input_529_dilations_0, groups = input_529_groups_0, pad = input_529_pad_0, pad_type = input_529_pad_type_0, strides = input_529_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_527_cast_fp16)[name = string("input_529_cast_fp16")]; + tensor var_18704_cast_fp16 = silu(x = input_529_cast_fp16)[name = string("op_18704_cast_fp16")]; + string var_18710_pad_type_0 = const()[name = string("op_18710_pad_type_0"), val = string("valid")]; + tensor var_18710_strides_0 = const()[name = string("op_18710_strides_0"), val = tensor([1, 1])]; + tensor var_18710_pad_0 = const()[name = string("op_18710_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18710_dilations_0 = const()[name = string("op_18710_dilations_0"), val = tensor([1, 1])]; + int32 var_18710_groups_0 = const()[name = string("op_18710_groups_0"), val = int32(1)]; + tensor var_18710_cast_fp16 = conv(dilations = var_18710_dilations_0, groups = var_18710_groups_0, pad = var_18710_pad_0, pad_type = var_18710_pad_type_0, strides = var_18710_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_527_cast_fp16)[name = string("op_18710_cast_fp16")]; + tensor input_531_cast_fp16 = mul(x = var_18704_cast_fp16, y = var_18710_cast_fp16)[name = string("input_531_cast_fp16")]; + string hidden_states_631_pad_type_0 = const()[name = string("hidden_states_631_pad_type_0"), val = string("valid")]; + tensor hidden_states_631_strides_0 = const()[name = string("hidden_states_631_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_631_pad_0 = const()[name = string("hidden_states_631_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_631_dilations_0 = const()[name = string("hidden_states_631_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_631_groups_0 = const()[name = string("hidden_states_631_groups_0"), val = int32(1)]; + tensor hidden_states_631_cast_fp16 = conv(dilations = hidden_states_631_dilations_0, groups = hidden_states_631_groups_0, pad = hidden_states_631_pad_0, pad_type = hidden_states_631_pad_type_0, strides = hidden_states_631_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_531_cast_fp16)[name = string("hidden_states_631_cast_fp16")]; + tensor inputs_511_cast_fp16 = add(x = inputs_509_cast_fp16, y = hidden_states_631_cast_fp16)[name = string("inputs_511_cast_fp16")]; + tensor obj_543_begin_0 = const()[name = string("obj_543_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_543_end_0 = const()[name = string("obj_543_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_543_end_mask_0 = const()[name = string("obj_543_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_543_cast_fp16 = slice_by_index(begin = obj_543_begin_0, end = obj_543_end_0, end_mask = obj_543_end_mask_0, x = key_caches_25_cast_fp16)[name = string("obj_543_cast_fp16")]; + tensor obj_545_begin_0 = const()[name = string("obj_545_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_545_end_0 = const()[name = string("obj_545_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_545_end_mask_0 = const()[name = string("obj_545_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_545_cast_fp16 = slice_by_index(begin = obj_545_begin_0, end = obj_545_end_0, end_mask = obj_545_end_mask_0, x = value_caches_25_cast_fp16)[name = string("obj_545_cast_fp16")]; + int32 var_18758 = const()[name = string("op_18758"), val = int32(3)]; + int32 var_18768 = const()[name = string("op_18768"), val = int32(-2)]; + tensor inputs_sq_511_cast_fp16 = mul(x = inputs_511_cast_fp16, y = inputs_511_cast_fp16)[name = string("inputs_sq_511_cast_fp16")]; + tensor variance_511_axes_0 = const()[name = string("variance_511_axes_0"), val = tensor([1])]; + bool variance_511_keep_dims_0 = const()[name = string("variance_511_keep_dims_0"), val = bool(true)]; + tensor variance_511_cast_fp16 = reduce_mean(axes = variance_511_axes_0, keep_dims = variance_511_keep_dims_0, x = inputs_sq_511_cast_fp16)[name = string("variance_511_cast_fp16")]; + fp16 var_18782_to_fp16 = const()[name = string("op_18782_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18783_cast_fp16 = add(x = variance_511_cast_fp16, y = var_18782_to_fp16)[name = string("op_18783_cast_fp16")]; + fp32 var_18784_epsilon_0 = const()[name = string("op_18784_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18784_cast_fp16 = rsqrt(epsilon = var_18784_epsilon_0, x = var_18783_cast_fp16)[name = string("op_18784_cast_fp16")]; + tensor hidden_states_633_cast_fp16 = mul(x = inputs_511_cast_fp16, y = var_18784_cast_fp16)[name = string("hidden_states_633_cast_fp16")]; + tensor obj_541_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_633_cast_fp16)[name = string("obj_541_cast_fp16")]; + string query_367_pad_type_0 = const()[name = string("query_367_pad_type_0"), val = string("valid")]; + tensor query_367_strides_0 = const()[name = string("query_367_strides_0"), val = tensor([1, 1])]; + tensor query_367_pad_0 = const()[name = string("query_367_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_367_dilations_0 = const()[name = string("query_367_dilations_0"), val = tensor([1, 1])]; + int32 query_367_groups_0 = const()[name = string("query_367_groups_0"), val = int32(1)]; + tensor query_367_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_367_dilations_0, groups = query_367_groups_0, pad = query_367_pad_0, pad_type = query_367_pad_type_0, strides = query_367_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_541_cast_fp16)[name = string("query_367_cast_fp16")]; + string current_key_245_pad_type_0 = const()[name = string("current_key_245_pad_type_0"), val = string("valid")]; + tensor current_key_245_strides_0 = const()[name = string("current_key_245_strides_0"), val = tensor([1, 1])]; + tensor current_key_245_pad_0 = const()[name = string("current_key_245_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_245_dilations_0 = const()[name = string("current_key_245_dilations_0"), val = tensor([1, 1])]; + int32 current_key_245_groups_0 = const()[name = string("current_key_245_groups_0"), val = int32(1)]; + tensor current_key_245_cast_fp16 = conv(dilations = current_key_245_dilations_0, groups = current_key_245_groups_0, pad = current_key_245_pad_0, pad_type = current_key_245_pad_type_0, strides = current_key_245_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_541_cast_fp16)[name = string("current_key_245_cast_fp16")]; + string current_value_123_pad_type_0 = const()[name = string("current_value_123_pad_type_0"), val = string("valid")]; + tensor current_value_123_strides_0 = const()[name = string("current_value_123_strides_0"), val = tensor([1, 1])]; + tensor current_value_123_pad_0 = const()[name = string("current_value_123_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_123_dilations_0 = const()[name = string("current_value_123_dilations_0"), val = tensor([1, 1])]; + int32 current_value_123_groups_0 = const()[name = string("current_value_123_groups_0"), val = int32(1)]; + tensor current_value_123_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_123_dilations_0, groups = current_value_123_groups_0, pad = current_value_123_pad_0, pad_type = current_value_123_pad_type_0, strides = current_value_123_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_541_cast_fp16)[name = string("current_value_123_cast_fp16")]; + tensor var_18821 = const()[name = string("op_18821"), val = tensor([16, 128, 1, 1])]; + tensor inputs_513_cast_fp16 = reshape(shape = var_18821, x = query_367_cast_fp16)[name = string("inputs_513_cast_fp16")]; + tensor inputs_sq_513_cast_fp16 = mul(x = inputs_513_cast_fp16, y = inputs_513_cast_fp16)[name = string("inputs_sq_513_cast_fp16")]; + tensor variance_513_axes_0 = const()[name = string("variance_513_axes_0"), val = tensor([1])]; + bool variance_513_keep_dims_0 = const()[name = string("variance_513_keep_dims_0"), val = bool(true)]; + tensor variance_513_cast_fp16 = reduce_mean(axes = variance_513_axes_0, keep_dims = variance_513_keep_dims_0, x = inputs_sq_513_cast_fp16)[name = string("variance_513_cast_fp16")]; + fp16 var_18827_to_fp16 = const()[name = string("op_18827_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18828_cast_fp16 = add(x = variance_513_cast_fp16, y = var_18827_to_fp16)[name = string("op_18828_cast_fp16")]; + fp32 var_18829_epsilon_0 = const()[name = string("op_18829_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18829_cast_fp16 = rsqrt(epsilon = var_18829_epsilon_0, x = var_18828_cast_fp16)[name = string("op_18829_cast_fp16")]; + tensor hidden_states_635_cast_fp16 = mul(x = inputs_513_cast_fp16, y = var_18829_cast_fp16)[name = string("hidden_states_635_cast_fp16")]; + tensor query_normed_123_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_635_cast_fp16)[name = string("query_normed_123_cast_fp16")]; + tensor var_18837 = const()[name = string("op_18837"), val = tensor([8, 128, 1, 1])]; + tensor inputs_515_cast_fp16 = reshape(shape = var_18837, x = current_key_245_cast_fp16)[name = string("inputs_515_cast_fp16")]; + tensor inputs_sq_515_cast_fp16 = mul(x = inputs_515_cast_fp16, y = inputs_515_cast_fp16)[name = string("inputs_sq_515_cast_fp16")]; + tensor variance_515_axes_0 = const()[name = string("variance_515_axes_0"), val = tensor([1])]; + bool variance_515_keep_dims_0 = const()[name = string("variance_515_keep_dims_0"), val = bool(true)]; + tensor variance_515_cast_fp16 = reduce_mean(axes = variance_515_axes_0, keep_dims = variance_515_keep_dims_0, x = inputs_sq_515_cast_fp16)[name = string("variance_515_cast_fp16")]; + fp16 var_18843_to_fp16 = const()[name = string("op_18843_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18844_cast_fp16 = add(x = variance_515_cast_fp16, y = var_18843_to_fp16)[name = string("op_18844_cast_fp16")]; + fp32 var_18845_epsilon_0 = const()[name = string("op_18845_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18845_cast_fp16 = rsqrt(epsilon = var_18845_epsilon_0, x = var_18844_cast_fp16)[name = string("op_18845_cast_fp16")]; + tensor hidden_states_637_cast_fp16 = mul(x = inputs_515_cast_fp16, y = var_18845_cast_fp16)[name = string("hidden_states_637_cast_fp16")]; + tensor current_key_normed_123_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_637_cast_fp16)[name = string("current_key_normed_123_cast_fp16")]; + tensor var_18863 = const()[name = string("op_18863"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_489_cast_fp16 = reshape(shape = var_18863, x = query_normed_123_cast_fp16)[name = string("mh_q_489_cast_fp16")]; + tensor var_18865 = const()[name = string("op_18865"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_489_cast_fp16 = reshape(shape = var_18865, x = current_key_normed_123_cast_fp16)[name = string("mh_k_489_cast_fp16")]; + tensor var_18869_cast_fp16 = mul(x = mh_q_489_cast_fp16, y = cos_121_to_fp16)[name = string("op_18869_cast_fp16")]; + tensor var_18874_begin_0 = const()[name = string("op_18874_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18874_end_0 = const()[name = string("op_18874_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_18874_end_mask_0 = const()[name = string("op_18874_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_18874_cast_fp16 = slice_by_index(begin = var_18874_begin_0, end = var_18874_end_0, end_mask = var_18874_end_mask_0, x = mh_q_489_cast_fp16)[name = string("op_18874_cast_fp16")]; + tensor var_18880_begin_0 = const()[name = string("op_18880_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_18880_end_0 = const()[name = string("op_18880_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_18880_end_mask_0 = const()[name = string("op_18880_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_18880_cast_fp16 = slice_by_index(begin = var_18880_begin_0, end = var_18880_end_0, end_mask = var_18880_end_mask_0, x = mh_q_489_cast_fp16)[name = string("op_18880_cast_fp16")]; + fp16 const_1246_promoted_to_fp16 = const()[name = string("const_1246_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18882_cast_fp16 = mul(x = var_18880_cast_fp16, y = const_1246_promoted_to_fp16)[name = string("op_18882_cast_fp16")]; + bool var_18884_interleave_0 = const()[name = string("op_18884_interleave_0"), val = bool(false)]; + tensor var_18884_cast_fp16 = concat(axis = var_18768, interleave = var_18884_interleave_0, values = (var_18882_cast_fp16, var_18874_cast_fp16))[name = string("op_18884_cast_fp16")]; + tensor var_18885_cast_fp16 = mul(x = var_18884_cast_fp16, y = sin_121_to_fp16)[name = string("op_18885_cast_fp16")]; + tensor mh_q_491_cast_fp16 = add(x = var_18869_cast_fp16, y = var_18885_cast_fp16)[name = string("mh_q_491_cast_fp16")]; + tensor var_18887_cast_fp16 = mul(x = mh_k_489_cast_fp16, y = cos_121_to_fp16)[name = string("op_18887_cast_fp16")]; + tensor var_18892_begin_0 = const()[name = string("op_18892_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18892_end_0 = const()[name = string("op_18892_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_18892_end_mask_0 = const()[name = string("op_18892_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_18892_cast_fp16 = slice_by_index(begin = var_18892_begin_0, end = var_18892_end_0, end_mask = var_18892_end_mask_0, x = mh_k_489_cast_fp16)[name = string("op_18892_cast_fp16")]; + tensor var_18898_begin_0 = const()[name = string("op_18898_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_18898_end_0 = const()[name = string("op_18898_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_18898_end_mask_0 = const()[name = string("op_18898_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_18898_cast_fp16 = slice_by_index(begin = var_18898_begin_0, end = var_18898_end_0, end_mask = var_18898_end_mask_0, x = mh_k_489_cast_fp16)[name = string("op_18898_cast_fp16")]; + fp16 const_1249_promoted_to_fp16 = const()[name = string("const_1249_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_18900_cast_fp16 = mul(x = var_18898_cast_fp16, y = const_1249_promoted_to_fp16)[name = string("op_18900_cast_fp16")]; + bool var_18902_interleave_0 = const()[name = string("op_18902_interleave_0"), val = bool(false)]; + tensor var_18902_cast_fp16 = concat(axis = var_18768, interleave = var_18902_interleave_0, values = (var_18900_cast_fp16, var_18892_cast_fp16))[name = string("op_18902_cast_fp16")]; + tensor var_18903_cast_fp16 = mul(x = var_18902_cast_fp16, y = sin_121_to_fp16)[name = string("op_18903_cast_fp16")]; + tensor mh_k_491_cast_fp16 = add(x = var_18887_cast_fp16, y = var_18903_cast_fp16)[name = string("mh_k_491_cast_fp16")]; + tensor var_18907 = const()[name = string("op_18907"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_247_cast_fp16 = reshape(shape = var_18907, x = mh_k_491_cast_fp16)[name = string("current_key_247_cast_fp16")]; + tensor var_18913_to_fp16 = const()[name = string("op_18913_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201472)))]; + tensor var_18914_cast_fp16 = mul(x = obj_543_cast_fp16, y = var_18913_to_fp16)[name = string("op_18914_cast_fp16")]; + tensor var_18911_to_fp16 = const()[name = string("op_18911_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201600)))]; + tensor var_18915_cast_fp16 = mul(x = current_key_247_cast_fp16, y = var_18911_to_fp16)[name = string("op_18915_cast_fp16")]; + tensor key_247_cast_fp16 = add(x = var_18914_cast_fp16, y = var_18915_cast_fp16)[name = string("key_247_cast_fp16")]; + tensor var_18917_to_fp16 = const()[name = string("op_18917_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201472)))]; + tensor var_18918_cast_fp16 = mul(x = obj_545_cast_fp16, y = var_18917_to_fp16)[name = string("op_18918_cast_fp16")]; + tensor var_18919_cast_fp16 = mul(x = current_value_123_cast_fp16, y = var_18911_to_fp16)[name = string("op_18919_cast_fp16")]; + tensor value_123_cast_fp16 = add(x = var_18918_cast_fp16, y = var_18919_cast_fp16)[name = string("value_123_cast_fp16")]; + fp16 var_18926_to_fp16 = const()[name = string("op_18926_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_495_cast_fp16 = mul(x = mh_q_491_cast_fp16, y = var_18926_to_fp16)[name = string("mh_q_495_cast_fp16")]; + tensor var_18928 = const()[name = string("op_18928"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_493_cast_fp16 = reshape(shape = var_18928, x = key_247_cast_fp16)[name = string("mh_k_493_cast_fp16")]; + tensor var_18930 = const()[name = string("op_18930"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_245_cast_fp16 = reshape(shape = var_18930, x = value_123_cast_fp16)[name = string("mh_v_245_cast_fp16")]; + tensor transpose_244_perm_0 = const()[name = string("transpose_244_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_122_reps_0 = const()[name = string("tile_122_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_244_cast_fp16 = transpose(perm = transpose_244_perm_0, x = mh_k_493_cast_fp16)[name = string("transpose_113")]; + tensor tile_122_cast_fp16 = tile(reps = tile_122_reps_0, x = transpose_244_cast_fp16)[name = string("tile_122_cast_fp16")]; + tensor concat_307 = const()[name = string("concat_307"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_244_cast_fp16 = reshape(shape = concat_307, x = tile_122_cast_fp16)[name = string("reshape_244_cast_fp16")]; + tensor transpose_245_perm_0 = const()[name = string("transpose_245_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_308 = const()[name = string("concat_308"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_245_cast_fp16 = transpose(perm = transpose_245_perm_0, x = reshape_244_cast_fp16)[name = string("transpose_112")]; + tensor reshape_245_cast_fp16 = reshape(shape = concat_308, x = transpose_245_cast_fp16)[name = string("reshape_245_cast_fp16")]; + tensor transpose_246_perm_0 = const()[name = string("transpose_246_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_123_reps_0 = const()[name = string("tile_123_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_246_cast_fp16 = transpose(perm = transpose_246_perm_0, x = mh_v_245_cast_fp16)[name = string("transpose_111")]; + tensor tile_123_cast_fp16 = tile(reps = tile_123_reps_0, x = transpose_246_cast_fp16)[name = string("tile_123_cast_fp16")]; + tensor concat_309 = const()[name = string("concat_309"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_246_cast_fp16 = reshape(shape = concat_309, x = tile_123_cast_fp16)[name = string("reshape_246_cast_fp16")]; + tensor transpose_247_perm_0 = const()[name = string("transpose_247_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_310 = const()[name = string("concat_310"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_247_cast_fp16 = transpose(perm = transpose_247_perm_0, x = reshape_246_cast_fp16)[name = string("transpose_110")]; + tensor reshape_247_cast_fp16 = reshape(shape = concat_310, x = transpose_247_cast_fp16)[name = string("reshape_247_cast_fp16")]; + tensor transpose_561_perm_0 = const()[name = string("transpose_561_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_367_transpose_x_1 = const()[name = string("mh_w_367_transpose_x_1"), val = bool(true)]; + bool mh_w_367_transpose_y_1 = const()[name = string("mh_w_367_transpose_y_1"), val = bool(false)]; + tensor transpose_561_cast_fp16 = transpose(perm = transpose_561_perm_0, x = reshape_245_cast_fp16)[name = string("transpose_109")]; + tensor mh_w_367_cast_fp16 = matmul(transpose_x = mh_w_367_transpose_x_1, transpose_y = mh_w_367_transpose_y_1, x = mh_q_495_cast_fp16, y = transpose_561_cast_fp16)[name = string("mh_w_367_cast_fp16")]; + tensor var_18938_to_fp16 = const()[name = string("op_18938_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201728)))]; + tensor mh_w_369_cast_fp16 = add(x = mh_w_367_cast_fp16, y = var_18938_to_fp16)[name = string("mh_w_369_cast_fp16")]; + tensor mh_w_371_cast_fp16 = softmax(axis = var_18758, x = mh_w_369_cast_fp16)[name = string("mh_w_371_cast_fp16")]; + tensor transpose_562_perm_0 = const()[name = string("transpose_562_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_123_transpose_x_1 = const()[name = string("attn_123_transpose_x_1"), val = bool(false)]; + bool attn_123_transpose_y_1 = const()[name = string("attn_123_transpose_y_1"), val = bool(true)]; + tensor transpose_562_cast_fp16 = transpose(perm = transpose_562_perm_0, x = reshape_247_cast_fp16)[name = string("transpose_108")]; + tensor attn_123_cast_fp16 = matmul(transpose_x = attn_123_transpose_x_1, transpose_y = attn_123_transpose_y_1, x = transpose_562_cast_fp16, y = mh_w_371_cast_fp16)[name = string("attn_123_cast_fp16")]; + tensor var_18944 = const()[name = string("op_18944"), val = tensor([1, 2048, 1, 1])]; + tensor input_533_cast_fp16 = reshape(shape = var_18944, x = attn_123_cast_fp16)[name = string("input_533_cast_fp16")]; + string obj_547_pad_type_0 = const()[name = string("obj_547_pad_type_0"), val = string("valid")]; + tensor obj_547_strides_0 = const()[name = string("obj_547_strides_0"), val = tensor([1, 1])]; + tensor obj_547_pad_0 = const()[name = string("obj_547_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_547_dilations_0 = const()[name = string("obj_547_dilations_0"), val = tensor([1, 1])]; + int32 obj_547_groups_0 = const()[name = string("obj_547_groups_0"), val = int32(1)]; + tensor obj_547_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_547_dilations_0, groups = obj_547_groups_0, pad = obj_547_pad_0, pad_type = obj_547_pad_type_0, strides = obj_547_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_533_cast_fp16)[name = string("obj_547_cast_fp16")]; + tensor inputs_517_cast_fp16 = add(x = inputs_511_cast_fp16, y = obj_547_cast_fp16)[name = string("inputs_517_cast_fp16")]; + tensor inputs_sq_517_cast_fp16 = mul(x = inputs_517_cast_fp16, y = inputs_517_cast_fp16)[name = string("inputs_sq_517_cast_fp16")]; + tensor variance_517_axes_0 = const()[name = string("variance_517_axes_0"), val = tensor([1])]; + bool variance_517_keep_dims_0 = const()[name = string("variance_517_keep_dims_0"), val = bool(true)]; + tensor variance_517_cast_fp16 = reduce_mean(axes = variance_517_axes_0, keep_dims = variance_517_keep_dims_0, x = inputs_sq_517_cast_fp16)[name = string("variance_517_cast_fp16")]; + fp16 var_18962_to_fp16 = const()[name = string("op_18962_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_18963_cast_fp16 = add(x = variance_517_cast_fp16, y = var_18962_to_fp16)[name = string("op_18963_cast_fp16")]; + fp32 var_18964_epsilon_0 = const()[name = string("op_18964_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_18964_cast_fp16 = rsqrt(epsilon = var_18964_epsilon_0, x = var_18963_cast_fp16)[name = string("op_18964_cast_fp16")]; + tensor hidden_states_639_cast_fp16 = mul(x = inputs_517_cast_fp16, y = var_18964_cast_fp16)[name = string("hidden_states_639_cast_fp16")]; + tensor input_535_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_639_cast_fp16)[name = string("input_535_cast_fp16")]; + string input_537_pad_type_0 = const()[name = string("input_537_pad_type_0"), val = string("valid")]; + tensor input_537_strides_0 = const()[name = string("input_537_strides_0"), val = tensor([1, 1])]; + tensor input_537_pad_0 = const()[name = string("input_537_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_537_dilations_0 = const()[name = string("input_537_dilations_0"), val = tensor([1, 1])]; + int32 input_537_groups_0 = const()[name = string("input_537_groups_0"), val = int32(1)]; + tensor input_537_cast_fp16 = conv(dilations = input_537_dilations_0, groups = input_537_groups_0, pad = input_537_pad_0, pad_type = input_537_pad_type_0, strides = input_537_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_535_cast_fp16)[name = string("input_537_cast_fp16")]; + tensor var_18978_cast_fp16 = silu(x = input_537_cast_fp16)[name = string("op_18978_cast_fp16")]; + string var_18984_pad_type_0 = const()[name = string("op_18984_pad_type_0"), val = string("valid")]; + tensor var_18984_strides_0 = const()[name = string("op_18984_strides_0"), val = tensor([1, 1])]; + tensor var_18984_pad_0 = const()[name = string("op_18984_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_18984_dilations_0 = const()[name = string("op_18984_dilations_0"), val = tensor([1, 1])]; + int32 var_18984_groups_0 = const()[name = string("op_18984_groups_0"), val = int32(1)]; + tensor var_18984_cast_fp16 = conv(dilations = var_18984_dilations_0, groups = var_18984_groups_0, pad = var_18984_pad_0, pad_type = var_18984_pad_type_0, strides = var_18984_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_535_cast_fp16)[name = string("op_18984_cast_fp16")]; + tensor input_539_cast_fp16 = mul(x = var_18978_cast_fp16, y = var_18984_cast_fp16)[name = string("input_539_cast_fp16")]; + string hidden_states_641_pad_type_0 = const()[name = string("hidden_states_641_pad_type_0"), val = string("valid")]; + tensor hidden_states_641_strides_0 = const()[name = string("hidden_states_641_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_641_pad_0 = const()[name = string("hidden_states_641_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_641_dilations_0 = const()[name = string("hidden_states_641_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_641_groups_0 = const()[name = string("hidden_states_641_groups_0"), val = int32(1)]; + tensor hidden_states_641_cast_fp16 = conv(dilations = hidden_states_641_dilations_0, groups = hidden_states_641_groups_0, pad = hidden_states_641_pad_0, pad_type = hidden_states_641_pad_type_0, strides = hidden_states_641_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_539_cast_fp16)[name = string("hidden_states_641_cast_fp16")]; + tensor inputs_519_cast_fp16 = add(x = inputs_517_cast_fp16, y = hidden_states_641_cast_fp16)[name = string("inputs_519_cast_fp16")]; + tensor obj_551_begin_0 = const()[name = string("obj_551_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_551_end_0 = const()[name = string("obj_551_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_551_end_mask_0 = const()[name = string("obj_551_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_551_cast_fp16 = slice_by_index(begin = obj_551_begin_0, end = obj_551_end_0, end_mask = obj_551_end_mask_0, x = key_caches_25_cast_fp16)[name = string("obj_551_cast_fp16")]; + tensor obj_553_begin_0 = const()[name = string("obj_553_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_553_end_0 = const()[name = string("obj_553_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_553_end_mask_0 = const()[name = string("obj_553_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_553_cast_fp16 = slice_by_index(begin = obj_553_begin_0, end = obj_553_end_0, end_mask = obj_553_end_mask_0, x = value_caches_25_cast_fp16)[name = string("obj_553_cast_fp16")]; + int32 var_19032 = const()[name = string("op_19032"), val = int32(3)]; + int32 var_19042 = const()[name = string("op_19042"), val = int32(-2)]; + tensor inputs_sq_519_cast_fp16 = mul(x = inputs_519_cast_fp16, y = inputs_519_cast_fp16)[name = string("inputs_sq_519_cast_fp16")]; + tensor variance_519_axes_0 = const()[name = string("variance_519_axes_0"), val = tensor([1])]; + bool variance_519_keep_dims_0 = const()[name = string("variance_519_keep_dims_0"), val = bool(true)]; + tensor variance_519_cast_fp16 = reduce_mean(axes = variance_519_axes_0, keep_dims = variance_519_keep_dims_0, x = inputs_sq_519_cast_fp16)[name = string("variance_519_cast_fp16")]; + fp16 var_19056_to_fp16 = const()[name = string("op_19056_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19057_cast_fp16 = add(x = variance_519_cast_fp16, y = var_19056_to_fp16)[name = string("op_19057_cast_fp16")]; + fp32 var_19058_epsilon_0 = const()[name = string("op_19058_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19058_cast_fp16 = rsqrt(epsilon = var_19058_epsilon_0, x = var_19057_cast_fp16)[name = string("op_19058_cast_fp16")]; + tensor hidden_states_643_cast_fp16 = mul(x = inputs_519_cast_fp16, y = var_19058_cast_fp16)[name = string("hidden_states_643_cast_fp16")]; + tensor obj_549_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_643_cast_fp16)[name = string("obj_549_cast_fp16")]; + string query_373_pad_type_0 = const()[name = string("query_373_pad_type_0"), val = string("valid")]; + tensor query_373_strides_0 = const()[name = string("query_373_strides_0"), val = tensor([1, 1])]; + tensor query_373_pad_0 = const()[name = string("query_373_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_373_dilations_0 = const()[name = string("query_373_dilations_0"), val = tensor([1, 1])]; + int32 query_373_groups_0 = const()[name = string("query_373_groups_0"), val = int32(1)]; + tensor query_373_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_373_dilations_0, groups = query_373_groups_0, pad = query_373_pad_0, pad_type = query_373_pad_type_0, strides = query_373_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_549_cast_fp16)[name = string("query_373_cast_fp16")]; + string current_key_249_pad_type_0 = const()[name = string("current_key_249_pad_type_0"), val = string("valid")]; + tensor current_key_249_strides_0 = const()[name = string("current_key_249_strides_0"), val = tensor([1, 1])]; + tensor current_key_249_pad_0 = const()[name = string("current_key_249_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_249_dilations_0 = const()[name = string("current_key_249_dilations_0"), val = tensor([1, 1])]; + int32 current_key_249_groups_0 = const()[name = string("current_key_249_groups_0"), val = int32(1)]; + tensor current_key_249_cast_fp16 = conv(dilations = current_key_249_dilations_0, groups = current_key_249_groups_0, pad = current_key_249_pad_0, pad_type = current_key_249_pad_type_0, strides = current_key_249_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_549_cast_fp16)[name = string("current_key_249_cast_fp16")]; + string current_value_125_pad_type_0 = const()[name = string("current_value_125_pad_type_0"), val = string("valid")]; + tensor current_value_125_strides_0 = const()[name = string("current_value_125_strides_0"), val = tensor([1, 1])]; + tensor current_value_125_pad_0 = const()[name = string("current_value_125_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_125_dilations_0 = const()[name = string("current_value_125_dilations_0"), val = tensor([1, 1])]; + int32 current_value_125_groups_0 = const()[name = string("current_value_125_groups_0"), val = int32(1)]; + tensor current_value_125_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_125_dilations_0, groups = current_value_125_groups_0, pad = current_value_125_pad_0, pad_type = current_value_125_pad_type_0, strides = current_value_125_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_549_cast_fp16)[name = string("current_value_125_cast_fp16")]; + tensor var_19095 = const()[name = string("op_19095"), val = tensor([16, 128, 1, 1])]; + tensor inputs_521_cast_fp16 = reshape(shape = var_19095, x = query_373_cast_fp16)[name = string("inputs_521_cast_fp16")]; + tensor inputs_sq_521_cast_fp16 = mul(x = inputs_521_cast_fp16, y = inputs_521_cast_fp16)[name = string("inputs_sq_521_cast_fp16")]; + tensor variance_521_axes_0 = const()[name = string("variance_521_axes_0"), val = tensor([1])]; + bool variance_521_keep_dims_0 = const()[name = string("variance_521_keep_dims_0"), val = bool(true)]; + tensor variance_521_cast_fp16 = reduce_mean(axes = variance_521_axes_0, keep_dims = variance_521_keep_dims_0, x = inputs_sq_521_cast_fp16)[name = string("variance_521_cast_fp16")]; + fp16 var_19101_to_fp16 = const()[name = string("op_19101_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19102_cast_fp16 = add(x = variance_521_cast_fp16, y = var_19101_to_fp16)[name = string("op_19102_cast_fp16")]; + fp32 var_19103_epsilon_0 = const()[name = string("op_19103_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19103_cast_fp16 = rsqrt(epsilon = var_19103_epsilon_0, x = var_19102_cast_fp16)[name = string("op_19103_cast_fp16")]; + tensor hidden_states_645_cast_fp16 = mul(x = inputs_521_cast_fp16, y = var_19103_cast_fp16)[name = string("hidden_states_645_cast_fp16")]; + tensor query_normed_125_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_645_cast_fp16)[name = string("query_normed_125_cast_fp16")]; + tensor var_19111 = const()[name = string("op_19111"), val = tensor([8, 128, 1, 1])]; + tensor inputs_523_cast_fp16 = reshape(shape = var_19111, x = current_key_249_cast_fp16)[name = string("inputs_523_cast_fp16")]; + tensor inputs_sq_523_cast_fp16 = mul(x = inputs_523_cast_fp16, y = inputs_523_cast_fp16)[name = string("inputs_sq_523_cast_fp16")]; + tensor variance_523_axes_0 = const()[name = string("variance_523_axes_0"), val = tensor([1])]; + bool variance_523_keep_dims_0 = const()[name = string("variance_523_keep_dims_0"), val = bool(true)]; + tensor variance_523_cast_fp16 = reduce_mean(axes = variance_523_axes_0, keep_dims = variance_523_keep_dims_0, x = inputs_sq_523_cast_fp16)[name = string("variance_523_cast_fp16")]; + fp16 var_19117_to_fp16 = const()[name = string("op_19117_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19118_cast_fp16 = add(x = variance_523_cast_fp16, y = var_19117_to_fp16)[name = string("op_19118_cast_fp16")]; + fp32 var_19119_epsilon_0 = const()[name = string("op_19119_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19119_cast_fp16 = rsqrt(epsilon = var_19119_epsilon_0, x = var_19118_cast_fp16)[name = string("op_19119_cast_fp16")]; + tensor hidden_states_647_cast_fp16 = mul(x = inputs_523_cast_fp16, y = var_19119_cast_fp16)[name = string("hidden_states_647_cast_fp16")]; + tensor current_key_normed_125_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_647_cast_fp16)[name = string("current_key_normed_125_cast_fp16")]; + tensor var_19137 = const()[name = string("op_19137"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_497_cast_fp16 = reshape(shape = var_19137, x = query_normed_125_cast_fp16)[name = string("mh_q_497_cast_fp16")]; + tensor var_19139 = const()[name = string("op_19139"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_497_cast_fp16 = reshape(shape = var_19139, x = current_key_normed_125_cast_fp16)[name = string("mh_k_497_cast_fp16")]; + tensor var_19143_cast_fp16 = mul(x = mh_q_497_cast_fp16, y = cos_121_to_fp16)[name = string("op_19143_cast_fp16")]; + tensor var_19148_begin_0 = const()[name = string("op_19148_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19148_end_0 = const()[name = string("op_19148_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_19148_end_mask_0 = const()[name = string("op_19148_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_19148_cast_fp16 = slice_by_index(begin = var_19148_begin_0, end = var_19148_end_0, end_mask = var_19148_end_mask_0, x = mh_q_497_cast_fp16)[name = string("op_19148_cast_fp16")]; + tensor var_19154_begin_0 = const()[name = string("op_19154_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_19154_end_0 = const()[name = string("op_19154_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_19154_end_mask_0 = const()[name = string("op_19154_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_19154_cast_fp16 = slice_by_index(begin = var_19154_begin_0, end = var_19154_end_0, end_mask = var_19154_end_mask_0, x = mh_q_497_cast_fp16)[name = string("op_19154_cast_fp16")]; + fp16 const_1266_promoted_to_fp16 = const()[name = string("const_1266_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19156_cast_fp16 = mul(x = var_19154_cast_fp16, y = const_1266_promoted_to_fp16)[name = string("op_19156_cast_fp16")]; + bool var_19158_interleave_0 = const()[name = string("op_19158_interleave_0"), val = bool(false)]; + tensor var_19158_cast_fp16 = concat(axis = var_19042, interleave = var_19158_interleave_0, values = (var_19156_cast_fp16, var_19148_cast_fp16))[name = string("op_19158_cast_fp16")]; + tensor var_19159_cast_fp16 = mul(x = var_19158_cast_fp16, y = sin_121_to_fp16)[name = string("op_19159_cast_fp16")]; + tensor mh_q_499_cast_fp16 = add(x = var_19143_cast_fp16, y = var_19159_cast_fp16)[name = string("mh_q_499_cast_fp16")]; + tensor var_19161_cast_fp16 = mul(x = mh_k_497_cast_fp16, y = cos_121_to_fp16)[name = string("op_19161_cast_fp16")]; + tensor var_19166_begin_0 = const()[name = string("op_19166_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19166_end_0 = const()[name = string("op_19166_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_19166_end_mask_0 = const()[name = string("op_19166_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_19166_cast_fp16 = slice_by_index(begin = var_19166_begin_0, end = var_19166_end_0, end_mask = var_19166_end_mask_0, x = mh_k_497_cast_fp16)[name = string("op_19166_cast_fp16")]; + tensor var_19172_begin_0 = const()[name = string("op_19172_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_19172_end_0 = const()[name = string("op_19172_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_19172_end_mask_0 = const()[name = string("op_19172_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_19172_cast_fp16 = slice_by_index(begin = var_19172_begin_0, end = var_19172_end_0, end_mask = var_19172_end_mask_0, x = mh_k_497_cast_fp16)[name = string("op_19172_cast_fp16")]; + fp16 const_1269_promoted_to_fp16 = const()[name = string("const_1269_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19174_cast_fp16 = mul(x = var_19172_cast_fp16, y = const_1269_promoted_to_fp16)[name = string("op_19174_cast_fp16")]; + bool var_19176_interleave_0 = const()[name = string("op_19176_interleave_0"), val = bool(false)]; + tensor var_19176_cast_fp16 = concat(axis = var_19042, interleave = var_19176_interleave_0, values = (var_19174_cast_fp16, var_19166_cast_fp16))[name = string("op_19176_cast_fp16")]; + tensor var_19177_cast_fp16 = mul(x = var_19176_cast_fp16, y = sin_121_to_fp16)[name = string("op_19177_cast_fp16")]; + tensor mh_k_499_cast_fp16 = add(x = var_19161_cast_fp16, y = var_19177_cast_fp16)[name = string("mh_k_499_cast_fp16")]; + tensor var_19181 = const()[name = string("op_19181"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_251_cast_fp16 = reshape(shape = var_19181, x = mh_k_499_cast_fp16)[name = string("current_key_251_cast_fp16")]; + tensor var_19187_to_fp16 = const()[name = string("op_19187_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201472)))]; + tensor var_19188_cast_fp16 = mul(x = obj_551_cast_fp16, y = var_19187_to_fp16)[name = string("op_19188_cast_fp16")]; + tensor var_19185_to_fp16 = const()[name = string("op_19185_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201600)))]; + tensor var_19189_cast_fp16 = mul(x = current_key_251_cast_fp16, y = var_19185_to_fp16)[name = string("op_19189_cast_fp16")]; + tensor key_251_cast_fp16 = add(x = var_19188_cast_fp16, y = var_19189_cast_fp16)[name = string("key_251_cast_fp16")]; + tensor var_19191_to_fp16 = const()[name = string("op_19191_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201472)))]; + tensor var_19192_cast_fp16 = mul(x = obj_553_cast_fp16, y = var_19191_to_fp16)[name = string("op_19192_cast_fp16")]; + tensor var_19193_cast_fp16 = mul(x = current_value_125_cast_fp16, y = var_19185_to_fp16)[name = string("op_19193_cast_fp16")]; + tensor value_125_cast_fp16 = add(x = var_19192_cast_fp16, y = var_19193_cast_fp16)[name = string("value_125_cast_fp16")]; + fp16 var_19200_to_fp16 = const()[name = string("op_19200_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_503_cast_fp16 = mul(x = mh_q_499_cast_fp16, y = var_19200_to_fp16)[name = string("mh_q_503_cast_fp16")]; + tensor var_19202 = const()[name = string("op_19202"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_501_cast_fp16 = reshape(shape = var_19202, x = key_251_cast_fp16)[name = string("mh_k_501_cast_fp16")]; + tensor var_19204 = const()[name = string("op_19204"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_249_cast_fp16 = reshape(shape = var_19204, x = value_125_cast_fp16)[name = string("mh_v_249_cast_fp16")]; + tensor transpose_248_perm_0 = const()[name = string("transpose_248_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_124_reps_0 = const()[name = string("tile_124_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_248_cast_fp16 = transpose(perm = transpose_248_perm_0, x = mh_k_501_cast_fp16)[name = string("transpose_107")]; + tensor tile_124_cast_fp16 = tile(reps = tile_124_reps_0, x = transpose_248_cast_fp16)[name = string("tile_124_cast_fp16")]; + tensor concat_311 = const()[name = string("concat_311"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_248_cast_fp16 = reshape(shape = concat_311, x = tile_124_cast_fp16)[name = string("reshape_248_cast_fp16")]; + tensor transpose_249_perm_0 = const()[name = string("transpose_249_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_312 = const()[name = string("concat_312"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_249_cast_fp16 = transpose(perm = transpose_249_perm_0, x = reshape_248_cast_fp16)[name = string("transpose_106")]; + tensor reshape_249_cast_fp16 = reshape(shape = concat_312, x = transpose_249_cast_fp16)[name = string("reshape_249_cast_fp16")]; + tensor transpose_250_perm_0 = const()[name = string("transpose_250_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_125_reps_0 = const()[name = string("tile_125_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_250_cast_fp16 = transpose(perm = transpose_250_perm_0, x = mh_v_249_cast_fp16)[name = string("transpose_105")]; + tensor tile_125_cast_fp16 = tile(reps = tile_125_reps_0, x = transpose_250_cast_fp16)[name = string("tile_125_cast_fp16")]; + tensor concat_313 = const()[name = string("concat_313"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_250_cast_fp16 = reshape(shape = concat_313, x = tile_125_cast_fp16)[name = string("reshape_250_cast_fp16")]; + tensor transpose_251_perm_0 = const()[name = string("transpose_251_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_314 = const()[name = string("concat_314"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_251_cast_fp16 = transpose(perm = transpose_251_perm_0, x = reshape_250_cast_fp16)[name = string("transpose_104")]; + tensor reshape_251_cast_fp16 = reshape(shape = concat_314, x = transpose_251_cast_fp16)[name = string("reshape_251_cast_fp16")]; + tensor transpose_565_perm_0 = const()[name = string("transpose_565_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_373_transpose_x_1 = const()[name = string("mh_w_373_transpose_x_1"), val = bool(true)]; + bool mh_w_373_transpose_y_1 = const()[name = string("mh_w_373_transpose_y_1"), val = bool(false)]; + tensor transpose_565_cast_fp16 = transpose(perm = transpose_565_perm_0, x = reshape_249_cast_fp16)[name = string("transpose_103")]; + tensor mh_w_373_cast_fp16 = matmul(transpose_x = mh_w_373_transpose_x_1, transpose_y = mh_w_373_transpose_y_1, x = mh_q_503_cast_fp16, y = transpose_565_cast_fp16)[name = string("mh_w_373_cast_fp16")]; + tensor var_19212_to_fp16 = const()[name = string("op_19212_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201728)))]; + tensor mh_w_375_cast_fp16 = add(x = mh_w_373_cast_fp16, y = var_19212_to_fp16)[name = string("mh_w_375_cast_fp16")]; + tensor mh_w_377_cast_fp16 = softmax(axis = var_19032, x = mh_w_375_cast_fp16)[name = string("mh_w_377_cast_fp16")]; + tensor transpose_566_perm_0 = const()[name = string("transpose_566_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_125_transpose_x_1 = const()[name = string("attn_125_transpose_x_1"), val = bool(false)]; + bool attn_125_transpose_y_1 = const()[name = string("attn_125_transpose_y_1"), val = bool(true)]; + tensor transpose_566_cast_fp16 = transpose(perm = transpose_566_perm_0, x = reshape_251_cast_fp16)[name = string("transpose_102")]; + tensor attn_125_cast_fp16 = matmul(transpose_x = attn_125_transpose_x_1, transpose_y = attn_125_transpose_y_1, x = transpose_566_cast_fp16, y = mh_w_377_cast_fp16)[name = string("attn_125_cast_fp16")]; + tensor var_19218 = const()[name = string("op_19218"), val = tensor([1, 2048, 1, 1])]; + tensor input_541_cast_fp16 = reshape(shape = var_19218, x = attn_125_cast_fp16)[name = string("input_541_cast_fp16")]; + string obj_555_pad_type_0 = const()[name = string("obj_555_pad_type_0"), val = string("valid")]; + tensor obj_555_strides_0 = const()[name = string("obj_555_strides_0"), val = tensor([1, 1])]; + tensor obj_555_pad_0 = const()[name = string("obj_555_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_555_dilations_0 = const()[name = string("obj_555_dilations_0"), val = tensor([1, 1])]; + int32 obj_555_groups_0 = const()[name = string("obj_555_groups_0"), val = int32(1)]; + tensor obj_555_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_555_dilations_0, groups = obj_555_groups_0, pad = obj_555_pad_0, pad_type = obj_555_pad_type_0, strides = obj_555_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_541_cast_fp16)[name = string("obj_555_cast_fp16")]; + tensor inputs_525_cast_fp16 = add(x = inputs_519_cast_fp16, y = obj_555_cast_fp16)[name = string("inputs_525_cast_fp16")]; + tensor inputs_sq_525_cast_fp16 = mul(x = inputs_525_cast_fp16, y = inputs_525_cast_fp16)[name = string("inputs_sq_525_cast_fp16")]; + tensor variance_525_axes_0 = const()[name = string("variance_525_axes_0"), val = tensor([1])]; + bool variance_525_keep_dims_0 = const()[name = string("variance_525_keep_dims_0"), val = bool(true)]; + tensor variance_525_cast_fp16 = reduce_mean(axes = variance_525_axes_0, keep_dims = variance_525_keep_dims_0, x = inputs_sq_525_cast_fp16)[name = string("variance_525_cast_fp16")]; + fp16 var_19236_to_fp16 = const()[name = string("op_19236_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19237_cast_fp16 = add(x = variance_525_cast_fp16, y = var_19236_to_fp16)[name = string("op_19237_cast_fp16")]; + fp32 var_19238_epsilon_0 = const()[name = string("op_19238_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19238_cast_fp16 = rsqrt(epsilon = var_19238_epsilon_0, x = var_19237_cast_fp16)[name = string("op_19238_cast_fp16")]; + tensor hidden_states_649_cast_fp16 = mul(x = inputs_525_cast_fp16, y = var_19238_cast_fp16)[name = string("hidden_states_649_cast_fp16")]; + tensor input_543_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_649_cast_fp16)[name = string("input_543_cast_fp16")]; + string input_545_pad_type_0 = const()[name = string("input_545_pad_type_0"), val = string("valid")]; + tensor input_545_strides_0 = const()[name = string("input_545_strides_0"), val = tensor([1, 1])]; + tensor input_545_pad_0 = const()[name = string("input_545_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_545_dilations_0 = const()[name = string("input_545_dilations_0"), val = tensor([1, 1])]; + int32 input_545_groups_0 = const()[name = string("input_545_groups_0"), val = int32(1)]; + tensor input_545_cast_fp16 = conv(dilations = input_545_dilations_0, groups = input_545_groups_0, pad = input_545_pad_0, pad_type = input_545_pad_type_0, strides = input_545_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_543_cast_fp16)[name = string("input_545_cast_fp16")]; + tensor var_19252_cast_fp16 = silu(x = input_545_cast_fp16)[name = string("op_19252_cast_fp16")]; + string var_19258_pad_type_0 = const()[name = string("op_19258_pad_type_0"), val = string("valid")]; + tensor var_19258_strides_0 = const()[name = string("op_19258_strides_0"), val = tensor([1, 1])]; + tensor var_19258_pad_0 = const()[name = string("op_19258_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19258_dilations_0 = const()[name = string("op_19258_dilations_0"), val = tensor([1, 1])]; + int32 var_19258_groups_0 = const()[name = string("op_19258_groups_0"), val = int32(1)]; + tensor var_19258_cast_fp16 = conv(dilations = var_19258_dilations_0, groups = var_19258_groups_0, pad = var_19258_pad_0, pad_type = var_19258_pad_type_0, strides = var_19258_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_543_cast_fp16)[name = string("op_19258_cast_fp16")]; + tensor input_547_cast_fp16 = mul(x = var_19252_cast_fp16, y = var_19258_cast_fp16)[name = string("input_547_cast_fp16")]; + string hidden_states_651_pad_type_0 = const()[name = string("hidden_states_651_pad_type_0"), val = string("valid")]; + tensor hidden_states_651_strides_0 = const()[name = string("hidden_states_651_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_651_pad_0 = const()[name = string("hidden_states_651_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_651_dilations_0 = const()[name = string("hidden_states_651_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_651_groups_0 = const()[name = string("hidden_states_651_groups_0"), val = int32(1)]; + tensor hidden_states_651_cast_fp16 = conv(dilations = hidden_states_651_dilations_0, groups = hidden_states_651_groups_0, pad = hidden_states_651_pad_0, pad_type = hidden_states_651_pad_type_0, strides = hidden_states_651_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_547_cast_fp16)[name = string("hidden_states_651_cast_fp16")]; + tensor inputs_527_cast_fp16 = add(x = inputs_525_cast_fp16, y = hidden_states_651_cast_fp16)[name = string("inputs_527_cast_fp16")]; + tensor obj_559_begin_0 = const()[name = string("obj_559_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_559_end_0 = const()[name = string("obj_559_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_559_end_mask_0 = const()[name = string("obj_559_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_559_cast_fp16 = slice_by_index(begin = obj_559_begin_0, end = obj_559_end_0, end_mask = obj_559_end_mask_0, x = key_caches_25_cast_fp16)[name = string("obj_559_cast_fp16")]; + tensor obj_561_begin_0 = const()[name = string("obj_561_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_561_end_0 = const()[name = string("obj_561_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_561_end_mask_0 = const()[name = string("obj_561_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_561_cast_fp16 = slice_by_index(begin = obj_561_begin_0, end = obj_561_end_0, end_mask = obj_561_end_mask_0, x = value_caches_25_cast_fp16)[name = string("obj_561_cast_fp16")]; + int32 var_19306 = const()[name = string("op_19306"), val = int32(3)]; + int32 var_19316 = const()[name = string("op_19316"), val = int32(-2)]; + tensor inputs_sq_527_cast_fp16 = mul(x = inputs_527_cast_fp16, y = inputs_527_cast_fp16)[name = string("inputs_sq_527_cast_fp16")]; + tensor variance_527_axes_0 = const()[name = string("variance_527_axes_0"), val = tensor([1])]; + bool variance_527_keep_dims_0 = const()[name = string("variance_527_keep_dims_0"), val = bool(true)]; + tensor variance_527_cast_fp16 = reduce_mean(axes = variance_527_axes_0, keep_dims = variance_527_keep_dims_0, x = inputs_sq_527_cast_fp16)[name = string("variance_527_cast_fp16")]; + fp16 var_19330_to_fp16 = const()[name = string("op_19330_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19331_cast_fp16 = add(x = variance_527_cast_fp16, y = var_19330_to_fp16)[name = string("op_19331_cast_fp16")]; + fp32 var_19332_epsilon_0 = const()[name = string("op_19332_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19332_cast_fp16 = rsqrt(epsilon = var_19332_epsilon_0, x = var_19331_cast_fp16)[name = string("op_19332_cast_fp16")]; + tensor hidden_states_653_cast_fp16 = mul(x = inputs_527_cast_fp16, y = var_19332_cast_fp16)[name = string("hidden_states_653_cast_fp16")]; + tensor obj_557_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_653_cast_fp16)[name = string("obj_557_cast_fp16")]; + string query_379_pad_type_0 = const()[name = string("query_379_pad_type_0"), val = string("valid")]; + tensor query_379_strides_0 = const()[name = string("query_379_strides_0"), val = tensor([1, 1])]; + tensor query_379_pad_0 = const()[name = string("query_379_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_379_dilations_0 = const()[name = string("query_379_dilations_0"), val = tensor([1, 1])]; + int32 query_379_groups_0 = const()[name = string("query_379_groups_0"), val = int32(1)]; + tensor query_379_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_379_dilations_0, groups = query_379_groups_0, pad = query_379_pad_0, pad_type = query_379_pad_type_0, strides = query_379_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_557_cast_fp16)[name = string("query_379_cast_fp16")]; + string current_key_253_pad_type_0 = const()[name = string("current_key_253_pad_type_0"), val = string("valid")]; + tensor current_key_253_strides_0 = const()[name = string("current_key_253_strides_0"), val = tensor([1, 1])]; + tensor current_key_253_pad_0 = const()[name = string("current_key_253_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_253_dilations_0 = const()[name = string("current_key_253_dilations_0"), val = tensor([1, 1])]; + int32 current_key_253_groups_0 = const()[name = string("current_key_253_groups_0"), val = int32(1)]; + tensor current_key_253_cast_fp16 = conv(dilations = current_key_253_dilations_0, groups = current_key_253_groups_0, pad = current_key_253_pad_0, pad_type = current_key_253_pad_type_0, strides = current_key_253_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_557_cast_fp16)[name = string("current_key_253_cast_fp16")]; + string current_value_127_pad_type_0 = const()[name = string("current_value_127_pad_type_0"), val = string("valid")]; + tensor current_value_127_strides_0 = const()[name = string("current_value_127_strides_0"), val = tensor([1, 1])]; + tensor current_value_127_pad_0 = const()[name = string("current_value_127_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_127_dilations_0 = const()[name = string("current_value_127_dilations_0"), val = tensor([1, 1])]; + int32 current_value_127_groups_0 = const()[name = string("current_value_127_groups_0"), val = int32(1)]; + tensor current_value_127_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_127_dilations_0, groups = current_value_127_groups_0, pad = current_value_127_pad_0, pad_type = current_value_127_pad_type_0, strides = current_value_127_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_557_cast_fp16)[name = string("current_value_127_cast_fp16")]; + tensor var_19369 = const()[name = string("op_19369"), val = tensor([16, 128, 1, 1])]; + tensor inputs_529_cast_fp16 = reshape(shape = var_19369, x = query_379_cast_fp16)[name = string("inputs_529_cast_fp16")]; + tensor inputs_sq_529_cast_fp16 = mul(x = inputs_529_cast_fp16, y = inputs_529_cast_fp16)[name = string("inputs_sq_529_cast_fp16")]; + tensor variance_529_axes_0 = const()[name = string("variance_529_axes_0"), val = tensor([1])]; + bool variance_529_keep_dims_0 = const()[name = string("variance_529_keep_dims_0"), val = bool(true)]; + tensor variance_529_cast_fp16 = reduce_mean(axes = variance_529_axes_0, keep_dims = variance_529_keep_dims_0, x = inputs_sq_529_cast_fp16)[name = string("variance_529_cast_fp16")]; + fp16 var_19375_to_fp16 = const()[name = string("op_19375_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19376_cast_fp16 = add(x = variance_529_cast_fp16, y = var_19375_to_fp16)[name = string("op_19376_cast_fp16")]; + fp32 var_19377_epsilon_0 = const()[name = string("op_19377_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19377_cast_fp16 = rsqrt(epsilon = var_19377_epsilon_0, x = var_19376_cast_fp16)[name = string("op_19377_cast_fp16")]; + tensor hidden_states_655_cast_fp16 = mul(x = inputs_529_cast_fp16, y = var_19377_cast_fp16)[name = string("hidden_states_655_cast_fp16")]; + tensor query_normed_127_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_655_cast_fp16)[name = string("query_normed_127_cast_fp16")]; + tensor var_19385 = const()[name = string("op_19385"), val = tensor([8, 128, 1, 1])]; + tensor inputs_531_cast_fp16 = reshape(shape = var_19385, x = current_key_253_cast_fp16)[name = string("inputs_531_cast_fp16")]; + tensor inputs_sq_531_cast_fp16 = mul(x = inputs_531_cast_fp16, y = inputs_531_cast_fp16)[name = string("inputs_sq_531_cast_fp16")]; + tensor variance_531_axes_0 = const()[name = string("variance_531_axes_0"), val = tensor([1])]; + bool variance_531_keep_dims_0 = const()[name = string("variance_531_keep_dims_0"), val = bool(true)]; + tensor variance_531_cast_fp16 = reduce_mean(axes = variance_531_axes_0, keep_dims = variance_531_keep_dims_0, x = inputs_sq_531_cast_fp16)[name = string("variance_531_cast_fp16")]; + fp16 var_19391_to_fp16 = const()[name = string("op_19391_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19392_cast_fp16 = add(x = variance_531_cast_fp16, y = var_19391_to_fp16)[name = string("op_19392_cast_fp16")]; + fp32 var_19393_epsilon_0 = const()[name = string("op_19393_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19393_cast_fp16 = rsqrt(epsilon = var_19393_epsilon_0, x = var_19392_cast_fp16)[name = string("op_19393_cast_fp16")]; + tensor hidden_states_657_cast_fp16 = mul(x = inputs_531_cast_fp16, y = var_19393_cast_fp16)[name = string("hidden_states_657_cast_fp16")]; + tensor current_key_normed_127_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_657_cast_fp16)[name = string("current_key_normed_127_cast_fp16")]; + tensor var_19411 = const()[name = string("op_19411"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_505_cast_fp16 = reshape(shape = var_19411, x = query_normed_127_cast_fp16)[name = string("mh_q_505_cast_fp16")]; + tensor var_19413 = const()[name = string("op_19413"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_505_cast_fp16 = reshape(shape = var_19413, x = current_key_normed_127_cast_fp16)[name = string("mh_k_505_cast_fp16")]; + tensor var_19417_cast_fp16 = mul(x = mh_q_505_cast_fp16, y = cos_121_to_fp16)[name = string("op_19417_cast_fp16")]; + tensor var_19422_begin_0 = const()[name = string("op_19422_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19422_end_0 = const()[name = string("op_19422_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_19422_end_mask_0 = const()[name = string("op_19422_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_19422_cast_fp16 = slice_by_index(begin = var_19422_begin_0, end = var_19422_end_0, end_mask = var_19422_end_mask_0, x = mh_q_505_cast_fp16)[name = string("op_19422_cast_fp16")]; + tensor var_19428_begin_0 = const()[name = string("op_19428_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_19428_end_0 = const()[name = string("op_19428_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_19428_end_mask_0 = const()[name = string("op_19428_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_19428_cast_fp16 = slice_by_index(begin = var_19428_begin_0, end = var_19428_end_0, end_mask = var_19428_end_mask_0, x = mh_q_505_cast_fp16)[name = string("op_19428_cast_fp16")]; + fp16 const_1286_promoted_to_fp16 = const()[name = string("const_1286_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19430_cast_fp16 = mul(x = var_19428_cast_fp16, y = const_1286_promoted_to_fp16)[name = string("op_19430_cast_fp16")]; + bool var_19432_interleave_0 = const()[name = string("op_19432_interleave_0"), val = bool(false)]; + tensor var_19432_cast_fp16 = concat(axis = var_19316, interleave = var_19432_interleave_0, values = (var_19430_cast_fp16, var_19422_cast_fp16))[name = string("op_19432_cast_fp16")]; + tensor var_19433_cast_fp16 = mul(x = var_19432_cast_fp16, y = sin_121_to_fp16)[name = string("op_19433_cast_fp16")]; + tensor mh_q_507_cast_fp16 = add(x = var_19417_cast_fp16, y = var_19433_cast_fp16)[name = string("mh_q_507_cast_fp16")]; + tensor var_19435_cast_fp16 = mul(x = mh_k_505_cast_fp16, y = cos_121_to_fp16)[name = string("op_19435_cast_fp16")]; + tensor var_19440_begin_0 = const()[name = string("op_19440_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19440_end_0 = const()[name = string("op_19440_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_19440_end_mask_0 = const()[name = string("op_19440_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_19440_cast_fp16 = slice_by_index(begin = var_19440_begin_0, end = var_19440_end_0, end_mask = var_19440_end_mask_0, x = mh_k_505_cast_fp16)[name = string("op_19440_cast_fp16")]; + tensor var_19446_begin_0 = const()[name = string("op_19446_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_19446_end_0 = const()[name = string("op_19446_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_19446_end_mask_0 = const()[name = string("op_19446_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_19446_cast_fp16 = slice_by_index(begin = var_19446_begin_0, end = var_19446_end_0, end_mask = var_19446_end_mask_0, x = mh_k_505_cast_fp16)[name = string("op_19446_cast_fp16")]; + fp16 const_1289_promoted_to_fp16 = const()[name = string("const_1289_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19448_cast_fp16 = mul(x = var_19446_cast_fp16, y = const_1289_promoted_to_fp16)[name = string("op_19448_cast_fp16")]; + bool var_19450_interleave_0 = const()[name = string("op_19450_interleave_0"), val = bool(false)]; + tensor var_19450_cast_fp16 = concat(axis = var_19316, interleave = var_19450_interleave_0, values = (var_19448_cast_fp16, var_19440_cast_fp16))[name = string("op_19450_cast_fp16")]; + tensor var_19451_cast_fp16 = mul(x = var_19450_cast_fp16, y = sin_121_to_fp16)[name = string("op_19451_cast_fp16")]; + tensor mh_k_507_cast_fp16 = add(x = var_19435_cast_fp16, y = var_19451_cast_fp16)[name = string("mh_k_507_cast_fp16")]; + tensor var_19455 = const()[name = string("op_19455"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_255_cast_fp16 = reshape(shape = var_19455, x = mh_k_507_cast_fp16)[name = string("current_key_255_cast_fp16")]; + tensor var_19461_to_fp16 = const()[name = string("op_19461_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201472)))]; + tensor var_19462_cast_fp16 = mul(x = obj_559_cast_fp16, y = var_19461_to_fp16)[name = string("op_19462_cast_fp16")]; + tensor var_19459_to_fp16 = const()[name = string("op_19459_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201600)))]; + tensor var_19463_cast_fp16 = mul(x = current_key_255_cast_fp16, y = var_19459_to_fp16)[name = string("op_19463_cast_fp16")]; + tensor key_255_cast_fp16 = add(x = var_19462_cast_fp16, y = var_19463_cast_fp16)[name = string("key_255_cast_fp16")]; + tensor var_19465_to_fp16 = const()[name = string("op_19465_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201472)))]; + tensor var_19466_cast_fp16 = mul(x = obj_561_cast_fp16, y = var_19465_to_fp16)[name = string("op_19466_cast_fp16")]; + tensor var_19467_cast_fp16 = mul(x = current_value_127_cast_fp16, y = var_19459_to_fp16)[name = string("op_19467_cast_fp16")]; + tensor value_127_cast_fp16 = add(x = var_19466_cast_fp16, y = var_19467_cast_fp16)[name = string("value_127_cast_fp16")]; + fp16 var_19474_to_fp16 = const()[name = string("op_19474_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_511_cast_fp16 = mul(x = mh_q_507_cast_fp16, y = var_19474_to_fp16)[name = string("mh_q_511_cast_fp16")]; + tensor var_19476 = const()[name = string("op_19476"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_509_cast_fp16 = reshape(shape = var_19476, x = key_255_cast_fp16)[name = string("mh_k_509_cast_fp16")]; + tensor var_19478 = const()[name = string("op_19478"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_253_cast_fp16 = reshape(shape = var_19478, x = value_127_cast_fp16)[name = string("mh_v_253_cast_fp16")]; + tensor transpose_252_perm_0 = const()[name = string("transpose_252_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_126_reps_0 = const()[name = string("tile_126_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_252_cast_fp16 = transpose(perm = transpose_252_perm_0, x = mh_k_509_cast_fp16)[name = string("transpose_101")]; + tensor tile_126_cast_fp16 = tile(reps = tile_126_reps_0, x = transpose_252_cast_fp16)[name = string("tile_126_cast_fp16")]; + tensor concat_315 = const()[name = string("concat_315"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_252_cast_fp16 = reshape(shape = concat_315, x = tile_126_cast_fp16)[name = string("reshape_252_cast_fp16")]; + tensor transpose_253_perm_0 = const()[name = string("transpose_253_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_316 = const()[name = string("concat_316"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_253_cast_fp16 = transpose(perm = transpose_253_perm_0, x = reshape_252_cast_fp16)[name = string("transpose_100")]; + tensor reshape_253_cast_fp16 = reshape(shape = concat_316, x = transpose_253_cast_fp16)[name = string("reshape_253_cast_fp16")]; + tensor transpose_254_perm_0 = const()[name = string("transpose_254_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_127_reps_0 = const()[name = string("tile_127_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_254_cast_fp16 = transpose(perm = transpose_254_perm_0, x = mh_v_253_cast_fp16)[name = string("transpose_99")]; + tensor tile_127_cast_fp16 = tile(reps = tile_127_reps_0, x = transpose_254_cast_fp16)[name = string("tile_127_cast_fp16")]; + tensor concat_317 = const()[name = string("concat_317"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_254_cast_fp16 = reshape(shape = concat_317, x = tile_127_cast_fp16)[name = string("reshape_254_cast_fp16")]; + tensor transpose_255_perm_0 = const()[name = string("transpose_255_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_318 = const()[name = string("concat_318"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_255_cast_fp16 = transpose(perm = transpose_255_perm_0, x = reshape_254_cast_fp16)[name = string("transpose_98")]; + tensor reshape_255_cast_fp16 = reshape(shape = concat_318, x = transpose_255_cast_fp16)[name = string("reshape_255_cast_fp16")]; + tensor transpose_569_perm_0 = const()[name = string("transpose_569_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_379_transpose_x_1 = const()[name = string("mh_w_379_transpose_x_1"), val = bool(true)]; + bool mh_w_379_transpose_y_1 = const()[name = string("mh_w_379_transpose_y_1"), val = bool(false)]; + tensor transpose_569_cast_fp16 = transpose(perm = transpose_569_perm_0, x = reshape_253_cast_fp16)[name = string("transpose_97")]; + tensor mh_w_379_cast_fp16 = matmul(transpose_x = mh_w_379_transpose_x_1, transpose_y = mh_w_379_transpose_y_1, x = mh_q_511_cast_fp16, y = transpose_569_cast_fp16)[name = string("mh_w_379_cast_fp16")]; + tensor var_19486_to_fp16 = const()[name = string("op_19486_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201728)))]; + tensor mh_w_381_cast_fp16 = add(x = mh_w_379_cast_fp16, y = var_19486_to_fp16)[name = string("mh_w_381_cast_fp16")]; + tensor mh_w_383_cast_fp16 = softmax(axis = var_19306, x = mh_w_381_cast_fp16)[name = string("mh_w_383_cast_fp16")]; + tensor transpose_570_perm_0 = const()[name = string("transpose_570_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_127_transpose_x_1 = const()[name = string("attn_127_transpose_x_1"), val = bool(false)]; + bool attn_127_transpose_y_1 = const()[name = string("attn_127_transpose_y_1"), val = bool(true)]; + tensor transpose_570_cast_fp16 = transpose(perm = transpose_570_perm_0, x = reshape_255_cast_fp16)[name = string("transpose_96")]; + tensor attn_127_cast_fp16 = matmul(transpose_x = attn_127_transpose_x_1, transpose_y = attn_127_transpose_y_1, x = transpose_570_cast_fp16, y = mh_w_383_cast_fp16)[name = string("attn_127_cast_fp16")]; + tensor var_19492 = const()[name = string("op_19492"), val = tensor([1, 2048, 1, 1])]; + tensor input_549_cast_fp16 = reshape(shape = var_19492, x = attn_127_cast_fp16)[name = string("input_549_cast_fp16")]; + string obj_563_pad_type_0 = const()[name = string("obj_563_pad_type_0"), val = string("valid")]; + tensor obj_563_strides_0 = const()[name = string("obj_563_strides_0"), val = tensor([1, 1])]; + tensor obj_563_pad_0 = const()[name = string("obj_563_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_563_dilations_0 = const()[name = string("obj_563_dilations_0"), val = tensor([1, 1])]; + int32 obj_563_groups_0 = const()[name = string("obj_563_groups_0"), val = int32(1)]; + tensor obj_563_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_563_dilations_0, groups = obj_563_groups_0, pad = obj_563_pad_0, pad_type = obj_563_pad_type_0, strides = obj_563_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_549_cast_fp16)[name = string("obj_563_cast_fp16")]; + tensor inputs_533_cast_fp16 = add(x = inputs_527_cast_fp16, y = obj_563_cast_fp16)[name = string("inputs_533_cast_fp16")]; + tensor inputs_sq_533_cast_fp16 = mul(x = inputs_533_cast_fp16, y = inputs_533_cast_fp16)[name = string("inputs_sq_533_cast_fp16")]; + tensor variance_533_axes_0 = const()[name = string("variance_533_axes_0"), val = tensor([1])]; + bool variance_533_keep_dims_0 = const()[name = string("variance_533_keep_dims_0"), val = bool(true)]; + tensor variance_533_cast_fp16 = reduce_mean(axes = variance_533_axes_0, keep_dims = variance_533_keep_dims_0, x = inputs_sq_533_cast_fp16)[name = string("variance_533_cast_fp16")]; + fp16 var_19510_to_fp16 = const()[name = string("op_19510_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19511_cast_fp16 = add(x = variance_533_cast_fp16, y = var_19510_to_fp16)[name = string("op_19511_cast_fp16")]; + fp32 var_19512_epsilon_0 = const()[name = string("op_19512_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19512_cast_fp16 = rsqrt(epsilon = var_19512_epsilon_0, x = var_19511_cast_fp16)[name = string("op_19512_cast_fp16")]; + tensor hidden_states_659_cast_fp16 = mul(x = inputs_533_cast_fp16, y = var_19512_cast_fp16)[name = string("hidden_states_659_cast_fp16")]; + tensor input_551_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_659_cast_fp16)[name = string("input_551_cast_fp16")]; + string input_553_pad_type_0 = const()[name = string("input_553_pad_type_0"), val = string("valid")]; + tensor input_553_strides_0 = const()[name = string("input_553_strides_0"), val = tensor([1, 1])]; + tensor input_553_pad_0 = const()[name = string("input_553_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_553_dilations_0 = const()[name = string("input_553_dilations_0"), val = tensor([1, 1])]; + int32 input_553_groups_0 = const()[name = string("input_553_groups_0"), val = int32(1)]; + tensor input_553_cast_fp16 = conv(dilations = input_553_dilations_0, groups = input_553_groups_0, pad = input_553_pad_0, pad_type = input_553_pad_type_0, strides = input_553_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_551_cast_fp16)[name = string("input_553_cast_fp16")]; + tensor var_19526_cast_fp16 = silu(x = input_553_cast_fp16)[name = string("op_19526_cast_fp16")]; + string var_19532_pad_type_0 = const()[name = string("op_19532_pad_type_0"), val = string("valid")]; + tensor var_19532_strides_0 = const()[name = string("op_19532_strides_0"), val = tensor([1, 1])]; + tensor var_19532_pad_0 = const()[name = string("op_19532_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19532_dilations_0 = const()[name = string("op_19532_dilations_0"), val = tensor([1, 1])]; + int32 var_19532_groups_0 = const()[name = string("op_19532_groups_0"), val = int32(1)]; + tensor var_19532_cast_fp16 = conv(dilations = var_19532_dilations_0, groups = var_19532_groups_0, pad = var_19532_pad_0, pad_type = var_19532_pad_type_0, strides = var_19532_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_551_cast_fp16)[name = string("op_19532_cast_fp16")]; + tensor input_555_cast_fp16 = mul(x = var_19526_cast_fp16, y = var_19532_cast_fp16)[name = string("input_555_cast_fp16")]; + string hidden_states_661_pad_type_0 = const()[name = string("hidden_states_661_pad_type_0"), val = string("valid")]; + tensor hidden_states_661_strides_0 = const()[name = string("hidden_states_661_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_661_pad_0 = const()[name = string("hidden_states_661_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_661_dilations_0 = const()[name = string("hidden_states_661_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_661_groups_0 = const()[name = string("hidden_states_661_groups_0"), val = int32(1)]; + tensor hidden_states_661_cast_fp16 = conv(dilations = hidden_states_661_dilations_0, groups = hidden_states_661_groups_0, pad = hidden_states_661_pad_0, pad_type = hidden_states_661_pad_type_0, strides = hidden_states_661_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_555_cast_fp16)[name = string("hidden_states_661_cast_fp16")]; + tensor inputs_535_cast_fp16 = add(x = inputs_533_cast_fp16, y = hidden_states_661_cast_fp16)[name = string("inputs_535_cast_fp16")]; + tensor obj_567_begin_0 = const()[name = string("obj_567_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_567_end_0 = const()[name = string("obj_567_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_567_end_mask_0 = const()[name = string("obj_567_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_567_cast_fp16 = slice_by_index(begin = obj_567_begin_0, end = obj_567_end_0, end_mask = obj_567_end_mask_0, x = key_caches_25_cast_fp16)[name = string("obj_567_cast_fp16")]; + tensor obj_569_begin_0 = const()[name = string("obj_569_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_569_end_0 = const()[name = string("obj_569_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_569_end_mask_0 = const()[name = string("obj_569_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_569_cast_fp16 = slice_by_index(begin = obj_569_begin_0, end = obj_569_end_0, end_mask = obj_569_end_mask_0, x = value_caches_25_cast_fp16)[name = string("obj_569_cast_fp16")]; + int32 var_19580 = const()[name = string("op_19580"), val = int32(3)]; + int32 var_19590 = const()[name = string("op_19590"), val = int32(-2)]; + tensor inputs_sq_535_cast_fp16 = mul(x = inputs_535_cast_fp16, y = inputs_535_cast_fp16)[name = string("inputs_sq_535_cast_fp16")]; + tensor variance_535_axes_0 = const()[name = string("variance_535_axes_0"), val = tensor([1])]; + bool variance_535_keep_dims_0 = const()[name = string("variance_535_keep_dims_0"), val = bool(true)]; + tensor variance_535_cast_fp16 = reduce_mean(axes = variance_535_axes_0, keep_dims = variance_535_keep_dims_0, x = inputs_sq_535_cast_fp16)[name = string("variance_535_cast_fp16")]; + fp16 var_19604_to_fp16 = const()[name = string("op_19604_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19605_cast_fp16 = add(x = variance_535_cast_fp16, y = var_19604_to_fp16)[name = string("op_19605_cast_fp16")]; + fp32 var_19606_epsilon_0 = const()[name = string("op_19606_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19606_cast_fp16 = rsqrt(epsilon = var_19606_epsilon_0, x = var_19605_cast_fp16)[name = string("op_19606_cast_fp16")]; + tensor hidden_states_663_cast_fp16 = mul(x = inputs_535_cast_fp16, y = var_19606_cast_fp16)[name = string("hidden_states_663_cast_fp16")]; + tensor obj_565_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_663_cast_fp16)[name = string("obj_565_cast_fp16")]; + string query_385_pad_type_0 = const()[name = string("query_385_pad_type_0"), val = string("valid")]; + tensor query_385_strides_0 = const()[name = string("query_385_strides_0"), val = tensor([1, 1])]; + tensor query_385_pad_0 = const()[name = string("query_385_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_385_dilations_0 = const()[name = string("query_385_dilations_0"), val = tensor([1, 1])]; + int32 query_385_groups_0 = const()[name = string("query_385_groups_0"), val = int32(1)]; + tensor query_385_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_385_dilations_0, groups = query_385_groups_0, pad = query_385_pad_0, pad_type = query_385_pad_type_0, strides = query_385_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_565_cast_fp16)[name = string("query_385_cast_fp16")]; + string current_key_257_pad_type_0 = const()[name = string("current_key_257_pad_type_0"), val = string("valid")]; + tensor current_key_257_strides_0 = const()[name = string("current_key_257_strides_0"), val = tensor([1, 1])]; + tensor current_key_257_pad_0 = const()[name = string("current_key_257_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_257_dilations_0 = const()[name = string("current_key_257_dilations_0"), val = tensor([1, 1])]; + int32 current_key_257_groups_0 = const()[name = string("current_key_257_groups_0"), val = int32(1)]; + tensor current_key_257_cast_fp16 = conv(dilations = current_key_257_dilations_0, groups = current_key_257_groups_0, pad = current_key_257_pad_0, pad_type = current_key_257_pad_type_0, strides = current_key_257_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_565_cast_fp16)[name = string("current_key_257_cast_fp16")]; + string current_value_129_pad_type_0 = const()[name = string("current_value_129_pad_type_0"), val = string("valid")]; + tensor current_value_129_strides_0 = const()[name = string("current_value_129_strides_0"), val = tensor([1, 1])]; + tensor current_value_129_pad_0 = const()[name = string("current_value_129_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_129_dilations_0 = const()[name = string("current_value_129_dilations_0"), val = tensor([1, 1])]; + int32 current_value_129_groups_0 = const()[name = string("current_value_129_groups_0"), val = int32(1)]; + tensor current_value_129_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_129_dilations_0, groups = current_value_129_groups_0, pad = current_value_129_pad_0, pad_type = current_value_129_pad_type_0, strides = current_value_129_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_565_cast_fp16)[name = string("current_value_129_cast_fp16")]; + tensor var_19643 = const()[name = string("op_19643"), val = tensor([16, 128, 1, 1])]; + tensor inputs_537_cast_fp16 = reshape(shape = var_19643, x = query_385_cast_fp16)[name = string("inputs_537_cast_fp16")]; + tensor inputs_sq_537_cast_fp16 = mul(x = inputs_537_cast_fp16, y = inputs_537_cast_fp16)[name = string("inputs_sq_537_cast_fp16")]; + tensor variance_537_axes_0 = const()[name = string("variance_537_axes_0"), val = tensor([1])]; + bool variance_537_keep_dims_0 = const()[name = string("variance_537_keep_dims_0"), val = bool(true)]; + tensor variance_537_cast_fp16 = reduce_mean(axes = variance_537_axes_0, keep_dims = variance_537_keep_dims_0, x = inputs_sq_537_cast_fp16)[name = string("variance_537_cast_fp16")]; + fp16 var_19649_to_fp16 = const()[name = string("op_19649_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19650_cast_fp16 = add(x = variance_537_cast_fp16, y = var_19649_to_fp16)[name = string("op_19650_cast_fp16")]; + fp32 var_19651_epsilon_0 = const()[name = string("op_19651_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19651_cast_fp16 = rsqrt(epsilon = var_19651_epsilon_0, x = var_19650_cast_fp16)[name = string("op_19651_cast_fp16")]; + tensor hidden_states_665_cast_fp16 = mul(x = inputs_537_cast_fp16, y = var_19651_cast_fp16)[name = string("hidden_states_665_cast_fp16")]; + tensor query_normed_129_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_665_cast_fp16)[name = string("query_normed_129_cast_fp16")]; + tensor var_19659 = const()[name = string("op_19659"), val = tensor([8, 128, 1, 1])]; + tensor inputs_539_cast_fp16 = reshape(shape = var_19659, x = current_key_257_cast_fp16)[name = string("inputs_539_cast_fp16")]; + tensor inputs_sq_539_cast_fp16 = mul(x = inputs_539_cast_fp16, y = inputs_539_cast_fp16)[name = string("inputs_sq_539_cast_fp16")]; + tensor variance_539_axes_0 = const()[name = string("variance_539_axes_0"), val = tensor([1])]; + bool variance_539_keep_dims_0 = const()[name = string("variance_539_keep_dims_0"), val = bool(true)]; + tensor variance_539_cast_fp16 = reduce_mean(axes = variance_539_axes_0, keep_dims = variance_539_keep_dims_0, x = inputs_sq_539_cast_fp16)[name = string("variance_539_cast_fp16")]; + fp16 var_19665_to_fp16 = const()[name = string("op_19665_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19666_cast_fp16 = add(x = variance_539_cast_fp16, y = var_19665_to_fp16)[name = string("op_19666_cast_fp16")]; + fp32 var_19667_epsilon_0 = const()[name = string("op_19667_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19667_cast_fp16 = rsqrt(epsilon = var_19667_epsilon_0, x = var_19666_cast_fp16)[name = string("op_19667_cast_fp16")]; + tensor hidden_states_667_cast_fp16 = mul(x = inputs_539_cast_fp16, y = var_19667_cast_fp16)[name = string("hidden_states_667_cast_fp16")]; + tensor current_key_normed_129_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_667_cast_fp16)[name = string("current_key_normed_129_cast_fp16")]; + tensor var_19685 = const()[name = string("op_19685"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_513_cast_fp16 = reshape(shape = var_19685, x = query_normed_129_cast_fp16)[name = string("mh_q_513_cast_fp16")]; + tensor var_19687 = const()[name = string("op_19687"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_513_cast_fp16 = reshape(shape = var_19687, x = current_key_normed_129_cast_fp16)[name = string("mh_k_513_cast_fp16")]; + tensor var_19691_cast_fp16 = mul(x = mh_q_513_cast_fp16, y = cos_121_to_fp16)[name = string("op_19691_cast_fp16")]; + tensor var_19696_begin_0 = const()[name = string("op_19696_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19696_end_0 = const()[name = string("op_19696_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_19696_end_mask_0 = const()[name = string("op_19696_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_19696_cast_fp16 = slice_by_index(begin = var_19696_begin_0, end = var_19696_end_0, end_mask = var_19696_end_mask_0, x = mh_q_513_cast_fp16)[name = string("op_19696_cast_fp16")]; + tensor var_19702_begin_0 = const()[name = string("op_19702_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_19702_end_0 = const()[name = string("op_19702_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_19702_end_mask_0 = const()[name = string("op_19702_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_19702_cast_fp16 = slice_by_index(begin = var_19702_begin_0, end = var_19702_end_0, end_mask = var_19702_end_mask_0, x = mh_q_513_cast_fp16)[name = string("op_19702_cast_fp16")]; + fp16 const_1306_promoted_to_fp16 = const()[name = string("const_1306_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19704_cast_fp16 = mul(x = var_19702_cast_fp16, y = const_1306_promoted_to_fp16)[name = string("op_19704_cast_fp16")]; + bool var_19706_interleave_0 = const()[name = string("op_19706_interleave_0"), val = bool(false)]; + tensor var_19706_cast_fp16 = concat(axis = var_19590, interleave = var_19706_interleave_0, values = (var_19704_cast_fp16, var_19696_cast_fp16))[name = string("op_19706_cast_fp16")]; + tensor var_19707_cast_fp16 = mul(x = var_19706_cast_fp16, y = sin_121_to_fp16)[name = string("op_19707_cast_fp16")]; + tensor mh_q_515_cast_fp16 = add(x = var_19691_cast_fp16, y = var_19707_cast_fp16)[name = string("mh_q_515_cast_fp16")]; + tensor var_19709_cast_fp16 = mul(x = mh_k_513_cast_fp16, y = cos_121_to_fp16)[name = string("op_19709_cast_fp16")]; + tensor var_19714_begin_0 = const()[name = string("op_19714_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19714_end_0 = const()[name = string("op_19714_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_19714_end_mask_0 = const()[name = string("op_19714_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_19714_cast_fp16 = slice_by_index(begin = var_19714_begin_0, end = var_19714_end_0, end_mask = var_19714_end_mask_0, x = mh_k_513_cast_fp16)[name = string("op_19714_cast_fp16")]; + tensor var_19720_begin_0 = const()[name = string("op_19720_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_19720_end_0 = const()[name = string("op_19720_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_19720_end_mask_0 = const()[name = string("op_19720_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_19720_cast_fp16 = slice_by_index(begin = var_19720_begin_0, end = var_19720_end_0, end_mask = var_19720_end_mask_0, x = mh_k_513_cast_fp16)[name = string("op_19720_cast_fp16")]; + fp16 const_1309_promoted_to_fp16 = const()[name = string("const_1309_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_19722_cast_fp16 = mul(x = var_19720_cast_fp16, y = const_1309_promoted_to_fp16)[name = string("op_19722_cast_fp16")]; + bool var_19724_interleave_0 = const()[name = string("op_19724_interleave_0"), val = bool(false)]; + tensor var_19724_cast_fp16 = concat(axis = var_19590, interleave = var_19724_interleave_0, values = (var_19722_cast_fp16, var_19714_cast_fp16))[name = string("op_19724_cast_fp16")]; + tensor var_19725_cast_fp16 = mul(x = var_19724_cast_fp16, y = sin_121_to_fp16)[name = string("op_19725_cast_fp16")]; + tensor mh_k_515_cast_fp16 = add(x = var_19709_cast_fp16, y = var_19725_cast_fp16)[name = string("mh_k_515_cast_fp16")]; + tensor var_19729 = const()[name = string("op_19729"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_259_cast_fp16 = reshape(shape = var_19729, x = mh_k_515_cast_fp16)[name = string("current_key_259_cast_fp16")]; + tensor var_19735_to_fp16 = const()[name = string("op_19735_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201472)))]; + tensor var_19736_cast_fp16 = mul(x = obj_567_cast_fp16, y = var_19735_to_fp16)[name = string("op_19736_cast_fp16")]; + tensor var_19733_to_fp16 = const()[name = string("op_19733_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201600)))]; + tensor var_19737_cast_fp16 = mul(x = current_key_259_cast_fp16, y = var_19733_to_fp16)[name = string("op_19737_cast_fp16")]; + tensor key_259_cast_fp16 = add(x = var_19736_cast_fp16, y = var_19737_cast_fp16)[name = string("key_259_cast_fp16")]; + tensor var_19739_to_fp16 = const()[name = string("op_19739_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201472)))]; + tensor var_19740_cast_fp16 = mul(x = obj_569_cast_fp16, y = var_19739_to_fp16)[name = string("op_19740_cast_fp16")]; + tensor var_19741_cast_fp16 = mul(x = current_value_129_cast_fp16, y = var_19733_to_fp16)[name = string("op_19741_cast_fp16")]; + tensor value_129_cast_fp16 = add(x = var_19740_cast_fp16, y = var_19741_cast_fp16)[name = string("value_129_cast_fp16")]; + fp16 var_19748_to_fp16 = const()[name = string("op_19748_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_519_cast_fp16 = mul(x = mh_q_515_cast_fp16, y = var_19748_to_fp16)[name = string("mh_q_519_cast_fp16")]; + tensor var_19750 = const()[name = string("op_19750"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_517_cast_fp16 = reshape(shape = var_19750, x = key_259_cast_fp16)[name = string("mh_k_517_cast_fp16")]; + tensor var_19752 = const()[name = string("op_19752"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_257_cast_fp16 = reshape(shape = var_19752, x = value_129_cast_fp16)[name = string("mh_v_257_cast_fp16")]; + tensor transpose_256_perm_0 = const()[name = string("transpose_256_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_128_reps_0 = const()[name = string("tile_128_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_256_cast_fp16 = transpose(perm = transpose_256_perm_0, x = mh_k_517_cast_fp16)[name = string("transpose_95")]; + tensor tile_128_cast_fp16 = tile(reps = tile_128_reps_0, x = transpose_256_cast_fp16)[name = string("tile_128_cast_fp16")]; + tensor concat_319 = const()[name = string("concat_319"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_256_cast_fp16 = reshape(shape = concat_319, x = tile_128_cast_fp16)[name = string("reshape_256_cast_fp16")]; + tensor transpose_257_perm_0 = const()[name = string("transpose_257_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_320 = const()[name = string("concat_320"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_257_cast_fp16 = transpose(perm = transpose_257_perm_0, x = reshape_256_cast_fp16)[name = string("transpose_94")]; + tensor reshape_257_cast_fp16 = reshape(shape = concat_320, x = transpose_257_cast_fp16)[name = string("reshape_257_cast_fp16")]; + tensor transpose_258_perm_0 = const()[name = string("transpose_258_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_129_reps_0 = const()[name = string("tile_129_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_258_cast_fp16 = transpose(perm = transpose_258_perm_0, x = mh_v_257_cast_fp16)[name = string("transpose_93")]; + tensor tile_129_cast_fp16 = tile(reps = tile_129_reps_0, x = transpose_258_cast_fp16)[name = string("tile_129_cast_fp16")]; + tensor concat_321 = const()[name = string("concat_321"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_258_cast_fp16 = reshape(shape = concat_321, x = tile_129_cast_fp16)[name = string("reshape_258_cast_fp16")]; + tensor transpose_259_perm_0 = const()[name = string("transpose_259_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_322 = const()[name = string("concat_322"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_259_cast_fp16 = transpose(perm = transpose_259_perm_0, x = reshape_258_cast_fp16)[name = string("transpose_92")]; + tensor reshape_259_cast_fp16 = reshape(shape = concat_322, x = transpose_259_cast_fp16)[name = string("reshape_259_cast_fp16")]; + tensor transpose_573_perm_0 = const()[name = string("transpose_573_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_385_transpose_x_1 = const()[name = string("mh_w_385_transpose_x_1"), val = bool(true)]; + bool mh_w_385_transpose_y_1 = const()[name = string("mh_w_385_transpose_y_1"), val = bool(false)]; + tensor transpose_573_cast_fp16 = transpose(perm = transpose_573_perm_0, x = reshape_257_cast_fp16)[name = string("transpose_91")]; + tensor mh_w_385_cast_fp16 = matmul(transpose_x = mh_w_385_transpose_x_1, transpose_y = mh_w_385_transpose_y_1, x = mh_q_519_cast_fp16, y = transpose_573_cast_fp16)[name = string("mh_w_385_cast_fp16")]; + tensor var_19760_to_fp16 = const()[name = string("op_19760_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201728)))]; + tensor mh_w_387_cast_fp16 = add(x = mh_w_385_cast_fp16, y = var_19760_to_fp16)[name = string("mh_w_387_cast_fp16")]; + tensor mh_w_389_cast_fp16 = softmax(axis = var_19580, x = mh_w_387_cast_fp16)[name = string("mh_w_389_cast_fp16")]; + tensor transpose_574_perm_0 = const()[name = string("transpose_574_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_129_transpose_x_1 = const()[name = string("attn_129_transpose_x_1"), val = bool(false)]; + bool attn_129_transpose_y_1 = const()[name = string("attn_129_transpose_y_1"), val = bool(true)]; + tensor transpose_574_cast_fp16 = transpose(perm = transpose_574_perm_0, x = reshape_259_cast_fp16)[name = string("transpose_90")]; + tensor attn_129_cast_fp16 = matmul(transpose_x = attn_129_transpose_x_1, transpose_y = attn_129_transpose_y_1, x = transpose_574_cast_fp16, y = mh_w_389_cast_fp16)[name = string("attn_129_cast_fp16")]; + tensor var_19766 = const()[name = string("op_19766"), val = tensor([1, 2048, 1, 1])]; + tensor input_557_cast_fp16 = reshape(shape = var_19766, x = attn_129_cast_fp16)[name = string("input_557_cast_fp16")]; + string obj_571_pad_type_0 = const()[name = string("obj_571_pad_type_0"), val = string("valid")]; + tensor obj_571_strides_0 = const()[name = string("obj_571_strides_0"), val = tensor([1, 1])]; + tensor obj_571_pad_0 = const()[name = string("obj_571_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_571_dilations_0 = const()[name = string("obj_571_dilations_0"), val = tensor([1, 1])]; + int32 obj_571_groups_0 = const()[name = string("obj_571_groups_0"), val = int32(1)]; + tensor obj_571_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_571_dilations_0, groups = obj_571_groups_0, pad = obj_571_pad_0, pad_type = obj_571_pad_type_0, strides = obj_571_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_557_cast_fp16)[name = string("obj_571_cast_fp16")]; + tensor inputs_541_cast_fp16 = add(x = inputs_535_cast_fp16, y = obj_571_cast_fp16)[name = string("inputs_541_cast_fp16")]; + tensor inputs_sq_541_cast_fp16 = mul(x = inputs_541_cast_fp16, y = inputs_541_cast_fp16)[name = string("inputs_sq_541_cast_fp16")]; + tensor variance_541_axes_0 = const()[name = string("variance_541_axes_0"), val = tensor([1])]; + bool variance_541_keep_dims_0 = const()[name = string("variance_541_keep_dims_0"), val = bool(true)]; + tensor variance_541_cast_fp16 = reduce_mean(axes = variance_541_axes_0, keep_dims = variance_541_keep_dims_0, x = inputs_sq_541_cast_fp16)[name = string("variance_541_cast_fp16")]; + fp16 var_19784_to_fp16 = const()[name = string("op_19784_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19785_cast_fp16 = add(x = variance_541_cast_fp16, y = var_19784_to_fp16)[name = string("op_19785_cast_fp16")]; + fp32 var_19786_epsilon_0 = const()[name = string("op_19786_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19786_cast_fp16 = rsqrt(epsilon = var_19786_epsilon_0, x = var_19785_cast_fp16)[name = string("op_19786_cast_fp16")]; + tensor hidden_states_669_cast_fp16 = mul(x = inputs_541_cast_fp16, y = var_19786_cast_fp16)[name = string("hidden_states_669_cast_fp16")]; + tensor input_559_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_669_cast_fp16)[name = string("input_559_cast_fp16")]; + string input_561_pad_type_0 = const()[name = string("input_561_pad_type_0"), val = string("valid")]; + tensor input_561_strides_0 = const()[name = string("input_561_strides_0"), val = tensor([1, 1])]; + tensor input_561_pad_0 = const()[name = string("input_561_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_561_dilations_0 = const()[name = string("input_561_dilations_0"), val = tensor([1, 1])]; + int32 input_561_groups_0 = const()[name = string("input_561_groups_0"), val = int32(1)]; + tensor input_561_cast_fp16 = conv(dilations = input_561_dilations_0, groups = input_561_groups_0, pad = input_561_pad_0, pad_type = input_561_pad_type_0, strides = input_561_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_559_cast_fp16)[name = string("input_561_cast_fp16")]; + tensor var_19800_cast_fp16 = silu(x = input_561_cast_fp16)[name = string("op_19800_cast_fp16")]; + string var_19806_pad_type_0 = const()[name = string("op_19806_pad_type_0"), val = string("valid")]; + tensor var_19806_strides_0 = const()[name = string("op_19806_strides_0"), val = tensor([1, 1])]; + tensor var_19806_pad_0 = const()[name = string("op_19806_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_19806_dilations_0 = const()[name = string("op_19806_dilations_0"), val = tensor([1, 1])]; + int32 var_19806_groups_0 = const()[name = string("op_19806_groups_0"), val = int32(1)]; + tensor var_19806_cast_fp16 = conv(dilations = var_19806_dilations_0, groups = var_19806_groups_0, pad = var_19806_pad_0, pad_type = var_19806_pad_type_0, strides = var_19806_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_559_cast_fp16)[name = string("op_19806_cast_fp16")]; + tensor input_563_cast_fp16 = mul(x = var_19800_cast_fp16, y = var_19806_cast_fp16)[name = string("input_563_cast_fp16")]; + string hidden_states_671_pad_type_0 = const()[name = string("hidden_states_671_pad_type_0"), val = string("valid")]; + tensor hidden_states_671_strides_0 = const()[name = string("hidden_states_671_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_671_pad_0 = const()[name = string("hidden_states_671_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_671_dilations_0 = const()[name = string("hidden_states_671_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_671_groups_0 = const()[name = string("hidden_states_671_groups_0"), val = int32(1)]; + tensor hidden_states_671_cast_fp16 = conv(dilations = hidden_states_671_dilations_0, groups = hidden_states_671_groups_0, pad = hidden_states_671_pad_0, pad_type = hidden_states_671_pad_type_0, strides = hidden_states_671_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_563_cast_fp16)[name = string("hidden_states_671_cast_fp16")]; + tensor inputs_543_cast_fp16 = add(x = inputs_541_cast_fp16, y = hidden_states_671_cast_fp16)[name = string("inputs_543_cast_fp16")]; + int32 var_19834 = const()[name = string("op_19834"), val = int32(1)]; + bool key_caches_27_interleave_0 = const()[name = string("key_caches_27_interleave_0"), val = bool(false)]; + tensor key_caches_27_cast_fp16 = concat(axis = var_19834, interleave = key_caches_27_interleave_0, values = (key_243_cast_fp16, key_247_cast_fp16, key_251_cast_fp16, key_255_cast_fp16, key_259_cast_fp16))[name = string("key_caches_27_cast_fp16")]; + int32 var_19837 = const()[name = string("op_19837"), val = int32(1)]; + bool value_caches_27_interleave_0 = const()[name = string("value_caches_27_interleave_0"), val = bool(false)]; + tensor value_caches_27_cast_fp16 = concat(axis = var_19837, interleave = value_caches_27_interleave_0, values = (value_121_cast_fp16, value_123_cast_fp16, value_125_cast_fp16, value_127_cast_fp16, value_129_cast_fp16))[name = string("value_caches_27_cast_fp16")]; + tensor inputs_sq_543_cast_fp16 = mul(x = inputs_543_cast_fp16, y = inputs_543_cast_fp16)[name = string("inputs_sq_543_cast_fp16")]; + tensor variance_543_axes_0 = const()[name = string("variance_543_axes_0"), val = tensor([1])]; + bool variance_543_keep_dims_0 = const()[name = string("variance_543_keep_dims_0"), val = bool(true)]; + tensor variance_543_cast_fp16 = reduce_mean(axes = variance_543_axes_0, keep_dims = variance_543_keep_dims_0, x = inputs_sq_543_cast_fp16)[name = string("variance_543_cast_fp16")]; + fp16 var_19847_to_fp16 = const()[name = string("op_19847_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_19848_cast_fp16 = add(x = variance_543_cast_fp16, y = var_19847_to_fp16)[name = string("op_19848_cast_fp16")]; + fp32 var_19849_epsilon_0 = const()[name = string("op_19849_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_19849_cast_fp16 = rsqrt(epsilon = var_19849_epsilon_0, x = var_19848_cast_fp16)[name = string("op_19849_cast_fp16")]; + tensor hidden_states_673_cast_fp16 = mul(x = inputs_543_cast_fp16, y = var_19849_cast_fp16)[name = string("hidden_states_673_cast_fp16")]; + tensor input_565_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_673_cast_fp16)[name = string("input_565_cast_fp16")]; + string logits_45_pad_type_0 = const()[name = string("logits_45_pad_type_0"), val = string("valid")]; + tensor logits_45_strides_0 = const()[name = string("logits_45_strides_0"), val = tensor([1, 1])]; + tensor logits_45_pad_0 = const()[name = string("logits_45_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_45_dilations_0 = const()[name = string("logits_45_dilations_0"), val = tensor([1, 1])]; + int32 logits_45_groups_0 = const()[name = string("logits_45_groups_0"), val = int32(1)]; + tensor lm_heads_11_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103882304))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105979520))))[name = string("lm_heads_11_weight_to_fp16_palettized")]; + tensor logits_45_cast_fp16 = conv(dilations = logits_45_dilations_0, groups = logits_45_groups_0, pad = logits_45_pad_0, pad_type = logits_45_pad_type_0, strides = logits_45_strides_0, weight = lm_heads_11_weight_to_fp16_palettized, x = input_565_cast_fp16)[name = string("logits_45_cast_fp16")]; + tensor var_19867 = const()[name = string("op_19867"), val = tensor([1, 2048])]; + tensor logits_47_cast_fp16 = reshape(shape = var_19867, x = logits_45_cast_fp16)[name = string("logits_47_cast_fp16")]; + tensor scaled_logits_23_cast_fp16 = real_div(x = logits_47_cast_fp16, y = temperature)[name = string("scaled_logits_23_cast_fp16")]; + int32 var_19877 = const()[name = string("op_19877"), val = int32(100)]; + int32 top_values_23_axis_0 = const()[name = string("top_values_23_axis_0"), val = int32(1)]; + bool top_values_23_ascending_0 = const()[name = string("top_values_23_ascending_0"), val = bool(false)]; + bool top_values_23_sort_0 = const()[name = string("top_values_23_sort_0"), val = bool(true)]; + bool top_values_23_return_indices_0 = const()[name = string("top_values_23_return_indices_0"), val = bool(true)]; + string top_values_23_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_23_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_23_cast_fp16_cast_uint16_0, tensor top_values_23_cast_fp16_cast_uint16_1 = topk(ascending = top_values_23_ascending_0, axis = top_values_23_axis_0, k = var_19877, output_indices_dtype = top_values_23_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_23_return_indices_0, sort = top_values_23_sort_0, x = scaled_logits_23_cast_fp16)[name = string("top_values_23_cast_fp16_cast_uint16")]; + tensor var_19883_cast_fp16 = mul(x = top_values_23_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_19883_cast_fp16")]; + tensor var_19887_cast_fp16 = add(x = var_19883_cast_fp16, y = var_3001_cast_fp16)[name = string("op_19887_cast_fp16")]; + tensor reduce_min_11_axes_0 = const()[name = string("reduce_min_11_axes_0"), val = tensor([1])]; + bool reduce_min_11_keep_dims_0 = const()[name = string("reduce_min_11_keep_dims_0"), val = bool(true)]; + tensor reduce_min_11_cast_fp16 = reduce_min(axes = reduce_min_11_axes_0, keep_dims = reduce_min_11_keep_dims_0, x = var_19887_cast_fp16)[name = string("reduce_min_11_cast_fp16")]; + tensor var_19890_cast_fp16 = greater_equal(x = scaled_logits_23_cast_fp16, y = reduce_min_11_cast_fp16)[name = string("op_19890_cast_fp16")]; + fp16 var_19891_value_0_to_fp16 = const()[name = string("op_19891_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_19891_cast_fp16 = fill_like(ref_tensor = scaled_logits_23_cast_fp16, value = var_19891_value_0_to_fp16)[name = string("op_19891_cast_fp16")]; + tensor masked_logits_23_cast_fp16 = select(a = scaled_logits_23_cast_fp16, b = var_19891_cast_fp16, cond = var_19890_cast_fp16)[name = string("masked_logits_23_cast_fp16")]; + tensor var_19895_begin_0 = const()[name = string("op_19895_begin_0"), val = tensor([11, 0])]; + tensor var_19895_end_0 = const()[name = string("op_19895_end_0"), val = tensor([12, 2048])]; + tensor var_19895_end_mask_0 = const()[name = string("op_19895_end_mask_0"), val = tensor([false, true])]; + tensor var_19895_squeeze_mask_0 = const()[name = string("op_19895_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_19895_cast_fp16 = slice_by_index(begin = var_19895_begin_0, end = var_19895_end_0, end_mask = var_19895_end_mask_0, squeeze_mask = var_19895_squeeze_mask_0, x = gumbel)[name = string("op_19895_cast_fp16")]; + tensor var_19898 = const()[name = string("op_19898"), val = tensor([1, 2048])]; + tensor var_19899_cast_fp16 = reshape(shape = var_19898, x = var_19895_cast_fp16)[name = string("op_19899_cast_fp16")]; + tensor noisy_logits_23_cast_fp16 = add(x = masked_logits_23_cast_fp16, y = var_19899_cast_fp16)[name = string("noisy_logits_23_cast_fp16")]; + int32 code_23_axis_0 = const()[name = string("code_23_axis_0"), val = int32(1)]; + bool code_23_keep_dims_0 = const()[name = string("code_23_keep_dims_0"), val = bool(false)]; + string code_23_output_dtype_0 = const()[name = string("code_23_output_dtype_0"), val = string("int32")]; + tensor code_23_cast_fp16 = reduce_argmax(axis = code_23_axis_0, keep_dims = code_23_keep_dims_0, output_dtype = code_23_output_dtype_0, x = noisy_logits_23_cast_fp16)[name = string("code_23_cast_fp16")]; + int32 var_19910 = const()[name = string("op_19910"), val = int32(22528)]; + tensor input_567 = add(x = code_23_cast_fp16, y = var_19910)[name = string("input_567")]; + int32 code_embed_45_axis_0 = const()[name = string("code_embed_45_axis_0"), val = int32(0)]; + int32 code_embed_45_batch_dims_0 = const()[name = string("code_embed_45_batch_dims_0"), val = int32(0)]; + bool code_embed_45_validate_indices_0 = const()[name = string("code_embed_45_validate_indices_0"), val = bool(false)]; + string input_567_to_uint16_dtype_0 = const()[name = string("input_567_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_567_to_uint16 = cast(dtype = input_567_to_uint16_dtype_0, x = input_567)[name = string("cast_3")]; + tensor code_embed_45_cast_fp16_cast_uint16 = gather(axis = code_embed_45_axis_0, batch_dims = code_embed_45_batch_dims_0, indices = input_567_to_uint16, validate_indices = code_embed_45_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_45_cast_fp16_cast_uint16")]; + tensor var_19914 = const()[name = string("op_19914"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_47_cast_fp16 = reshape(shape = var_19914, x = code_embed_45_cast_fp16_cast_uint16)[name = string("code_embed_47_cast_fp16")]; + tensor embed_sum_25_cast_fp16 = add(x = embed_sum_23_cast_fp16, y = code_embed_47_cast_fp16)[name = string("embed_sum_25_cast_fp16")]; + string inputs_545_pad_type_0 = const()[name = string("inputs_545_pad_type_0"), val = string("valid")]; + tensor inputs_545_strides_0 = const()[name = string("inputs_545_strides_0"), val = tensor([1, 1])]; + tensor inputs_545_pad_0 = const()[name = string("inputs_545_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_545_dilations_0 = const()[name = string("inputs_545_dilations_0"), val = tensor([1, 1])]; + int32 inputs_545_groups_0 = const()[name = string("inputs_545_groups_0"), val = int32(1)]; + tensor inputs_545_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_545_dilations_0, groups = inputs_545_groups_0, pad = inputs_545_pad_0, pad_type = inputs_545_pad_type_0, strides = inputs_545_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_47_cast_fp16)[name = string("inputs_545_cast_fp16")]; + tensor obj_575_begin_0 = const()[name = string("obj_575_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_575_end_0 = const()[name = string("obj_575_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_575_end_mask_0 = const()[name = string("obj_575_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_575_cast_fp16 = slice_by_index(begin = obj_575_begin_0, end = obj_575_end_0, end_mask = obj_575_end_mask_0, x = key_caches_27_cast_fp16)[name = string("obj_575_cast_fp16")]; + tensor obj_577_begin_0 = const()[name = string("obj_577_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_577_end_0 = const()[name = string("obj_577_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_577_end_mask_0 = const()[name = string("obj_577_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_577_cast_fp16 = slice_by_index(begin = obj_577_begin_0, end = obj_577_end_0, end_mask = obj_577_end_mask_0, x = value_caches_27_cast_fp16)[name = string("obj_577_cast_fp16")]; + int32 var_20019 = const()[name = string("op_20019"), val = int32(3)]; + int32 var_20029 = const()[name = string("op_20029"), val = int32(-2)]; + tensor inputs_sq_545_cast_fp16 = mul(x = inputs_545_cast_fp16, y = inputs_545_cast_fp16)[name = string("inputs_sq_545_cast_fp16")]; + tensor variance_545_axes_0 = const()[name = string("variance_545_axes_0"), val = tensor([1])]; + bool variance_545_keep_dims_0 = const()[name = string("variance_545_keep_dims_0"), val = bool(true)]; + tensor variance_545_cast_fp16 = reduce_mean(axes = variance_545_axes_0, keep_dims = variance_545_keep_dims_0, x = inputs_sq_545_cast_fp16)[name = string("variance_545_cast_fp16")]; + fp16 var_20043_to_fp16 = const()[name = string("op_20043_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20044_cast_fp16 = add(x = variance_545_cast_fp16, y = var_20043_to_fp16)[name = string("op_20044_cast_fp16")]; + fp32 var_20045_epsilon_0 = const()[name = string("op_20045_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20045_cast_fp16 = rsqrt(epsilon = var_20045_epsilon_0, x = var_20044_cast_fp16)[name = string("op_20045_cast_fp16")]; + tensor hidden_states_675_cast_fp16 = mul(x = inputs_545_cast_fp16, y = var_20045_cast_fp16)[name = string("hidden_states_675_cast_fp16")]; + tensor obj_573_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_675_cast_fp16)[name = string("obj_573_cast_fp16")]; + string query_391_pad_type_0 = const()[name = string("query_391_pad_type_0"), val = string("valid")]; + tensor query_391_strides_0 = const()[name = string("query_391_strides_0"), val = tensor([1, 1])]; + tensor query_391_pad_0 = const()[name = string("query_391_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_391_dilations_0 = const()[name = string("query_391_dilations_0"), val = tensor([1, 1])]; + int32 query_391_groups_0 = const()[name = string("query_391_groups_0"), val = int32(1)]; + tensor query_391_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_391_dilations_0, groups = query_391_groups_0, pad = query_391_pad_0, pad_type = query_391_pad_type_0, strides = query_391_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_573_cast_fp16)[name = string("query_391_cast_fp16")]; + string current_key_261_pad_type_0 = const()[name = string("current_key_261_pad_type_0"), val = string("valid")]; + tensor current_key_261_strides_0 = const()[name = string("current_key_261_strides_0"), val = tensor([1, 1])]; + tensor current_key_261_pad_0 = const()[name = string("current_key_261_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_261_dilations_0 = const()[name = string("current_key_261_dilations_0"), val = tensor([1, 1])]; + int32 current_key_261_groups_0 = const()[name = string("current_key_261_groups_0"), val = int32(1)]; + tensor current_key_261_cast_fp16 = conv(dilations = current_key_261_dilations_0, groups = current_key_261_groups_0, pad = current_key_261_pad_0, pad_type = current_key_261_pad_type_0, strides = current_key_261_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_573_cast_fp16)[name = string("current_key_261_cast_fp16")]; + string current_value_131_pad_type_0 = const()[name = string("current_value_131_pad_type_0"), val = string("valid")]; + tensor current_value_131_strides_0 = const()[name = string("current_value_131_strides_0"), val = tensor([1, 1])]; + tensor current_value_131_pad_0 = const()[name = string("current_value_131_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_131_dilations_0 = const()[name = string("current_value_131_dilations_0"), val = tensor([1, 1])]; + int32 current_value_131_groups_0 = const()[name = string("current_value_131_groups_0"), val = int32(1)]; + tensor current_value_131_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_131_dilations_0, groups = current_value_131_groups_0, pad = current_value_131_pad_0, pad_type = current_value_131_pad_type_0, strides = current_value_131_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_573_cast_fp16)[name = string("current_value_131_cast_fp16")]; + tensor var_20082 = const()[name = string("op_20082"), val = tensor([16, 128, 1, 1])]; + tensor inputs_547_cast_fp16 = reshape(shape = var_20082, x = query_391_cast_fp16)[name = string("inputs_547_cast_fp16")]; + tensor inputs_sq_547_cast_fp16 = mul(x = inputs_547_cast_fp16, y = inputs_547_cast_fp16)[name = string("inputs_sq_547_cast_fp16")]; + tensor variance_547_axes_0 = const()[name = string("variance_547_axes_0"), val = tensor([1])]; + bool variance_547_keep_dims_0 = const()[name = string("variance_547_keep_dims_0"), val = bool(true)]; + tensor variance_547_cast_fp16 = reduce_mean(axes = variance_547_axes_0, keep_dims = variance_547_keep_dims_0, x = inputs_sq_547_cast_fp16)[name = string("variance_547_cast_fp16")]; + fp16 var_20088_to_fp16 = const()[name = string("op_20088_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20089_cast_fp16 = add(x = variance_547_cast_fp16, y = var_20088_to_fp16)[name = string("op_20089_cast_fp16")]; + fp32 var_20090_epsilon_0 = const()[name = string("op_20090_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20090_cast_fp16 = rsqrt(epsilon = var_20090_epsilon_0, x = var_20089_cast_fp16)[name = string("op_20090_cast_fp16")]; + tensor hidden_states_677_cast_fp16 = mul(x = inputs_547_cast_fp16, y = var_20090_cast_fp16)[name = string("hidden_states_677_cast_fp16")]; + tensor query_normed_131_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_677_cast_fp16)[name = string("query_normed_131_cast_fp16")]; + tensor var_20098 = const()[name = string("op_20098"), val = tensor([8, 128, 1, 1])]; + tensor inputs_549_cast_fp16 = reshape(shape = var_20098, x = current_key_261_cast_fp16)[name = string("inputs_549_cast_fp16")]; + tensor inputs_sq_549_cast_fp16 = mul(x = inputs_549_cast_fp16, y = inputs_549_cast_fp16)[name = string("inputs_sq_549_cast_fp16")]; + tensor variance_549_axes_0 = const()[name = string("variance_549_axes_0"), val = tensor([1])]; + bool variance_549_keep_dims_0 = const()[name = string("variance_549_keep_dims_0"), val = bool(true)]; + tensor variance_549_cast_fp16 = reduce_mean(axes = variance_549_axes_0, keep_dims = variance_549_keep_dims_0, x = inputs_sq_549_cast_fp16)[name = string("variance_549_cast_fp16")]; + fp16 var_20104_to_fp16 = const()[name = string("op_20104_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20105_cast_fp16 = add(x = variance_549_cast_fp16, y = var_20104_to_fp16)[name = string("op_20105_cast_fp16")]; + fp32 var_20106_epsilon_0 = const()[name = string("op_20106_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20106_cast_fp16 = rsqrt(epsilon = var_20106_epsilon_0, x = var_20105_cast_fp16)[name = string("op_20106_cast_fp16")]; + tensor hidden_states_679_cast_fp16 = mul(x = inputs_549_cast_fp16, y = var_20106_cast_fp16)[name = string("hidden_states_679_cast_fp16")]; + tensor current_key_normed_131_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_679_cast_fp16)[name = string("current_key_normed_131_cast_fp16")]; + tensor var_20124 = const()[name = string("op_20124"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_521_cast_fp16 = reshape(shape = var_20124, x = query_normed_131_cast_fp16)[name = string("mh_q_521_cast_fp16")]; + tensor var_20126 = const()[name = string("op_20126"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_521_cast_fp16 = reshape(shape = var_20126, x = current_key_normed_131_cast_fp16)[name = string("mh_k_521_cast_fp16")]; + tensor cos_131_to_fp16 = const()[name = string("cos_131_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175201856)))]; + tensor var_20130_cast_fp16 = mul(x = mh_q_521_cast_fp16, y = cos_131_to_fp16)[name = string("op_20130_cast_fp16")]; + tensor var_20135_begin_0 = const()[name = string("op_20135_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_20135_end_0 = const()[name = string("op_20135_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_20135_end_mask_0 = const()[name = string("op_20135_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_20135_cast_fp16 = slice_by_index(begin = var_20135_begin_0, end = var_20135_end_0, end_mask = var_20135_end_mask_0, x = mh_q_521_cast_fp16)[name = string("op_20135_cast_fp16")]; + tensor var_20141_begin_0 = const()[name = string("op_20141_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_20141_end_0 = const()[name = string("op_20141_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_20141_end_mask_0 = const()[name = string("op_20141_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_20141_cast_fp16 = slice_by_index(begin = var_20141_begin_0, end = var_20141_end_0, end_mask = var_20141_end_mask_0, x = mh_q_521_cast_fp16)[name = string("op_20141_cast_fp16")]; + fp16 const_1327_promoted_to_fp16 = const()[name = string("const_1327_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_20143_cast_fp16 = mul(x = var_20141_cast_fp16, y = const_1327_promoted_to_fp16)[name = string("op_20143_cast_fp16")]; + bool var_20145_interleave_0 = const()[name = string("op_20145_interleave_0"), val = bool(false)]; + tensor var_20145_cast_fp16 = concat(axis = var_20029, interleave = var_20145_interleave_0, values = (var_20143_cast_fp16, var_20135_cast_fp16))[name = string("op_20145_cast_fp16")]; + tensor sin_131_to_fp16 = const()[name = string("sin_131_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202176)))]; + tensor var_20146_cast_fp16 = mul(x = var_20145_cast_fp16, y = sin_131_to_fp16)[name = string("op_20146_cast_fp16")]; + tensor mh_q_523_cast_fp16 = add(x = var_20130_cast_fp16, y = var_20146_cast_fp16)[name = string("mh_q_523_cast_fp16")]; + tensor var_20148_cast_fp16 = mul(x = mh_k_521_cast_fp16, y = cos_131_to_fp16)[name = string("op_20148_cast_fp16")]; + tensor var_20153_begin_0 = const()[name = string("op_20153_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_20153_end_0 = const()[name = string("op_20153_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_20153_end_mask_0 = const()[name = string("op_20153_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_20153_cast_fp16 = slice_by_index(begin = var_20153_begin_0, end = var_20153_end_0, end_mask = var_20153_end_mask_0, x = mh_k_521_cast_fp16)[name = string("op_20153_cast_fp16")]; + tensor var_20159_begin_0 = const()[name = string("op_20159_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_20159_end_0 = const()[name = string("op_20159_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_20159_end_mask_0 = const()[name = string("op_20159_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_20159_cast_fp16 = slice_by_index(begin = var_20159_begin_0, end = var_20159_end_0, end_mask = var_20159_end_mask_0, x = mh_k_521_cast_fp16)[name = string("op_20159_cast_fp16")]; + fp16 const_1330_promoted_to_fp16 = const()[name = string("const_1330_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_20161_cast_fp16 = mul(x = var_20159_cast_fp16, y = const_1330_promoted_to_fp16)[name = string("op_20161_cast_fp16")]; + bool var_20163_interleave_0 = const()[name = string("op_20163_interleave_0"), val = bool(false)]; + tensor var_20163_cast_fp16 = concat(axis = var_20029, interleave = var_20163_interleave_0, values = (var_20161_cast_fp16, var_20153_cast_fp16))[name = string("op_20163_cast_fp16")]; + tensor var_20164_cast_fp16 = mul(x = var_20163_cast_fp16, y = sin_131_to_fp16)[name = string("op_20164_cast_fp16")]; + tensor mh_k_523_cast_fp16 = add(x = var_20148_cast_fp16, y = var_20164_cast_fp16)[name = string("mh_k_523_cast_fp16")]; + tensor var_20168 = const()[name = string("op_20168"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_263_cast_fp16 = reshape(shape = var_20168, x = mh_k_523_cast_fp16)[name = string("current_key_263_cast_fp16")]; + tensor var_20174_to_fp16 = const()[name = string("op_20174_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202496)))]; + tensor var_20175_cast_fp16 = mul(x = obj_575_cast_fp16, y = var_20174_to_fp16)[name = string("op_20175_cast_fp16")]; + tensor var_20172_to_fp16 = const()[name = string("op_20172_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202624)))]; + tensor var_20176_cast_fp16 = mul(x = current_key_263_cast_fp16, y = var_20172_to_fp16)[name = string("op_20176_cast_fp16")]; + tensor key_263_cast_fp16 = add(x = var_20175_cast_fp16, y = var_20176_cast_fp16)[name = string("key_263_cast_fp16")]; + tensor var_20178_to_fp16 = const()[name = string("op_20178_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202496)))]; + tensor var_20179_cast_fp16 = mul(x = obj_577_cast_fp16, y = var_20178_to_fp16)[name = string("op_20179_cast_fp16")]; + tensor var_20180_cast_fp16 = mul(x = current_value_131_cast_fp16, y = var_20172_to_fp16)[name = string("op_20180_cast_fp16")]; + tensor value_131_cast_fp16 = add(x = var_20179_cast_fp16, y = var_20180_cast_fp16)[name = string("value_131_cast_fp16")]; + fp16 var_20187_to_fp16 = const()[name = string("op_20187_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_527_cast_fp16 = mul(x = mh_q_523_cast_fp16, y = var_20187_to_fp16)[name = string("mh_q_527_cast_fp16")]; + tensor var_20189 = const()[name = string("op_20189"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_525_cast_fp16 = reshape(shape = var_20189, x = key_263_cast_fp16)[name = string("mh_k_525_cast_fp16")]; + tensor var_20191 = const()[name = string("op_20191"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_261_cast_fp16 = reshape(shape = var_20191, x = value_131_cast_fp16)[name = string("mh_v_261_cast_fp16")]; + tensor transpose_260_perm_0 = const()[name = string("transpose_260_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_130_reps_0 = const()[name = string("tile_130_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_260_cast_fp16 = transpose(perm = transpose_260_perm_0, x = mh_k_525_cast_fp16)[name = string("transpose_89")]; + tensor tile_130_cast_fp16 = tile(reps = tile_130_reps_0, x = transpose_260_cast_fp16)[name = string("tile_130_cast_fp16")]; + tensor concat_328 = const()[name = string("concat_328"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_260_cast_fp16 = reshape(shape = concat_328, x = tile_130_cast_fp16)[name = string("reshape_260_cast_fp16")]; + tensor transpose_261_perm_0 = const()[name = string("transpose_261_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_329 = const()[name = string("concat_329"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_261_cast_fp16 = transpose(perm = transpose_261_perm_0, x = reshape_260_cast_fp16)[name = string("transpose_88")]; + tensor reshape_261_cast_fp16 = reshape(shape = concat_329, x = transpose_261_cast_fp16)[name = string("reshape_261_cast_fp16")]; + tensor transpose_262_perm_0 = const()[name = string("transpose_262_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_131_reps_0 = const()[name = string("tile_131_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_262_cast_fp16 = transpose(perm = transpose_262_perm_0, x = mh_v_261_cast_fp16)[name = string("transpose_87")]; + tensor tile_131_cast_fp16 = tile(reps = tile_131_reps_0, x = transpose_262_cast_fp16)[name = string("tile_131_cast_fp16")]; + tensor concat_330 = const()[name = string("concat_330"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_262_cast_fp16 = reshape(shape = concat_330, x = tile_131_cast_fp16)[name = string("reshape_262_cast_fp16")]; + tensor transpose_263_perm_0 = const()[name = string("transpose_263_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_331 = const()[name = string("concat_331"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_263_cast_fp16 = transpose(perm = transpose_263_perm_0, x = reshape_262_cast_fp16)[name = string("transpose_86")]; + tensor reshape_263_cast_fp16 = reshape(shape = concat_331, x = transpose_263_cast_fp16)[name = string("reshape_263_cast_fp16")]; + tensor transpose_577_perm_0 = const()[name = string("transpose_577_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_391_transpose_x_1 = const()[name = string("mh_w_391_transpose_x_1"), val = bool(true)]; + bool mh_w_391_transpose_y_1 = const()[name = string("mh_w_391_transpose_y_1"), val = bool(false)]; + tensor transpose_577_cast_fp16 = transpose(perm = transpose_577_perm_0, x = reshape_261_cast_fp16)[name = string("transpose_85")]; + tensor mh_w_391_cast_fp16 = matmul(transpose_x = mh_w_391_transpose_x_1, transpose_y = mh_w_391_transpose_y_1, x = mh_q_527_cast_fp16, y = transpose_577_cast_fp16)[name = string("mh_w_391_cast_fp16")]; + tensor var_20199_to_fp16 = const()[name = string("op_20199_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202752)))]; + tensor mh_w_393_cast_fp16 = add(x = mh_w_391_cast_fp16, y = var_20199_to_fp16)[name = string("mh_w_393_cast_fp16")]; + tensor mh_w_395_cast_fp16 = softmax(axis = var_20019, x = mh_w_393_cast_fp16)[name = string("mh_w_395_cast_fp16")]; + tensor transpose_578_perm_0 = const()[name = string("transpose_578_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_131_transpose_x_1 = const()[name = string("attn_131_transpose_x_1"), val = bool(false)]; + bool attn_131_transpose_y_1 = const()[name = string("attn_131_transpose_y_1"), val = bool(true)]; + tensor transpose_578_cast_fp16 = transpose(perm = transpose_578_perm_0, x = reshape_263_cast_fp16)[name = string("transpose_84")]; + tensor attn_131_cast_fp16 = matmul(transpose_x = attn_131_transpose_x_1, transpose_y = attn_131_transpose_y_1, x = transpose_578_cast_fp16, y = mh_w_395_cast_fp16)[name = string("attn_131_cast_fp16")]; + tensor var_20205 = const()[name = string("op_20205"), val = tensor([1, 2048, 1, 1])]; + tensor input_569_cast_fp16 = reshape(shape = var_20205, x = attn_131_cast_fp16)[name = string("input_569_cast_fp16")]; + string obj_583_pad_type_0 = const()[name = string("obj_583_pad_type_0"), val = string("valid")]; + tensor obj_583_strides_0 = const()[name = string("obj_583_strides_0"), val = tensor([1, 1])]; + tensor obj_583_pad_0 = const()[name = string("obj_583_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_583_dilations_0 = const()[name = string("obj_583_dilations_0"), val = tensor([1, 1])]; + int32 obj_583_groups_0 = const()[name = string("obj_583_groups_0"), val = int32(1)]; + tensor obj_583_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_583_dilations_0, groups = obj_583_groups_0, pad = obj_583_pad_0, pad_type = obj_583_pad_type_0, strides = obj_583_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_569_cast_fp16)[name = string("obj_583_cast_fp16")]; + tensor inputs_551_cast_fp16 = add(x = inputs_545_cast_fp16, y = obj_583_cast_fp16)[name = string("inputs_551_cast_fp16")]; + tensor inputs_sq_551_cast_fp16 = mul(x = inputs_551_cast_fp16, y = inputs_551_cast_fp16)[name = string("inputs_sq_551_cast_fp16")]; + tensor variance_551_axes_0 = const()[name = string("variance_551_axes_0"), val = tensor([1])]; + bool variance_551_keep_dims_0 = const()[name = string("variance_551_keep_dims_0"), val = bool(true)]; + tensor variance_551_cast_fp16 = reduce_mean(axes = variance_551_axes_0, keep_dims = variance_551_keep_dims_0, x = inputs_sq_551_cast_fp16)[name = string("variance_551_cast_fp16")]; + fp16 var_20223_to_fp16 = const()[name = string("op_20223_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20224_cast_fp16 = add(x = variance_551_cast_fp16, y = var_20223_to_fp16)[name = string("op_20224_cast_fp16")]; + fp32 var_20225_epsilon_0 = const()[name = string("op_20225_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20225_cast_fp16 = rsqrt(epsilon = var_20225_epsilon_0, x = var_20224_cast_fp16)[name = string("op_20225_cast_fp16")]; + tensor hidden_states_681_cast_fp16 = mul(x = inputs_551_cast_fp16, y = var_20225_cast_fp16)[name = string("hidden_states_681_cast_fp16")]; + tensor input_571_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_681_cast_fp16)[name = string("input_571_cast_fp16")]; + string input_573_pad_type_0 = const()[name = string("input_573_pad_type_0"), val = string("valid")]; + tensor input_573_strides_0 = const()[name = string("input_573_strides_0"), val = tensor([1, 1])]; + tensor input_573_pad_0 = const()[name = string("input_573_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_573_dilations_0 = const()[name = string("input_573_dilations_0"), val = tensor([1, 1])]; + int32 input_573_groups_0 = const()[name = string("input_573_groups_0"), val = int32(1)]; + tensor input_573_cast_fp16 = conv(dilations = input_573_dilations_0, groups = input_573_groups_0, pad = input_573_pad_0, pad_type = input_573_pad_type_0, strides = input_573_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_571_cast_fp16)[name = string("input_573_cast_fp16")]; + tensor var_20239_cast_fp16 = silu(x = input_573_cast_fp16)[name = string("op_20239_cast_fp16")]; + string var_20245_pad_type_0 = const()[name = string("op_20245_pad_type_0"), val = string("valid")]; + tensor var_20245_strides_0 = const()[name = string("op_20245_strides_0"), val = tensor([1, 1])]; + tensor var_20245_pad_0 = const()[name = string("op_20245_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_20245_dilations_0 = const()[name = string("op_20245_dilations_0"), val = tensor([1, 1])]; + int32 var_20245_groups_0 = const()[name = string("op_20245_groups_0"), val = int32(1)]; + tensor var_20245_cast_fp16 = conv(dilations = var_20245_dilations_0, groups = var_20245_groups_0, pad = var_20245_pad_0, pad_type = var_20245_pad_type_0, strides = var_20245_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_571_cast_fp16)[name = string("op_20245_cast_fp16")]; + tensor input_575_cast_fp16 = mul(x = var_20239_cast_fp16, y = var_20245_cast_fp16)[name = string("input_575_cast_fp16")]; + string hidden_states_683_pad_type_0 = const()[name = string("hidden_states_683_pad_type_0"), val = string("valid")]; + tensor hidden_states_683_strides_0 = const()[name = string("hidden_states_683_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_683_pad_0 = const()[name = string("hidden_states_683_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_683_dilations_0 = const()[name = string("hidden_states_683_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_683_groups_0 = const()[name = string("hidden_states_683_groups_0"), val = int32(1)]; + tensor hidden_states_683_cast_fp16 = conv(dilations = hidden_states_683_dilations_0, groups = hidden_states_683_groups_0, pad = hidden_states_683_pad_0, pad_type = hidden_states_683_pad_type_0, strides = hidden_states_683_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_575_cast_fp16)[name = string("hidden_states_683_cast_fp16")]; + tensor inputs_553_cast_fp16 = add(x = inputs_551_cast_fp16, y = hidden_states_683_cast_fp16)[name = string("inputs_553_cast_fp16")]; + tensor obj_587_begin_0 = const()[name = string("obj_587_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_587_end_0 = const()[name = string("obj_587_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_587_end_mask_0 = const()[name = string("obj_587_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_587_cast_fp16 = slice_by_index(begin = obj_587_begin_0, end = obj_587_end_0, end_mask = obj_587_end_mask_0, x = key_caches_27_cast_fp16)[name = string("obj_587_cast_fp16")]; + tensor obj_589_begin_0 = const()[name = string("obj_589_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_589_end_0 = const()[name = string("obj_589_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_589_end_mask_0 = const()[name = string("obj_589_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_589_cast_fp16 = slice_by_index(begin = obj_589_begin_0, end = obj_589_end_0, end_mask = obj_589_end_mask_0, x = value_caches_27_cast_fp16)[name = string("obj_589_cast_fp16")]; + int32 var_20293 = const()[name = string("op_20293"), val = int32(3)]; + int32 var_20303 = const()[name = string("op_20303"), val = int32(-2)]; + tensor inputs_sq_553_cast_fp16 = mul(x = inputs_553_cast_fp16, y = inputs_553_cast_fp16)[name = string("inputs_sq_553_cast_fp16")]; + tensor variance_553_axes_0 = const()[name = string("variance_553_axes_0"), val = tensor([1])]; + bool variance_553_keep_dims_0 = const()[name = string("variance_553_keep_dims_0"), val = bool(true)]; + tensor variance_553_cast_fp16 = reduce_mean(axes = variance_553_axes_0, keep_dims = variance_553_keep_dims_0, x = inputs_sq_553_cast_fp16)[name = string("variance_553_cast_fp16")]; + fp16 var_20317_to_fp16 = const()[name = string("op_20317_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20318_cast_fp16 = add(x = variance_553_cast_fp16, y = var_20317_to_fp16)[name = string("op_20318_cast_fp16")]; + fp32 var_20319_epsilon_0 = const()[name = string("op_20319_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20319_cast_fp16 = rsqrt(epsilon = var_20319_epsilon_0, x = var_20318_cast_fp16)[name = string("op_20319_cast_fp16")]; + tensor hidden_states_685_cast_fp16 = mul(x = inputs_553_cast_fp16, y = var_20319_cast_fp16)[name = string("hidden_states_685_cast_fp16")]; + tensor obj_585_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_685_cast_fp16)[name = string("obj_585_cast_fp16")]; + string query_397_pad_type_0 = const()[name = string("query_397_pad_type_0"), val = string("valid")]; + tensor query_397_strides_0 = const()[name = string("query_397_strides_0"), val = tensor([1, 1])]; + tensor query_397_pad_0 = const()[name = string("query_397_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_397_dilations_0 = const()[name = string("query_397_dilations_0"), val = tensor([1, 1])]; + int32 query_397_groups_0 = const()[name = string("query_397_groups_0"), val = int32(1)]; + tensor query_397_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_397_dilations_0, groups = query_397_groups_0, pad = query_397_pad_0, pad_type = query_397_pad_type_0, strides = query_397_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_585_cast_fp16)[name = string("query_397_cast_fp16")]; + string current_key_265_pad_type_0 = const()[name = string("current_key_265_pad_type_0"), val = string("valid")]; + tensor current_key_265_strides_0 = const()[name = string("current_key_265_strides_0"), val = tensor([1, 1])]; + tensor current_key_265_pad_0 = const()[name = string("current_key_265_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_265_dilations_0 = const()[name = string("current_key_265_dilations_0"), val = tensor([1, 1])]; + int32 current_key_265_groups_0 = const()[name = string("current_key_265_groups_0"), val = int32(1)]; + tensor current_key_265_cast_fp16 = conv(dilations = current_key_265_dilations_0, groups = current_key_265_groups_0, pad = current_key_265_pad_0, pad_type = current_key_265_pad_type_0, strides = current_key_265_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_585_cast_fp16)[name = string("current_key_265_cast_fp16")]; + string current_value_133_pad_type_0 = const()[name = string("current_value_133_pad_type_0"), val = string("valid")]; + tensor current_value_133_strides_0 = const()[name = string("current_value_133_strides_0"), val = tensor([1, 1])]; + tensor current_value_133_pad_0 = const()[name = string("current_value_133_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_133_dilations_0 = const()[name = string("current_value_133_dilations_0"), val = tensor([1, 1])]; + int32 current_value_133_groups_0 = const()[name = string("current_value_133_groups_0"), val = int32(1)]; + tensor current_value_133_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_133_dilations_0, groups = current_value_133_groups_0, pad = current_value_133_pad_0, pad_type = current_value_133_pad_type_0, strides = current_value_133_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_585_cast_fp16)[name = string("current_value_133_cast_fp16")]; + tensor var_20356 = const()[name = string("op_20356"), val = tensor([16, 128, 1, 1])]; + tensor inputs_555_cast_fp16 = reshape(shape = var_20356, x = query_397_cast_fp16)[name = string("inputs_555_cast_fp16")]; + tensor inputs_sq_555_cast_fp16 = mul(x = inputs_555_cast_fp16, y = inputs_555_cast_fp16)[name = string("inputs_sq_555_cast_fp16")]; + tensor variance_555_axes_0 = const()[name = string("variance_555_axes_0"), val = tensor([1])]; + bool variance_555_keep_dims_0 = const()[name = string("variance_555_keep_dims_0"), val = bool(true)]; + tensor variance_555_cast_fp16 = reduce_mean(axes = variance_555_axes_0, keep_dims = variance_555_keep_dims_0, x = inputs_sq_555_cast_fp16)[name = string("variance_555_cast_fp16")]; + fp16 var_20362_to_fp16 = const()[name = string("op_20362_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20363_cast_fp16 = add(x = variance_555_cast_fp16, y = var_20362_to_fp16)[name = string("op_20363_cast_fp16")]; + fp32 var_20364_epsilon_0 = const()[name = string("op_20364_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20364_cast_fp16 = rsqrt(epsilon = var_20364_epsilon_0, x = var_20363_cast_fp16)[name = string("op_20364_cast_fp16")]; + tensor hidden_states_687_cast_fp16 = mul(x = inputs_555_cast_fp16, y = var_20364_cast_fp16)[name = string("hidden_states_687_cast_fp16")]; + tensor query_normed_133_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_687_cast_fp16)[name = string("query_normed_133_cast_fp16")]; + tensor var_20372 = const()[name = string("op_20372"), val = tensor([8, 128, 1, 1])]; + tensor inputs_557_cast_fp16 = reshape(shape = var_20372, x = current_key_265_cast_fp16)[name = string("inputs_557_cast_fp16")]; + tensor inputs_sq_557_cast_fp16 = mul(x = inputs_557_cast_fp16, y = inputs_557_cast_fp16)[name = string("inputs_sq_557_cast_fp16")]; + tensor variance_557_axes_0 = const()[name = string("variance_557_axes_0"), val = tensor([1])]; + bool variance_557_keep_dims_0 = const()[name = string("variance_557_keep_dims_0"), val = bool(true)]; + tensor variance_557_cast_fp16 = reduce_mean(axes = variance_557_axes_0, keep_dims = variance_557_keep_dims_0, x = inputs_sq_557_cast_fp16)[name = string("variance_557_cast_fp16")]; + fp16 var_20378_to_fp16 = const()[name = string("op_20378_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20379_cast_fp16 = add(x = variance_557_cast_fp16, y = var_20378_to_fp16)[name = string("op_20379_cast_fp16")]; + fp32 var_20380_epsilon_0 = const()[name = string("op_20380_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20380_cast_fp16 = rsqrt(epsilon = var_20380_epsilon_0, x = var_20379_cast_fp16)[name = string("op_20380_cast_fp16")]; + tensor hidden_states_689_cast_fp16 = mul(x = inputs_557_cast_fp16, y = var_20380_cast_fp16)[name = string("hidden_states_689_cast_fp16")]; + tensor current_key_normed_133_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_689_cast_fp16)[name = string("current_key_normed_133_cast_fp16")]; + tensor var_20398 = const()[name = string("op_20398"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_529_cast_fp16 = reshape(shape = var_20398, x = query_normed_133_cast_fp16)[name = string("mh_q_529_cast_fp16")]; + tensor var_20400 = const()[name = string("op_20400"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_529_cast_fp16 = reshape(shape = var_20400, x = current_key_normed_133_cast_fp16)[name = string("mh_k_529_cast_fp16")]; + tensor var_20404_cast_fp16 = mul(x = mh_q_529_cast_fp16, y = cos_131_to_fp16)[name = string("op_20404_cast_fp16")]; + tensor var_20409_begin_0 = const()[name = string("op_20409_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_20409_end_0 = const()[name = string("op_20409_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_20409_end_mask_0 = const()[name = string("op_20409_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_20409_cast_fp16 = slice_by_index(begin = var_20409_begin_0, end = var_20409_end_0, end_mask = var_20409_end_mask_0, x = mh_q_529_cast_fp16)[name = string("op_20409_cast_fp16")]; + tensor var_20415_begin_0 = const()[name = string("op_20415_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_20415_end_0 = const()[name = string("op_20415_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_20415_end_mask_0 = const()[name = string("op_20415_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_20415_cast_fp16 = slice_by_index(begin = var_20415_begin_0, end = var_20415_end_0, end_mask = var_20415_end_mask_0, x = mh_q_529_cast_fp16)[name = string("op_20415_cast_fp16")]; + fp16 const_1347_promoted_to_fp16 = const()[name = string("const_1347_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_20417_cast_fp16 = mul(x = var_20415_cast_fp16, y = const_1347_promoted_to_fp16)[name = string("op_20417_cast_fp16")]; + bool var_20419_interleave_0 = const()[name = string("op_20419_interleave_0"), val = bool(false)]; + tensor var_20419_cast_fp16 = concat(axis = var_20303, interleave = var_20419_interleave_0, values = (var_20417_cast_fp16, var_20409_cast_fp16))[name = string("op_20419_cast_fp16")]; + tensor var_20420_cast_fp16 = mul(x = var_20419_cast_fp16, y = sin_131_to_fp16)[name = string("op_20420_cast_fp16")]; + tensor mh_q_531_cast_fp16 = add(x = var_20404_cast_fp16, y = var_20420_cast_fp16)[name = string("mh_q_531_cast_fp16")]; + tensor var_20422_cast_fp16 = mul(x = mh_k_529_cast_fp16, y = cos_131_to_fp16)[name = string("op_20422_cast_fp16")]; + tensor var_20427_begin_0 = const()[name = string("op_20427_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_20427_end_0 = const()[name = string("op_20427_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_20427_end_mask_0 = const()[name = string("op_20427_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_20427_cast_fp16 = slice_by_index(begin = var_20427_begin_0, end = var_20427_end_0, end_mask = var_20427_end_mask_0, x = mh_k_529_cast_fp16)[name = string("op_20427_cast_fp16")]; + tensor var_20433_begin_0 = const()[name = string("op_20433_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_20433_end_0 = const()[name = string("op_20433_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_20433_end_mask_0 = const()[name = string("op_20433_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_20433_cast_fp16 = slice_by_index(begin = var_20433_begin_0, end = var_20433_end_0, end_mask = var_20433_end_mask_0, x = mh_k_529_cast_fp16)[name = string("op_20433_cast_fp16")]; + fp16 const_1350_promoted_to_fp16 = const()[name = string("const_1350_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_20435_cast_fp16 = mul(x = var_20433_cast_fp16, y = const_1350_promoted_to_fp16)[name = string("op_20435_cast_fp16")]; + bool var_20437_interleave_0 = const()[name = string("op_20437_interleave_0"), val = bool(false)]; + tensor var_20437_cast_fp16 = concat(axis = var_20303, interleave = var_20437_interleave_0, values = (var_20435_cast_fp16, var_20427_cast_fp16))[name = string("op_20437_cast_fp16")]; + tensor var_20438_cast_fp16 = mul(x = var_20437_cast_fp16, y = sin_131_to_fp16)[name = string("op_20438_cast_fp16")]; + tensor mh_k_531_cast_fp16 = add(x = var_20422_cast_fp16, y = var_20438_cast_fp16)[name = string("mh_k_531_cast_fp16")]; + tensor var_20442 = const()[name = string("op_20442"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_267_cast_fp16 = reshape(shape = var_20442, x = mh_k_531_cast_fp16)[name = string("current_key_267_cast_fp16")]; + tensor var_20448_to_fp16 = const()[name = string("op_20448_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202496)))]; + tensor var_20449_cast_fp16 = mul(x = obj_587_cast_fp16, y = var_20448_to_fp16)[name = string("op_20449_cast_fp16")]; + tensor var_20446_to_fp16 = const()[name = string("op_20446_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202624)))]; + tensor var_20450_cast_fp16 = mul(x = current_key_267_cast_fp16, y = var_20446_to_fp16)[name = string("op_20450_cast_fp16")]; + tensor key_267_cast_fp16 = add(x = var_20449_cast_fp16, y = var_20450_cast_fp16)[name = string("key_267_cast_fp16")]; + tensor var_20452_to_fp16 = const()[name = string("op_20452_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202496)))]; + tensor var_20453_cast_fp16 = mul(x = obj_589_cast_fp16, y = var_20452_to_fp16)[name = string("op_20453_cast_fp16")]; + tensor var_20454_cast_fp16 = mul(x = current_value_133_cast_fp16, y = var_20446_to_fp16)[name = string("op_20454_cast_fp16")]; + tensor value_133_cast_fp16 = add(x = var_20453_cast_fp16, y = var_20454_cast_fp16)[name = string("value_133_cast_fp16")]; + fp16 var_20461_to_fp16 = const()[name = string("op_20461_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_535_cast_fp16 = mul(x = mh_q_531_cast_fp16, y = var_20461_to_fp16)[name = string("mh_q_535_cast_fp16")]; + tensor var_20463 = const()[name = string("op_20463"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_533_cast_fp16 = reshape(shape = var_20463, x = key_267_cast_fp16)[name = string("mh_k_533_cast_fp16")]; + tensor var_20465 = const()[name = string("op_20465"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_265_cast_fp16 = reshape(shape = var_20465, x = value_133_cast_fp16)[name = string("mh_v_265_cast_fp16")]; + tensor transpose_264_perm_0 = const()[name = string("transpose_264_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_132_reps_0 = const()[name = string("tile_132_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_264_cast_fp16 = transpose(perm = transpose_264_perm_0, x = mh_k_533_cast_fp16)[name = string("transpose_83")]; + tensor tile_132_cast_fp16 = tile(reps = tile_132_reps_0, x = transpose_264_cast_fp16)[name = string("tile_132_cast_fp16")]; + tensor concat_332 = const()[name = string("concat_332"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_264_cast_fp16 = reshape(shape = concat_332, x = tile_132_cast_fp16)[name = string("reshape_264_cast_fp16")]; + tensor transpose_265_perm_0 = const()[name = string("transpose_265_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_333 = const()[name = string("concat_333"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_265_cast_fp16 = transpose(perm = transpose_265_perm_0, x = reshape_264_cast_fp16)[name = string("transpose_82")]; + tensor reshape_265_cast_fp16 = reshape(shape = concat_333, x = transpose_265_cast_fp16)[name = string("reshape_265_cast_fp16")]; + tensor transpose_266_perm_0 = const()[name = string("transpose_266_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_133_reps_0 = const()[name = string("tile_133_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_266_cast_fp16 = transpose(perm = transpose_266_perm_0, x = mh_v_265_cast_fp16)[name = string("transpose_81")]; + tensor tile_133_cast_fp16 = tile(reps = tile_133_reps_0, x = transpose_266_cast_fp16)[name = string("tile_133_cast_fp16")]; + tensor concat_334 = const()[name = string("concat_334"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_266_cast_fp16 = reshape(shape = concat_334, x = tile_133_cast_fp16)[name = string("reshape_266_cast_fp16")]; + tensor transpose_267_perm_0 = const()[name = string("transpose_267_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_335 = const()[name = string("concat_335"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_267_cast_fp16 = transpose(perm = transpose_267_perm_0, x = reshape_266_cast_fp16)[name = string("transpose_80")]; + tensor reshape_267_cast_fp16 = reshape(shape = concat_335, x = transpose_267_cast_fp16)[name = string("reshape_267_cast_fp16")]; + tensor transpose_581_perm_0 = const()[name = string("transpose_581_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_397_transpose_x_1 = const()[name = string("mh_w_397_transpose_x_1"), val = bool(true)]; + bool mh_w_397_transpose_y_1 = const()[name = string("mh_w_397_transpose_y_1"), val = bool(false)]; + tensor transpose_581_cast_fp16 = transpose(perm = transpose_581_perm_0, x = reshape_265_cast_fp16)[name = string("transpose_79")]; + tensor mh_w_397_cast_fp16 = matmul(transpose_x = mh_w_397_transpose_x_1, transpose_y = mh_w_397_transpose_y_1, x = mh_q_535_cast_fp16, y = transpose_581_cast_fp16)[name = string("mh_w_397_cast_fp16")]; + tensor var_20473_to_fp16 = const()[name = string("op_20473_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202752)))]; + tensor mh_w_399_cast_fp16 = add(x = mh_w_397_cast_fp16, y = var_20473_to_fp16)[name = string("mh_w_399_cast_fp16")]; + tensor mh_w_401_cast_fp16 = softmax(axis = var_20293, x = mh_w_399_cast_fp16)[name = string("mh_w_401_cast_fp16")]; + tensor transpose_582_perm_0 = const()[name = string("transpose_582_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_133_transpose_x_1 = const()[name = string("attn_133_transpose_x_1"), val = bool(false)]; + bool attn_133_transpose_y_1 = const()[name = string("attn_133_transpose_y_1"), val = bool(true)]; + tensor transpose_582_cast_fp16 = transpose(perm = transpose_582_perm_0, x = reshape_267_cast_fp16)[name = string("transpose_78")]; + tensor attn_133_cast_fp16 = matmul(transpose_x = attn_133_transpose_x_1, transpose_y = attn_133_transpose_y_1, x = transpose_582_cast_fp16, y = mh_w_401_cast_fp16)[name = string("attn_133_cast_fp16")]; + tensor var_20479 = const()[name = string("op_20479"), val = tensor([1, 2048, 1, 1])]; + tensor input_577_cast_fp16 = reshape(shape = var_20479, x = attn_133_cast_fp16)[name = string("input_577_cast_fp16")]; + string obj_591_pad_type_0 = const()[name = string("obj_591_pad_type_0"), val = string("valid")]; + tensor obj_591_strides_0 = const()[name = string("obj_591_strides_0"), val = tensor([1, 1])]; + tensor obj_591_pad_0 = const()[name = string("obj_591_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_591_dilations_0 = const()[name = string("obj_591_dilations_0"), val = tensor([1, 1])]; + int32 obj_591_groups_0 = const()[name = string("obj_591_groups_0"), val = int32(1)]; + tensor obj_591_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_591_dilations_0, groups = obj_591_groups_0, pad = obj_591_pad_0, pad_type = obj_591_pad_type_0, strides = obj_591_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_577_cast_fp16)[name = string("obj_591_cast_fp16")]; + tensor inputs_559_cast_fp16 = add(x = inputs_553_cast_fp16, y = obj_591_cast_fp16)[name = string("inputs_559_cast_fp16")]; + tensor inputs_sq_559_cast_fp16 = mul(x = inputs_559_cast_fp16, y = inputs_559_cast_fp16)[name = string("inputs_sq_559_cast_fp16")]; + tensor variance_559_axes_0 = const()[name = string("variance_559_axes_0"), val = tensor([1])]; + bool variance_559_keep_dims_0 = const()[name = string("variance_559_keep_dims_0"), val = bool(true)]; + tensor variance_559_cast_fp16 = reduce_mean(axes = variance_559_axes_0, keep_dims = variance_559_keep_dims_0, x = inputs_sq_559_cast_fp16)[name = string("variance_559_cast_fp16")]; + fp16 var_20497_to_fp16 = const()[name = string("op_20497_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20498_cast_fp16 = add(x = variance_559_cast_fp16, y = var_20497_to_fp16)[name = string("op_20498_cast_fp16")]; + fp32 var_20499_epsilon_0 = const()[name = string("op_20499_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20499_cast_fp16 = rsqrt(epsilon = var_20499_epsilon_0, x = var_20498_cast_fp16)[name = string("op_20499_cast_fp16")]; + tensor hidden_states_691_cast_fp16 = mul(x = inputs_559_cast_fp16, y = var_20499_cast_fp16)[name = string("hidden_states_691_cast_fp16")]; + tensor input_579_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_691_cast_fp16)[name = string("input_579_cast_fp16")]; + string input_581_pad_type_0 = const()[name = string("input_581_pad_type_0"), val = string("valid")]; + tensor input_581_strides_0 = const()[name = string("input_581_strides_0"), val = tensor([1, 1])]; + tensor input_581_pad_0 = const()[name = string("input_581_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_581_dilations_0 = const()[name = string("input_581_dilations_0"), val = tensor([1, 1])]; + int32 input_581_groups_0 = const()[name = string("input_581_groups_0"), val = int32(1)]; + tensor input_581_cast_fp16 = conv(dilations = input_581_dilations_0, groups = input_581_groups_0, pad = input_581_pad_0, pad_type = input_581_pad_type_0, strides = input_581_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_579_cast_fp16)[name = string("input_581_cast_fp16")]; + tensor var_20513_cast_fp16 = silu(x = input_581_cast_fp16)[name = string("op_20513_cast_fp16")]; + string var_20519_pad_type_0 = const()[name = string("op_20519_pad_type_0"), val = string("valid")]; + tensor var_20519_strides_0 = const()[name = string("op_20519_strides_0"), val = tensor([1, 1])]; + tensor var_20519_pad_0 = const()[name = string("op_20519_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_20519_dilations_0 = const()[name = string("op_20519_dilations_0"), val = tensor([1, 1])]; + int32 var_20519_groups_0 = const()[name = string("op_20519_groups_0"), val = int32(1)]; + tensor var_20519_cast_fp16 = conv(dilations = var_20519_dilations_0, groups = var_20519_groups_0, pad = var_20519_pad_0, pad_type = var_20519_pad_type_0, strides = var_20519_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_579_cast_fp16)[name = string("op_20519_cast_fp16")]; + tensor input_583_cast_fp16 = mul(x = var_20513_cast_fp16, y = var_20519_cast_fp16)[name = string("input_583_cast_fp16")]; + string hidden_states_693_pad_type_0 = const()[name = string("hidden_states_693_pad_type_0"), val = string("valid")]; + tensor hidden_states_693_strides_0 = const()[name = string("hidden_states_693_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_693_pad_0 = const()[name = string("hidden_states_693_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_693_dilations_0 = const()[name = string("hidden_states_693_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_693_groups_0 = const()[name = string("hidden_states_693_groups_0"), val = int32(1)]; + tensor hidden_states_693_cast_fp16 = conv(dilations = hidden_states_693_dilations_0, groups = hidden_states_693_groups_0, pad = hidden_states_693_pad_0, pad_type = hidden_states_693_pad_type_0, strides = hidden_states_693_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_583_cast_fp16)[name = string("hidden_states_693_cast_fp16")]; + tensor inputs_561_cast_fp16 = add(x = inputs_559_cast_fp16, y = hidden_states_693_cast_fp16)[name = string("inputs_561_cast_fp16")]; + tensor obj_595_begin_0 = const()[name = string("obj_595_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_595_end_0 = const()[name = string("obj_595_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_595_end_mask_0 = const()[name = string("obj_595_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_595_cast_fp16 = slice_by_index(begin = obj_595_begin_0, end = obj_595_end_0, end_mask = obj_595_end_mask_0, x = key_caches_27_cast_fp16)[name = string("obj_595_cast_fp16")]; + tensor obj_597_begin_0 = const()[name = string("obj_597_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_597_end_0 = const()[name = string("obj_597_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_597_end_mask_0 = const()[name = string("obj_597_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_597_cast_fp16 = slice_by_index(begin = obj_597_begin_0, end = obj_597_end_0, end_mask = obj_597_end_mask_0, x = value_caches_27_cast_fp16)[name = string("obj_597_cast_fp16")]; + int32 var_20567 = const()[name = string("op_20567"), val = int32(3)]; + int32 var_20577 = const()[name = string("op_20577"), val = int32(-2)]; + tensor inputs_sq_561_cast_fp16 = mul(x = inputs_561_cast_fp16, y = inputs_561_cast_fp16)[name = string("inputs_sq_561_cast_fp16")]; + tensor variance_561_axes_0 = const()[name = string("variance_561_axes_0"), val = tensor([1])]; + bool variance_561_keep_dims_0 = const()[name = string("variance_561_keep_dims_0"), val = bool(true)]; + tensor variance_561_cast_fp16 = reduce_mean(axes = variance_561_axes_0, keep_dims = variance_561_keep_dims_0, x = inputs_sq_561_cast_fp16)[name = string("variance_561_cast_fp16")]; + fp16 var_20591_to_fp16 = const()[name = string("op_20591_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20592_cast_fp16 = add(x = variance_561_cast_fp16, y = var_20591_to_fp16)[name = string("op_20592_cast_fp16")]; + fp32 var_20593_epsilon_0 = const()[name = string("op_20593_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20593_cast_fp16 = rsqrt(epsilon = var_20593_epsilon_0, x = var_20592_cast_fp16)[name = string("op_20593_cast_fp16")]; + tensor hidden_states_695_cast_fp16 = mul(x = inputs_561_cast_fp16, y = var_20593_cast_fp16)[name = string("hidden_states_695_cast_fp16")]; + tensor obj_593_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_695_cast_fp16)[name = string("obj_593_cast_fp16")]; + string query_403_pad_type_0 = const()[name = string("query_403_pad_type_0"), val = string("valid")]; + tensor query_403_strides_0 = const()[name = string("query_403_strides_0"), val = tensor([1, 1])]; + tensor query_403_pad_0 = const()[name = string("query_403_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_403_dilations_0 = const()[name = string("query_403_dilations_0"), val = tensor([1, 1])]; + int32 query_403_groups_0 = const()[name = string("query_403_groups_0"), val = int32(1)]; + tensor query_403_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_403_dilations_0, groups = query_403_groups_0, pad = query_403_pad_0, pad_type = query_403_pad_type_0, strides = query_403_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_593_cast_fp16)[name = string("query_403_cast_fp16")]; + string current_key_269_pad_type_0 = const()[name = string("current_key_269_pad_type_0"), val = string("valid")]; + tensor current_key_269_strides_0 = const()[name = string("current_key_269_strides_0"), val = tensor([1, 1])]; + tensor current_key_269_pad_0 = const()[name = string("current_key_269_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_269_dilations_0 = const()[name = string("current_key_269_dilations_0"), val = tensor([1, 1])]; + int32 current_key_269_groups_0 = const()[name = string("current_key_269_groups_0"), val = int32(1)]; + tensor current_key_269_cast_fp16 = conv(dilations = current_key_269_dilations_0, groups = current_key_269_groups_0, pad = current_key_269_pad_0, pad_type = current_key_269_pad_type_0, strides = current_key_269_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_593_cast_fp16)[name = string("current_key_269_cast_fp16")]; + string current_value_135_pad_type_0 = const()[name = string("current_value_135_pad_type_0"), val = string("valid")]; + tensor current_value_135_strides_0 = const()[name = string("current_value_135_strides_0"), val = tensor([1, 1])]; + tensor current_value_135_pad_0 = const()[name = string("current_value_135_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_135_dilations_0 = const()[name = string("current_value_135_dilations_0"), val = tensor([1, 1])]; + int32 current_value_135_groups_0 = const()[name = string("current_value_135_groups_0"), val = int32(1)]; + tensor current_value_135_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_135_dilations_0, groups = current_value_135_groups_0, pad = current_value_135_pad_0, pad_type = current_value_135_pad_type_0, strides = current_value_135_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_593_cast_fp16)[name = string("current_value_135_cast_fp16")]; + tensor var_20630 = const()[name = string("op_20630"), val = tensor([16, 128, 1, 1])]; + tensor inputs_563_cast_fp16 = reshape(shape = var_20630, x = query_403_cast_fp16)[name = string("inputs_563_cast_fp16")]; + tensor inputs_sq_563_cast_fp16 = mul(x = inputs_563_cast_fp16, y = inputs_563_cast_fp16)[name = string("inputs_sq_563_cast_fp16")]; + tensor variance_563_axes_0 = const()[name = string("variance_563_axes_0"), val = tensor([1])]; + bool variance_563_keep_dims_0 = const()[name = string("variance_563_keep_dims_0"), val = bool(true)]; + tensor variance_563_cast_fp16 = reduce_mean(axes = variance_563_axes_0, keep_dims = variance_563_keep_dims_0, x = inputs_sq_563_cast_fp16)[name = string("variance_563_cast_fp16")]; + fp16 var_20636_to_fp16 = const()[name = string("op_20636_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20637_cast_fp16 = add(x = variance_563_cast_fp16, y = var_20636_to_fp16)[name = string("op_20637_cast_fp16")]; + fp32 var_20638_epsilon_0 = const()[name = string("op_20638_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20638_cast_fp16 = rsqrt(epsilon = var_20638_epsilon_0, x = var_20637_cast_fp16)[name = string("op_20638_cast_fp16")]; + tensor hidden_states_697_cast_fp16 = mul(x = inputs_563_cast_fp16, y = var_20638_cast_fp16)[name = string("hidden_states_697_cast_fp16")]; + tensor query_normed_135_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_697_cast_fp16)[name = string("query_normed_135_cast_fp16")]; + tensor var_20646 = const()[name = string("op_20646"), val = tensor([8, 128, 1, 1])]; + tensor inputs_565_cast_fp16 = reshape(shape = var_20646, x = current_key_269_cast_fp16)[name = string("inputs_565_cast_fp16")]; + tensor inputs_sq_565_cast_fp16 = mul(x = inputs_565_cast_fp16, y = inputs_565_cast_fp16)[name = string("inputs_sq_565_cast_fp16")]; + tensor variance_565_axes_0 = const()[name = string("variance_565_axes_0"), val = tensor([1])]; + bool variance_565_keep_dims_0 = const()[name = string("variance_565_keep_dims_0"), val = bool(true)]; + tensor variance_565_cast_fp16 = reduce_mean(axes = variance_565_axes_0, keep_dims = variance_565_keep_dims_0, x = inputs_sq_565_cast_fp16)[name = string("variance_565_cast_fp16")]; + fp16 var_20652_to_fp16 = const()[name = string("op_20652_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20653_cast_fp16 = add(x = variance_565_cast_fp16, y = var_20652_to_fp16)[name = string("op_20653_cast_fp16")]; + fp32 var_20654_epsilon_0 = const()[name = string("op_20654_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20654_cast_fp16 = rsqrt(epsilon = var_20654_epsilon_0, x = var_20653_cast_fp16)[name = string("op_20654_cast_fp16")]; + tensor hidden_states_699_cast_fp16 = mul(x = inputs_565_cast_fp16, y = var_20654_cast_fp16)[name = string("hidden_states_699_cast_fp16")]; + tensor current_key_normed_135_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_699_cast_fp16)[name = string("current_key_normed_135_cast_fp16")]; + tensor var_20672 = const()[name = string("op_20672"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_537_cast_fp16 = reshape(shape = var_20672, x = query_normed_135_cast_fp16)[name = string("mh_q_537_cast_fp16")]; + tensor var_20674 = const()[name = string("op_20674"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_537_cast_fp16 = reshape(shape = var_20674, x = current_key_normed_135_cast_fp16)[name = string("mh_k_537_cast_fp16")]; + tensor var_20678_cast_fp16 = mul(x = mh_q_537_cast_fp16, y = cos_131_to_fp16)[name = string("op_20678_cast_fp16")]; + tensor var_20683_begin_0 = const()[name = string("op_20683_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_20683_end_0 = const()[name = string("op_20683_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_20683_end_mask_0 = const()[name = string("op_20683_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_20683_cast_fp16 = slice_by_index(begin = var_20683_begin_0, end = var_20683_end_0, end_mask = var_20683_end_mask_0, x = mh_q_537_cast_fp16)[name = string("op_20683_cast_fp16")]; + tensor var_20689_begin_0 = const()[name = string("op_20689_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_20689_end_0 = const()[name = string("op_20689_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_20689_end_mask_0 = const()[name = string("op_20689_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_20689_cast_fp16 = slice_by_index(begin = var_20689_begin_0, end = var_20689_end_0, end_mask = var_20689_end_mask_0, x = mh_q_537_cast_fp16)[name = string("op_20689_cast_fp16")]; + fp16 const_1367_promoted_to_fp16 = const()[name = string("const_1367_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_20691_cast_fp16 = mul(x = var_20689_cast_fp16, y = const_1367_promoted_to_fp16)[name = string("op_20691_cast_fp16")]; + bool var_20693_interleave_0 = const()[name = string("op_20693_interleave_0"), val = bool(false)]; + tensor var_20693_cast_fp16 = concat(axis = var_20577, interleave = var_20693_interleave_0, values = (var_20691_cast_fp16, var_20683_cast_fp16))[name = string("op_20693_cast_fp16")]; + tensor var_20694_cast_fp16 = mul(x = var_20693_cast_fp16, y = sin_131_to_fp16)[name = string("op_20694_cast_fp16")]; + tensor mh_q_539_cast_fp16 = add(x = var_20678_cast_fp16, y = var_20694_cast_fp16)[name = string("mh_q_539_cast_fp16")]; + tensor var_20696_cast_fp16 = mul(x = mh_k_537_cast_fp16, y = cos_131_to_fp16)[name = string("op_20696_cast_fp16")]; + tensor var_20701_begin_0 = const()[name = string("op_20701_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_20701_end_0 = const()[name = string("op_20701_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_20701_end_mask_0 = const()[name = string("op_20701_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_20701_cast_fp16 = slice_by_index(begin = var_20701_begin_0, end = var_20701_end_0, end_mask = var_20701_end_mask_0, x = mh_k_537_cast_fp16)[name = string("op_20701_cast_fp16")]; + tensor var_20707_begin_0 = const()[name = string("op_20707_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_20707_end_0 = const()[name = string("op_20707_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_20707_end_mask_0 = const()[name = string("op_20707_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_20707_cast_fp16 = slice_by_index(begin = var_20707_begin_0, end = var_20707_end_0, end_mask = var_20707_end_mask_0, x = mh_k_537_cast_fp16)[name = string("op_20707_cast_fp16")]; + fp16 const_1370_promoted_to_fp16 = const()[name = string("const_1370_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_20709_cast_fp16 = mul(x = var_20707_cast_fp16, y = const_1370_promoted_to_fp16)[name = string("op_20709_cast_fp16")]; + bool var_20711_interleave_0 = const()[name = string("op_20711_interleave_0"), val = bool(false)]; + tensor var_20711_cast_fp16 = concat(axis = var_20577, interleave = var_20711_interleave_0, values = (var_20709_cast_fp16, var_20701_cast_fp16))[name = string("op_20711_cast_fp16")]; + tensor var_20712_cast_fp16 = mul(x = var_20711_cast_fp16, y = sin_131_to_fp16)[name = string("op_20712_cast_fp16")]; + tensor mh_k_539_cast_fp16 = add(x = var_20696_cast_fp16, y = var_20712_cast_fp16)[name = string("mh_k_539_cast_fp16")]; + tensor var_20716 = const()[name = string("op_20716"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_271_cast_fp16 = reshape(shape = var_20716, x = mh_k_539_cast_fp16)[name = string("current_key_271_cast_fp16")]; + tensor var_20722_to_fp16 = const()[name = string("op_20722_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202496)))]; + tensor var_20723_cast_fp16 = mul(x = obj_595_cast_fp16, y = var_20722_to_fp16)[name = string("op_20723_cast_fp16")]; + tensor var_20720_to_fp16 = const()[name = string("op_20720_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202624)))]; + tensor var_20724_cast_fp16 = mul(x = current_key_271_cast_fp16, y = var_20720_to_fp16)[name = string("op_20724_cast_fp16")]; + tensor key_271_cast_fp16 = add(x = var_20723_cast_fp16, y = var_20724_cast_fp16)[name = string("key_271_cast_fp16")]; + tensor var_20726_to_fp16 = const()[name = string("op_20726_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202496)))]; + tensor var_20727_cast_fp16 = mul(x = obj_597_cast_fp16, y = var_20726_to_fp16)[name = string("op_20727_cast_fp16")]; + tensor var_20728_cast_fp16 = mul(x = current_value_135_cast_fp16, y = var_20720_to_fp16)[name = string("op_20728_cast_fp16")]; + tensor value_135_cast_fp16 = add(x = var_20727_cast_fp16, y = var_20728_cast_fp16)[name = string("value_135_cast_fp16")]; + fp16 var_20735_to_fp16 = const()[name = string("op_20735_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_543_cast_fp16 = mul(x = mh_q_539_cast_fp16, y = var_20735_to_fp16)[name = string("mh_q_543_cast_fp16")]; + tensor var_20737 = const()[name = string("op_20737"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_541_cast_fp16 = reshape(shape = var_20737, x = key_271_cast_fp16)[name = string("mh_k_541_cast_fp16")]; + tensor var_20739 = const()[name = string("op_20739"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_269_cast_fp16 = reshape(shape = var_20739, x = value_135_cast_fp16)[name = string("mh_v_269_cast_fp16")]; + tensor transpose_268_perm_0 = const()[name = string("transpose_268_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_134_reps_0 = const()[name = string("tile_134_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_268_cast_fp16 = transpose(perm = transpose_268_perm_0, x = mh_k_541_cast_fp16)[name = string("transpose_77")]; + tensor tile_134_cast_fp16 = tile(reps = tile_134_reps_0, x = transpose_268_cast_fp16)[name = string("tile_134_cast_fp16")]; + tensor concat_336 = const()[name = string("concat_336"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_268_cast_fp16 = reshape(shape = concat_336, x = tile_134_cast_fp16)[name = string("reshape_268_cast_fp16")]; + tensor transpose_269_perm_0 = const()[name = string("transpose_269_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_337 = const()[name = string("concat_337"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_269_cast_fp16 = transpose(perm = transpose_269_perm_0, x = reshape_268_cast_fp16)[name = string("transpose_76")]; + tensor reshape_269_cast_fp16 = reshape(shape = concat_337, x = transpose_269_cast_fp16)[name = string("reshape_269_cast_fp16")]; + tensor transpose_270_perm_0 = const()[name = string("transpose_270_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_135_reps_0 = const()[name = string("tile_135_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_270_cast_fp16 = transpose(perm = transpose_270_perm_0, x = mh_v_269_cast_fp16)[name = string("transpose_75")]; + tensor tile_135_cast_fp16 = tile(reps = tile_135_reps_0, x = transpose_270_cast_fp16)[name = string("tile_135_cast_fp16")]; + tensor concat_338 = const()[name = string("concat_338"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_270_cast_fp16 = reshape(shape = concat_338, x = tile_135_cast_fp16)[name = string("reshape_270_cast_fp16")]; + tensor transpose_271_perm_0 = const()[name = string("transpose_271_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_339 = const()[name = string("concat_339"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_271_cast_fp16 = transpose(perm = transpose_271_perm_0, x = reshape_270_cast_fp16)[name = string("transpose_74")]; + tensor reshape_271_cast_fp16 = reshape(shape = concat_339, x = transpose_271_cast_fp16)[name = string("reshape_271_cast_fp16")]; + tensor transpose_585_perm_0 = const()[name = string("transpose_585_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_403_transpose_x_1 = const()[name = string("mh_w_403_transpose_x_1"), val = bool(true)]; + bool mh_w_403_transpose_y_1 = const()[name = string("mh_w_403_transpose_y_1"), val = bool(false)]; + tensor transpose_585_cast_fp16 = transpose(perm = transpose_585_perm_0, x = reshape_269_cast_fp16)[name = string("transpose_73")]; + tensor mh_w_403_cast_fp16 = matmul(transpose_x = mh_w_403_transpose_x_1, transpose_y = mh_w_403_transpose_y_1, x = mh_q_543_cast_fp16, y = transpose_585_cast_fp16)[name = string("mh_w_403_cast_fp16")]; + tensor var_20747_to_fp16 = const()[name = string("op_20747_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202752)))]; + tensor mh_w_405_cast_fp16 = add(x = mh_w_403_cast_fp16, y = var_20747_to_fp16)[name = string("mh_w_405_cast_fp16")]; + tensor mh_w_407_cast_fp16 = softmax(axis = var_20567, x = mh_w_405_cast_fp16)[name = string("mh_w_407_cast_fp16")]; + tensor transpose_586_perm_0 = const()[name = string("transpose_586_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_135_transpose_x_1 = const()[name = string("attn_135_transpose_x_1"), val = bool(false)]; + bool attn_135_transpose_y_1 = const()[name = string("attn_135_transpose_y_1"), val = bool(true)]; + tensor transpose_586_cast_fp16 = transpose(perm = transpose_586_perm_0, x = reshape_271_cast_fp16)[name = string("transpose_72")]; + tensor attn_135_cast_fp16 = matmul(transpose_x = attn_135_transpose_x_1, transpose_y = attn_135_transpose_y_1, x = transpose_586_cast_fp16, y = mh_w_407_cast_fp16)[name = string("attn_135_cast_fp16")]; + tensor var_20753 = const()[name = string("op_20753"), val = tensor([1, 2048, 1, 1])]; + tensor input_585_cast_fp16 = reshape(shape = var_20753, x = attn_135_cast_fp16)[name = string("input_585_cast_fp16")]; + string obj_599_pad_type_0 = const()[name = string("obj_599_pad_type_0"), val = string("valid")]; + tensor obj_599_strides_0 = const()[name = string("obj_599_strides_0"), val = tensor([1, 1])]; + tensor obj_599_pad_0 = const()[name = string("obj_599_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_599_dilations_0 = const()[name = string("obj_599_dilations_0"), val = tensor([1, 1])]; + int32 obj_599_groups_0 = const()[name = string("obj_599_groups_0"), val = int32(1)]; + tensor obj_599_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_599_dilations_0, groups = obj_599_groups_0, pad = obj_599_pad_0, pad_type = obj_599_pad_type_0, strides = obj_599_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_585_cast_fp16)[name = string("obj_599_cast_fp16")]; + tensor inputs_567_cast_fp16 = add(x = inputs_561_cast_fp16, y = obj_599_cast_fp16)[name = string("inputs_567_cast_fp16")]; + tensor inputs_sq_567_cast_fp16 = mul(x = inputs_567_cast_fp16, y = inputs_567_cast_fp16)[name = string("inputs_sq_567_cast_fp16")]; + tensor variance_567_axes_0 = const()[name = string("variance_567_axes_0"), val = tensor([1])]; + bool variance_567_keep_dims_0 = const()[name = string("variance_567_keep_dims_0"), val = bool(true)]; + tensor variance_567_cast_fp16 = reduce_mean(axes = variance_567_axes_0, keep_dims = variance_567_keep_dims_0, x = inputs_sq_567_cast_fp16)[name = string("variance_567_cast_fp16")]; + fp16 var_20771_to_fp16 = const()[name = string("op_20771_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20772_cast_fp16 = add(x = variance_567_cast_fp16, y = var_20771_to_fp16)[name = string("op_20772_cast_fp16")]; + fp32 var_20773_epsilon_0 = const()[name = string("op_20773_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20773_cast_fp16 = rsqrt(epsilon = var_20773_epsilon_0, x = var_20772_cast_fp16)[name = string("op_20773_cast_fp16")]; + tensor hidden_states_701_cast_fp16 = mul(x = inputs_567_cast_fp16, y = var_20773_cast_fp16)[name = string("hidden_states_701_cast_fp16")]; + tensor input_587_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_701_cast_fp16)[name = string("input_587_cast_fp16")]; + string input_589_pad_type_0 = const()[name = string("input_589_pad_type_0"), val = string("valid")]; + tensor input_589_strides_0 = const()[name = string("input_589_strides_0"), val = tensor([1, 1])]; + tensor input_589_pad_0 = const()[name = string("input_589_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_589_dilations_0 = const()[name = string("input_589_dilations_0"), val = tensor([1, 1])]; + int32 input_589_groups_0 = const()[name = string("input_589_groups_0"), val = int32(1)]; + tensor input_589_cast_fp16 = conv(dilations = input_589_dilations_0, groups = input_589_groups_0, pad = input_589_pad_0, pad_type = input_589_pad_type_0, strides = input_589_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_587_cast_fp16)[name = string("input_589_cast_fp16")]; + tensor var_20787_cast_fp16 = silu(x = input_589_cast_fp16)[name = string("op_20787_cast_fp16")]; + string var_20793_pad_type_0 = const()[name = string("op_20793_pad_type_0"), val = string("valid")]; + tensor var_20793_strides_0 = const()[name = string("op_20793_strides_0"), val = tensor([1, 1])]; + tensor var_20793_pad_0 = const()[name = string("op_20793_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_20793_dilations_0 = const()[name = string("op_20793_dilations_0"), val = tensor([1, 1])]; + int32 var_20793_groups_0 = const()[name = string("op_20793_groups_0"), val = int32(1)]; + tensor var_20793_cast_fp16 = conv(dilations = var_20793_dilations_0, groups = var_20793_groups_0, pad = var_20793_pad_0, pad_type = var_20793_pad_type_0, strides = var_20793_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_587_cast_fp16)[name = string("op_20793_cast_fp16")]; + tensor input_591_cast_fp16 = mul(x = var_20787_cast_fp16, y = var_20793_cast_fp16)[name = string("input_591_cast_fp16")]; + string hidden_states_703_pad_type_0 = const()[name = string("hidden_states_703_pad_type_0"), val = string("valid")]; + tensor hidden_states_703_strides_0 = const()[name = string("hidden_states_703_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_703_pad_0 = const()[name = string("hidden_states_703_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_703_dilations_0 = const()[name = string("hidden_states_703_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_703_groups_0 = const()[name = string("hidden_states_703_groups_0"), val = int32(1)]; + tensor hidden_states_703_cast_fp16 = conv(dilations = hidden_states_703_dilations_0, groups = hidden_states_703_groups_0, pad = hidden_states_703_pad_0, pad_type = hidden_states_703_pad_type_0, strides = hidden_states_703_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_591_cast_fp16)[name = string("hidden_states_703_cast_fp16")]; + tensor inputs_569_cast_fp16 = add(x = inputs_567_cast_fp16, y = hidden_states_703_cast_fp16)[name = string("inputs_569_cast_fp16")]; + tensor obj_603_begin_0 = const()[name = string("obj_603_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_603_end_0 = const()[name = string("obj_603_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_603_end_mask_0 = const()[name = string("obj_603_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_603_cast_fp16 = slice_by_index(begin = obj_603_begin_0, end = obj_603_end_0, end_mask = obj_603_end_mask_0, x = key_caches_27_cast_fp16)[name = string("obj_603_cast_fp16")]; + tensor obj_605_begin_0 = const()[name = string("obj_605_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_605_end_0 = const()[name = string("obj_605_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_605_end_mask_0 = const()[name = string("obj_605_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_605_cast_fp16 = slice_by_index(begin = obj_605_begin_0, end = obj_605_end_0, end_mask = obj_605_end_mask_0, x = value_caches_27_cast_fp16)[name = string("obj_605_cast_fp16")]; + int32 var_20841 = const()[name = string("op_20841"), val = int32(3)]; + int32 var_20851 = const()[name = string("op_20851"), val = int32(-2)]; + tensor inputs_sq_569_cast_fp16 = mul(x = inputs_569_cast_fp16, y = inputs_569_cast_fp16)[name = string("inputs_sq_569_cast_fp16")]; + tensor variance_569_axes_0 = const()[name = string("variance_569_axes_0"), val = tensor([1])]; + bool variance_569_keep_dims_0 = const()[name = string("variance_569_keep_dims_0"), val = bool(true)]; + tensor variance_569_cast_fp16 = reduce_mean(axes = variance_569_axes_0, keep_dims = variance_569_keep_dims_0, x = inputs_sq_569_cast_fp16)[name = string("variance_569_cast_fp16")]; + fp16 var_20865_to_fp16 = const()[name = string("op_20865_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20866_cast_fp16 = add(x = variance_569_cast_fp16, y = var_20865_to_fp16)[name = string("op_20866_cast_fp16")]; + fp32 var_20867_epsilon_0 = const()[name = string("op_20867_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20867_cast_fp16 = rsqrt(epsilon = var_20867_epsilon_0, x = var_20866_cast_fp16)[name = string("op_20867_cast_fp16")]; + tensor hidden_states_705_cast_fp16 = mul(x = inputs_569_cast_fp16, y = var_20867_cast_fp16)[name = string("hidden_states_705_cast_fp16")]; + tensor obj_601_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_705_cast_fp16)[name = string("obj_601_cast_fp16")]; + string query_409_pad_type_0 = const()[name = string("query_409_pad_type_0"), val = string("valid")]; + tensor query_409_strides_0 = const()[name = string("query_409_strides_0"), val = tensor([1, 1])]; + tensor query_409_pad_0 = const()[name = string("query_409_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_409_dilations_0 = const()[name = string("query_409_dilations_0"), val = tensor([1, 1])]; + int32 query_409_groups_0 = const()[name = string("query_409_groups_0"), val = int32(1)]; + tensor query_409_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_409_dilations_0, groups = query_409_groups_0, pad = query_409_pad_0, pad_type = query_409_pad_type_0, strides = query_409_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_601_cast_fp16)[name = string("query_409_cast_fp16")]; + string current_key_273_pad_type_0 = const()[name = string("current_key_273_pad_type_0"), val = string("valid")]; + tensor current_key_273_strides_0 = const()[name = string("current_key_273_strides_0"), val = tensor([1, 1])]; + tensor current_key_273_pad_0 = const()[name = string("current_key_273_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_273_dilations_0 = const()[name = string("current_key_273_dilations_0"), val = tensor([1, 1])]; + int32 current_key_273_groups_0 = const()[name = string("current_key_273_groups_0"), val = int32(1)]; + tensor current_key_273_cast_fp16 = conv(dilations = current_key_273_dilations_0, groups = current_key_273_groups_0, pad = current_key_273_pad_0, pad_type = current_key_273_pad_type_0, strides = current_key_273_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_601_cast_fp16)[name = string("current_key_273_cast_fp16")]; + string current_value_137_pad_type_0 = const()[name = string("current_value_137_pad_type_0"), val = string("valid")]; + tensor current_value_137_strides_0 = const()[name = string("current_value_137_strides_0"), val = tensor([1, 1])]; + tensor current_value_137_pad_0 = const()[name = string("current_value_137_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_137_dilations_0 = const()[name = string("current_value_137_dilations_0"), val = tensor([1, 1])]; + int32 current_value_137_groups_0 = const()[name = string("current_value_137_groups_0"), val = int32(1)]; + tensor current_value_137_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_137_dilations_0, groups = current_value_137_groups_0, pad = current_value_137_pad_0, pad_type = current_value_137_pad_type_0, strides = current_value_137_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_601_cast_fp16)[name = string("current_value_137_cast_fp16")]; + tensor var_20904 = const()[name = string("op_20904"), val = tensor([16, 128, 1, 1])]; + tensor inputs_571_cast_fp16 = reshape(shape = var_20904, x = query_409_cast_fp16)[name = string("inputs_571_cast_fp16")]; + tensor inputs_sq_571_cast_fp16 = mul(x = inputs_571_cast_fp16, y = inputs_571_cast_fp16)[name = string("inputs_sq_571_cast_fp16")]; + tensor variance_571_axes_0 = const()[name = string("variance_571_axes_0"), val = tensor([1])]; + bool variance_571_keep_dims_0 = const()[name = string("variance_571_keep_dims_0"), val = bool(true)]; + tensor variance_571_cast_fp16 = reduce_mean(axes = variance_571_axes_0, keep_dims = variance_571_keep_dims_0, x = inputs_sq_571_cast_fp16)[name = string("variance_571_cast_fp16")]; + fp16 var_20910_to_fp16 = const()[name = string("op_20910_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20911_cast_fp16 = add(x = variance_571_cast_fp16, y = var_20910_to_fp16)[name = string("op_20911_cast_fp16")]; + fp32 var_20912_epsilon_0 = const()[name = string("op_20912_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20912_cast_fp16 = rsqrt(epsilon = var_20912_epsilon_0, x = var_20911_cast_fp16)[name = string("op_20912_cast_fp16")]; + tensor hidden_states_707_cast_fp16 = mul(x = inputs_571_cast_fp16, y = var_20912_cast_fp16)[name = string("hidden_states_707_cast_fp16")]; + tensor query_normed_137_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_707_cast_fp16)[name = string("query_normed_137_cast_fp16")]; + tensor var_20920 = const()[name = string("op_20920"), val = tensor([8, 128, 1, 1])]; + tensor inputs_573_cast_fp16 = reshape(shape = var_20920, x = current_key_273_cast_fp16)[name = string("inputs_573_cast_fp16")]; + tensor inputs_sq_573_cast_fp16 = mul(x = inputs_573_cast_fp16, y = inputs_573_cast_fp16)[name = string("inputs_sq_573_cast_fp16")]; + tensor variance_573_axes_0 = const()[name = string("variance_573_axes_0"), val = tensor([1])]; + bool variance_573_keep_dims_0 = const()[name = string("variance_573_keep_dims_0"), val = bool(true)]; + tensor variance_573_cast_fp16 = reduce_mean(axes = variance_573_axes_0, keep_dims = variance_573_keep_dims_0, x = inputs_sq_573_cast_fp16)[name = string("variance_573_cast_fp16")]; + fp16 var_20926_to_fp16 = const()[name = string("op_20926_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_20927_cast_fp16 = add(x = variance_573_cast_fp16, y = var_20926_to_fp16)[name = string("op_20927_cast_fp16")]; + fp32 var_20928_epsilon_0 = const()[name = string("op_20928_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_20928_cast_fp16 = rsqrt(epsilon = var_20928_epsilon_0, x = var_20927_cast_fp16)[name = string("op_20928_cast_fp16")]; + tensor hidden_states_709_cast_fp16 = mul(x = inputs_573_cast_fp16, y = var_20928_cast_fp16)[name = string("hidden_states_709_cast_fp16")]; + tensor current_key_normed_137_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_709_cast_fp16)[name = string("current_key_normed_137_cast_fp16")]; + tensor var_20946 = const()[name = string("op_20946"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_545_cast_fp16 = reshape(shape = var_20946, x = query_normed_137_cast_fp16)[name = string("mh_q_545_cast_fp16")]; + tensor var_20948 = const()[name = string("op_20948"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_545_cast_fp16 = reshape(shape = var_20948, x = current_key_normed_137_cast_fp16)[name = string("mh_k_545_cast_fp16")]; + tensor var_20952_cast_fp16 = mul(x = mh_q_545_cast_fp16, y = cos_131_to_fp16)[name = string("op_20952_cast_fp16")]; + tensor var_20957_begin_0 = const()[name = string("op_20957_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_20957_end_0 = const()[name = string("op_20957_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_20957_end_mask_0 = const()[name = string("op_20957_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_20957_cast_fp16 = slice_by_index(begin = var_20957_begin_0, end = var_20957_end_0, end_mask = var_20957_end_mask_0, x = mh_q_545_cast_fp16)[name = string("op_20957_cast_fp16")]; + tensor var_20963_begin_0 = const()[name = string("op_20963_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_20963_end_0 = const()[name = string("op_20963_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_20963_end_mask_0 = const()[name = string("op_20963_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_20963_cast_fp16 = slice_by_index(begin = var_20963_begin_0, end = var_20963_end_0, end_mask = var_20963_end_mask_0, x = mh_q_545_cast_fp16)[name = string("op_20963_cast_fp16")]; + fp16 const_1387_promoted_to_fp16 = const()[name = string("const_1387_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_20965_cast_fp16 = mul(x = var_20963_cast_fp16, y = const_1387_promoted_to_fp16)[name = string("op_20965_cast_fp16")]; + bool var_20967_interleave_0 = const()[name = string("op_20967_interleave_0"), val = bool(false)]; + tensor var_20967_cast_fp16 = concat(axis = var_20851, interleave = var_20967_interleave_0, values = (var_20965_cast_fp16, var_20957_cast_fp16))[name = string("op_20967_cast_fp16")]; + tensor var_20968_cast_fp16 = mul(x = var_20967_cast_fp16, y = sin_131_to_fp16)[name = string("op_20968_cast_fp16")]; + tensor mh_q_547_cast_fp16 = add(x = var_20952_cast_fp16, y = var_20968_cast_fp16)[name = string("mh_q_547_cast_fp16")]; + tensor var_20970_cast_fp16 = mul(x = mh_k_545_cast_fp16, y = cos_131_to_fp16)[name = string("op_20970_cast_fp16")]; + tensor var_20975_begin_0 = const()[name = string("op_20975_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_20975_end_0 = const()[name = string("op_20975_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_20975_end_mask_0 = const()[name = string("op_20975_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_20975_cast_fp16 = slice_by_index(begin = var_20975_begin_0, end = var_20975_end_0, end_mask = var_20975_end_mask_0, x = mh_k_545_cast_fp16)[name = string("op_20975_cast_fp16")]; + tensor var_20981_begin_0 = const()[name = string("op_20981_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_20981_end_0 = const()[name = string("op_20981_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_20981_end_mask_0 = const()[name = string("op_20981_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_20981_cast_fp16 = slice_by_index(begin = var_20981_begin_0, end = var_20981_end_0, end_mask = var_20981_end_mask_0, x = mh_k_545_cast_fp16)[name = string("op_20981_cast_fp16")]; + fp16 const_1390_promoted_to_fp16 = const()[name = string("const_1390_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_20983_cast_fp16 = mul(x = var_20981_cast_fp16, y = const_1390_promoted_to_fp16)[name = string("op_20983_cast_fp16")]; + bool var_20985_interleave_0 = const()[name = string("op_20985_interleave_0"), val = bool(false)]; + tensor var_20985_cast_fp16 = concat(axis = var_20851, interleave = var_20985_interleave_0, values = (var_20983_cast_fp16, var_20975_cast_fp16))[name = string("op_20985_cast_fp16")]; + tensor var_20986_cast_fp16 = mul(x = var_20985_cast_fp16, y = sin_131_to_fp16)[name = string("op_20986_cast_fp16")]; + tensor mh_k_547_cast_fp16 = add(x = var_20970_cast_fp16, y = var_20986_cast_fp16)[name = string("mh_k_547_cast_fp16")]; + tensor var_20990 = const()[name = string("op_20990"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_275_cast_fp16 = reshape(shape = var_20990, x = mh_k_547_cast_fp16)[name = string("current_key_275_cast_fp16")]; + tensor var_20996_to_fp16 = const()[name = string("op_20996_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202496)))]; + tensor var_20997_cast_fp16 = mul(x = obj_603_cast_fp16, y = var_20996_to_fp16)[name = string("op_20997_cast_fp16")]; + tensor var_20994_to_fp16 = const()[name = string("op_20994_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202624)))]; + tensor var_20998_cast_fp16 = mul(x = current_key_275_cast_fp16, y = var_20994_to_fp16)[name = string("op_20998_cast_fp16")]; + tensor key_275_cast_fp16 = add(x = var_20997_cast_fp16, y = var_20998_cast_fp16)[name = string("key_275_cast_fp16")]; + tensor var_21000_to_fp16 = const()[name = string("op_21000_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202496)))]; + tensor var_21001_cast_fp16 = mul(x = obj_605_cast_fp16, y = var_21000_to_fp16)[name = string("op_21001_cast_fp16")]; + tensor var_21002_cast_fp16 = mul(x = current_value_137_cast_fp16, y = var_20994_to_fp16)[name = string("op_21002_cast_fp16")]; + tensor value_137_cast_fp16 = add(x = var_21001_cast_fp16, y = var_21002_cast_fp16)[name = string("value_137_cast_fp16")]; + fp16 var_21009_to_fp16 = const()[name = string("op_21009_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_551_cast_fp16 = mul(x = mh_q_547_cast_fp16, y = var_21009_to_fp16)[name = string("mh_q_551_cast_fp16")]; + tensor var_21011 = const()[name = string("op_21011"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_549_cast_fp16 = reshape(shape = var_21011, x = key_275_cast_fp16)[name = string("mh_k_549_cast_fp16")]; + tensor var_21013 = const()[name = string("op_21013"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_273_cast_fp16 = reshape(shape = var_21013, x = value_137_cast_fp16)[name = string("mh_v_273_cast_fp16")]; + tensor transpose_272_perm_0 = const()[name = string("transpose_272_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_136_reps_0 = const()[name = string("tile_136_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_272_cast_fp16 = transpose(perm = transpose_272_perm_0, x = mh_k_549_cast_fp16)[name = string("transpose_71")]; + tensor tile_136_cast_fp16 = tile(reps = tile_136_reps_0, x = transpose_272_cast_fp16)[name = string("tile_136_cast_fp16")]; + tensor concat_340 = const()[name = string("concat_340"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_272_cast_fp16 = reshape(shape = concat_340, x = tile_136_cast_fp16)[name = string("reshape_272_cast_fp16")]; + tensor transpose_273_perm_0 = const()[name = string("transpose_273_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_341 = const()[name = string("concat_341"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_273_cast_fp16 = transpose(perm = transpose_273_perm_0, x = reshape_272_cast_fp16)[name = string("transpose_70")]; + tensor reshape_273_cast_fp16 = reshape(shape = concat_341, x = transpose_273_cast_fp16)[name = string("reshape_273_cast_fp16")]; + tensor transpose_274_perm_0 = const()[name = string("transpose_274_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_137_reps_0 = const()[name = string("tile_137_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_274_cast_fp16 = transpose(perm = transpose_274_perm_0, x = mh_v_273_cast_fp16)[name = string("transpose_69")]; + tensor tile_137_cast_fp16 = tile(reps = tile_137_reps_0, x = transpose_274_cast_fp16)[name = string("tile_137_cast_fp16")]; + tensor concat_342 = const()[name = string("concat_342"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_274_cast_fp16 = reshape(shape = concat_342, x = tile_137_cast_fp16)[name = string("reshape_274_cast_fp16")]; + tensor transpose_275_perm_0 = const()[name = string("transpose_275_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_343 = const()[name = string("concat_343"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_275_cast_fp16 = transpose(perm = transpose_275_perm_0, x = reshape_274_cast_fp16)[name = string("transpose_68")]; + tensor reshape_275_cast_fp16 = reshape(shape = concat_343, x = transpose_275_cast_fp16)[name = string("reshape_275_cast_fp16")]; + tensor transpose_589_perm_0 = const()[name = string("transpose_589_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_409_transpose_x_1 = const()[name = string("mh_w_409_transpose_x_1"), val = bool(true)]; + bool mh_w_409_transpose_y_1 = const()[name = string("mh_w_409_transpose_y_1"), val = bool(false)]; + tensor transpose_589_cast_fp16 = transpose(perm = transpose_589_perm_0, x = reshape_273_cast_fp16)[name = string("transpose_67")]; + tensor mh_w_409_cast_fp16 = matmul(transpose_x = mh_w_409_transpose_x_1, transpose_y = mh_w_409_transpose_y_1, x = mh_q_551_cast_fp16, y = transpose_589_cast_fp16)[name = string("mh_w_409_cast_fp16")]; + tensor var_21021_to_fp16 = const()[name = string("op_21021_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202752)))]; + tensor mh_w_411_cast_fp16 = add(x = mh_w_409_cast_fp16, y = var_21021_to_fp16)[name = string("mh_w_411_cast_fp16")]; + tensor mh_w_413_cast_fp16 = softmax(axis = var_20841, x = mh_w_411_cast_fp16)[name = string("mh_w_413_cast_fp16")]; + tensor transpose_590_perm_0 = const()[name = string("transpose_590_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_137_transpose_x_1 = const()[name = string("attn_137_transpose_x_1"), val = bool(false)]; + bool attn_137_transpose_y_1 = const()[name = string("attn_137_transpose_y_1"), val = bool(true)]; + tensor transpose_590_cast_fp16 = transpose(perm = transpose_590_perm_0, x = reshape_275_cast_fp16)[name = string("transpose_66")]; + tensor attn_137_cast_fp16 = matmul(transpose_x = attn_137_transpose_x_1, transpose_y = attn_137_transpose_y_1, x = transpose_590_cast_fp16, y = mh_w_413_cast_fp16)[name = string("attn_137_cast_fp16")]; + tensor var_21027 = const()[name = string("op_21027"), val = tensor([1, 2048, 1, 1])]; + tensor input_593_cast_fp16 = reshape(shape = var_21027, x = attn_137_cast_fp16)[name = string("input_593_cast_fp16")]; + string obj_607_pad_type_0 = const()[name = string("obj_607_pad_type_0"), val = string("valid")]; + tensor obj_607_strides_0 = const()[name = string("obj_607_strides_0"), val = tensor([1, 1])]; + tensor obj_607_pad_0 = const()[name = string("obj_607_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_607_dilations_0 = const()[name = string("obj_607_dilations_0"), val = tensor([1, 1])]; + int32 obj_607_groups_0 = const()[name = string("obj_607_groups_0"), val = int32(1)]; + tensor obj_607_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_607_dilations_0, groups = obj_607_groups_0, pad = obj_607_pad_0, pad_type = obj_607_pad_type_0, strides = obj_607_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_593_cast_fp16)[name = string("obj_607_cast_fp16")]; + tensor inputs_575_cast_fp16 = add(x = inputs_569_cast_fp16, y = obj_607_cast_fp16)[name = string("inputs_575_cast_fp16")]; + tensor inputs_sq_575_cast_fp16 = mul(x = inputs_575_cast_fp16, y = inputs_575_cast_fp16)[name = string("inputs_sq_575_cast_fp16")]; + tensor variance_575_axes_0 = const()[name = string("variance_575_axes_0"), val = tensor([1])]; + bool variance_575_keep_dims_0 = const()[name = string("variance_575_keep_dims_0"), val = bool(true)]; + tensor variance_575_cast_fp16 = reduce_mean(axes = variance_575_axes_0, keep_dims = variance_575_keep_dims_0, x = inputs_sq_575_cast_fp16)[name = string("variance_575_cast_fp16")]; + fp16 var_21045_to_fp16 = const()[name = string("op_21045_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21046_cast_fp16 = add(x = variance_575_cast_fp16, y = var_21045_to_fp16)[name = string("op_21046_cast_fp16")]; + fp32 var_21047_epsilon_0 = const()[name = string("op_21047_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21047_cast_fp16 = rsqrt(epsilon = var_21047_epsilon_0, x = var_21046_cast_fp16)[name = string("op_21047_cast_fp16")]; + tensor hidden_states_711_cast_fp16 = mul(x = inputs_575_cast_fp16, y = var_21047_cast_fp16)[name = string("hidden_states_711_cast_fp16")]; + tensor input_595_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_711_cast_fp16)[name = string("input_595_cast_fp16")]; + string input_597_pad_type_0 = const()[name = string("input_597_pad_type_0"), val = string("valid")]; + tensor input_597_strides_0 = const()[name = string("input_597_strides_0"), val = tensor([1, 1])]; + tensor input_597_pad_0 = const()[name = string("input_597_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_597_dilations_0 = const()[name = string("input_597_dilations_0"), val = tensor([1, 1])]; + int32 input_597_groups_0 = const()[name = string("input_597_groups_0"), val = int32(1)]; + tensor input_597_cast_fp16 = conv(dilations = input_597_dilations_0, groups = input_597_groups_0, pad = input_597_pad_0, pad_type = input_597_pad_type_0, strides = input_597_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_595_cast_fp16)[name = string("input_597_cast_fp16")]; + tensor var_21061_cast_fp16 = silu(x = input_597_cast_fp16)[name = string("op_21061_cast_fp16")]; + string var_21067_pad_type_0 = const()[name = string("op_21067_pad_type_0"), val = string("valid")]; + tensor var_21067_strides_0 = const()[name = string("op_21067_strides_0"), val = tensor([1, 1])]; + tensor var_21067_pad_0 = const()[name = string("op_21067_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_21067_dilations_0 = const()[name = string("op_21067_dilations_0"), val = tensor([1, 1])]; + int32 var_21067_groups_0 = const()[name = string("op_21067_groups_0"), val = int32(1)]; + tensor var_21067_cast_fp16 = conv(dilations = var_21067_dilations_0, groups = var_21067_groups_0, pad = var_21067_pad_0, pad_type = var_21067_pad_type_0, strides = var_21067_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_595_cast_fp16)[name = string("op_21067_cast_fp16")]; + tensor input_599_cast_fp16 = mul(x = var_21061_cast_fp16, y = var_21067_cast_fp16)[name = string("input_599_cast_fp16")]; + string hidden_states_713_pad_type_0 = const()[name = string("hidden_states_713_pad_type_0"), val = string("valid")]; + tensor hidden_states_713_strides_0 = const()[name = string("hidden_states_713_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_713_pad_0 = const()[name = string("hidden_states_713_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_713_dilations_0 = const()[name = string("hidden_states_713_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_713_groups_0 = const()[name = string("hidden_states_713_groups_0"), val = int32(1)]; + tensor hidden_states_713_cast_fp16 = conv(dilations = hidden_states_713_dilations_0, groups = hidden_states_713_groups_0, pad = hidden_states_713_pad_0, pad_type = hidden_states_713_pad_type_0, strides = hidden_states_713_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_599_cast_fp16)[name = string("hidden_states_713_cast_fp16")]; + tensor inputs_577_cast_fp16 = add(x = inputs_575_cast_fp16, y = hidden_states_713_cast_fp16)[name = string("inputs_577_cast_fp16")]; + tensor obj_611_begin_0 = const()[name = string("obj_611_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_611_end_0 = const()[name = string("obj_611_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_611_end_mask_0 = const()[name = string("obj_611_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_611_cast_fp16 = slice_by_index(begin = obj_611_begin_0, end = obj_611_end_0, end_mask = obj_611_end_mask_0, x = key_caches_27_cast_fp16)[name = string("obj_611_cast_fp16")]; + tensor obj_613_begin_0 = const()[name = string("obj_613_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_613_end_0 = const()[name = string("obj_613_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_613_end_mask_0 = const()[name = string("obj_613_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_613_cast_fp16 = slice_by_index(begin = obj_613_begin_0, end = obj_613_end_0, end_mask = obj_613_end_mask_0, x = value_caches_27_cast_fp16)[name = string("obj_613_cast_fp16")]; + int32 var_21115 = const()[name = string("op_21115"), val = int32(3)]; + int32 var_21125 = const()[name = string("op_21125"), val = int32(-2)]; + tensor inputs_sq_577_cast_fp16 = mul(x = inputs_577_cast_fp16, y = inputs_577_cast_fp16)[name = string("inputs_sq_577_cast_fp16")]; + tensor variance_577_axes_0 = const()[name = string("variance_577_axes_0"), val = tensor([1])]; + bool variance_577_keep_dims_0 = const()[name = string("variance_577_keep_dims_0"), val = bool(true)]; + tensor variance_577_cast_fp16 = reduce_mean(axes = variance_577_axes_0, keep_dims = variance_577_keep_dims_0, x = inputs_sq_577_cast_fp16)[name = string("variance_577_cast_fp16")]; + fp16 var_21139_to_fp16 = const()[name = string("op_21139_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21140_cast_fp16 = add(x = variance_577_cast_fp16, y = var_21139_to_fp16)[name = string("op_21140_cast_fp16")]; + fp32 var_21141_epsilon_0 = const()[name = string("op_21141_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21141_cast_fp16 = rsqrt(epsilon = var_21141_epsilon_0, x = var_21140_cast_fp16)[name = string("op_21141_cast_fp16")]; + tensor hidden_states_715_cast_fp16 = mul(x = inputs_577_cast_fp16, y = var_21141_cast_fp16)[name = string("hidden_states_715_cast_fp16")]; + tensor obj_609_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_715_cast_fp16)[name = string("obj_609_cast_fp16")]; + string query_415_pad_type_0 = const()[name = string("query_415_pad_type_0"), val = string("valid")]; + tensor query_415_strides_0 = const()[name = string("query_415_strides_0"), val = tensor([1, 1])]; + tensor query_415_pad_0 = const()[name = string("query_415_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_415_dilations_0 = const()[name = string("query_415_dilations_0"), val = tensor([1, 1])]; + int32 query_415_groups_0 = const()[name = string("query_415_groups_0"), val = int32(1)]; + tensor query_415_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_415_dilations_0, groups = query_415_groups_0, pad = query_415_pad_0, pad_type = query_415_pad_type_0, strides = query_415_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_609_cast_fp16)[name = string("query_415_cast_fp16")]; + string current_key_277_pad_type_0 = const()[name = string("current_key_277_pad_type_0"), val = string("valid")]; + tensor current_key_277_strides_0 = const()[name = string("current_key_277_strides_0"), val = tensor([1, 1])]; + tensor current_key_277_pad_0 = const()[name = string("current_key_277_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_277_dilations_0 = const()[name = string("current_key_277_dilations_0"), val = tensor([1, 1])]; + int32 current_key_277_groups_0 = const()[name = string("current_key_277_groups_0"), val = int32(1)]; + tensor current_key_277_cast_fp16 = conv(dilations = current_key_277_dilations_0, groups = current_key_277_groups_0, pad = current_key_277_pad_0, pad_type = current_key_277_pad_type_0, strides = current_key_277_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_609_cast_fp16)[name = string("current_key_277_cast_fp16")]; + string current_value_139_pad_type_0 = const()[name = string("current_value_139_pad_type_0"), val = string("valid")]; + tensor current_value_139_strides_0 = const()[name = string("current_value_139_strides_0"), val = tensor([1, 1])]; + tensor current_value_139_pad_0 = const()[name = string("current_value_139_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_139_dilations_0 = const()[name = string("current_value_139_dilations_0"), val = tensor([1, 1])]; + int32 current_value_139_groups_0 = const()[name = string("current_value_139_groups_0"), val = int32(1)]; + tensor current_value_139_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_139_dilations_0, groups = current_value_139_groups_0, pad = current_value_139_pad_0, pad_type = current_value_139_pad_type_0, strides = current_value_139_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_609_cast_fp16)[name = string("current_value_139_cast_fp16")]; + tensor var_21178 = const()[name = string("op_21178"), val = tensor([16, 128, 1, 1])]; + tensor inputs_579_cast_fp16 = reshape(shape = var_21178, x = query_415_cast_fp16)[name = string("inputs_579_cast_fp16")]; + tensor inputs_sq_579_cast_fp16 = mul(x = inputs_579_cast_fp16, y = inputs_579_cast_fp16)[name = string("inputs_sq_579_cast_fp16")]; + tensor variance_579_axes_0 = const()[name = string("variance_579_axes_0"), val = tensor([1])]; + bool variance_579_keep_dims_0 = const()[name = string("variance_579_keep_dims_0"), val = bool(true)]; + tensor variance_579_cast_fp16 = reduce_mean(axes = variance_579_axes_0, keep_dims = variance_579_keep_dims_0, x = inputs_sq_579_cast_fp16)[name = string("variance_579_cast_fp16")]; + fp16 var_21184_to_fp16 = const()[name = string("op_21184_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21185_cast_fp16 = add(x = variance_579_cast_fp16, y = var_21184_to_fp16)[name = string("op_21185_cast_fp16")]; + fp32 var_21186_epsilon_0 = const()[name = string("op_21186_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21186_cast_fp16 = rsqrt(epsilon = var_21186_epsilon_0, x = var_21185_cast_fp16)[name = string("op_21186_cast_fp16")]; + tensor hidden_states_717_cast_fp16 = mul(x = inputs_579_cast_fp16, y = var_21186_cast_fp16)[name = string("hidden_states_717_cast_fp16")]; + tensor query_normed_139_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_717_cast_fp16)[name = string("query_normed_139_cast_fp16")]; + tensor var_21194 = const()[name = string("op_21194"), val = tensor([8, 128, 1, 1])]; + tensor inputs_581_cast_fp16 = reshape(shape = var_21194, x = current_key_277_cast_fp16)[name = string("inputs_581_cast_fp16")]; + tensor inputs_sq_581_cast_fp16 = mul(x = inputs_581_cast_fp16, y = inputs_581_cast_fp16)[name = string("inputs_sq_581_cast_fp16")]; + tensor variance_581_axes_0 = const()[name = string("variance_581_axes_0"), val = tensor([1])]; + bool variance_581_keep_dims_0 = const()[name = string("variance_581_keep_dims_0"), val = bool(true)]; + tensor variance_581_cast_fp16 = reduce_mean(axes = variance_581_axes_0, keep_dims = variance_581_keep_dims_0, x = inputs_sq_581_cast_fp16)[name = string("variance_581_cast_fp16")]; + fp16 var_21200_to_fp16 = const()[name = string("op_21200_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21201_cast_fp16 = add(x = variance_581_cast_fp16, y = var_21200_to_fp16)[name = string("op_21201_cast_fp16")]; + fp32 var_21202_epsilon_0 = const()[name = string("op_21202_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21202_cast_fp16 = rsqrt(epsilon = var_21202_epsilon_0, x = var_21201_cast_fp16)[name = string("op_21202_cast_fp16")]; + tensor hidden_states_719_cast_fp16 = mul(x = inputs_581_cast_fp16, y = var_21202_cast_fp16)[name = string("hidden_states_719_cast_fp16")]; + tensor current_key_normed_139_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_719_cast_fp16)[name = string("current_key_normed_139_cast_fp16")]; + tensor var_21220 = const()[name = string("op_21220"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_553_cast_fp16 = reshape(shape = var_21220, x = query_normed_139_cast_fp16)[name = string("mh_q_553_cast_fp16")]; + tensor var_21222 = const()[name = string("op_21222"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_553_cast_fp16 = reshape(shape = var_21222, x = current_key_normed_139_cast_fp16)[name = string("mh_k_553_cast_fp16")]; + tensor var_21226_cast_fp16 = mul(x = mh_q_553_cast_fp16, y = cos_131_to_fp16)[name = string("op_21226_cast_fp16")]; + tensor var_21231_begin_0 = const()[name = string("op_21231_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_21231_end_0 = const()[name = string("op_21231_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_21231_end_mask_0 = const()[name = string("op_21231_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_21231_cast_fp16 = slice_by_index(begin = var_21231_begin_0, end = var_21231_end_0, end_mask = var_21231_end_mask_0, x = mh_q_553_cast_fp16)[name = string("op_21231_cast_fp16")]; + tensor var_21237_begin_0 = const()[name = string("op_21237_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_21237_end_0 = const()[name = string("op_21237_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_21237_end_mask_0 = const()[name = string("op_21237_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_21237_cast_fp16 = slice_by_index(begin = var_21237_begin_0, end = var_21237_end_0, end_mask = var_21237_end_mask_0, x = mh_q_553_cast_fp16)[name = string("op_21237_cast_fp16")]; + fp16 const_1407_promoted_to_fp16 = const()[name = string("const_1407_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_21239_cast_fp16 = mul(x = var_21237_cast_fp16, y = const_1407_promoted_to_fp16)[name = string("op_21239_cast_fp16")]; + bool var_21241_interleave_0 = const()[name = string("op_21241_interleave_0"), val = bool(false)]; + tensor var_21241_cast_fp16 = concat(axis = var_21125, interleave = var_21241_interleave_0, values = (var_21239_cast_fp16, var_21231_cast_fp16))[name = string("op_21241_cast_fp16")]; + tensor var_21242_cast_fp16 = mul(x = var_21241_cast_fp16, y = sin_131_to_fp16)[name = string("op_21242_cast_fp16")]; + tensor mh_q_555_cast_fp16 = add(x = var_21226_cast_fp16, y = var_21242_cast_fp16)[name = string("mh_q_555_cast_fp16")]; + tensor var_21244_cast_fp16 = mul(x = mh_k_553_cast_fp16, y = cos_131_to_fp16)[name = string("op_21244_cast_fp16")]; + tensor var_21249_begin_0 = const()[name = string("op_21249_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_21249_end_0 = const()[name = string("op_21249_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_21249_end_mask_0 = const()[name = string("op_21249_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_21249_cast_fp16 = slice_by_index(begin = var_21249_begin_0, end = var_21249_end_0, end_mask = var_21249_end_mask_0, x = mh_k_553_cast_fp16)[name = string("op_21249_cast_fp16")]; + tensor var_21255_begin_0 = const()[name = string("op_21255_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_21255_end_0 = const()[name = string("op_21255_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_21255_end_mask_0 = const()[name = string("op_21255_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_21255_cast_fp16 = slice_by_index(begin = var_21255_begin_0, end = var_21255_end_0, end_mask = var_21255_end_mask_0, x = mh_k_553_cast_fp16)[name = string("op_21255_cast_fp16")]; + fp16 const_1410_promoted_to_fp16 = const()[name = string("const_1410_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_21257_cast_fp16 = mul(x = var_21255_cast_fp16, y = const_1410_promoted_to_fp16)[name = string("op_21257_cast_fp16")]; + bool var_21259_interleave_0 = const()[name = string("op_21259_interleave_0"), val = bool(false)]; + tensor var_21259_cast_fp16 = concat(axis = var_21125, interleave = var_21259_interleave_0, values = (var_21257_cast_fp16, var_21249_cast_fp16))[name = string("op_21259_cast_fp16")]; + tensor var_21260_cast_fp16 = mul(x = var_21259_cast_fp16, y = sin_131_to_fp16)[name = string("op_21260_cast_fp16")]; + tensor mh_k_555_cast_fp16 = add(x = var_21244_cast_fp16, y = var_21260_cast_fp16)[name = string("mh_k_555_cast_fp16")]; + tensor var_21264 = const()[name = string("op_21264"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_279_cast_fp16 = reshape(shape = var_21264, x = mh_k_555_cast_fp16)[name = string("current_key_279_cast_fp16")]; + tensor var_21270_to_fp16 = const()[name = string("op_21270_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202496)))]; + tensor var_21271_cast_fp16 = mul(x = obj_611_cast_fp16, y = var_21270_to_fp16)[name = string("op_21271_cast_fp16")]; + tensor var_21268_to_fp16 = const()[name = string("op_21268_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202624)))]; + tensor var_21272_cast_fp16 = mul(x = current_key_279_cast_fp16, y = var_21268_to_fp16)[name = string("op_21272_cast_fp16")]; + tensor key_279_cast_fp16 = add(x = var_21271_cast_fp16, y = var_21272_cast_fp16)[name = string("key_279_cast_fp16")]; + tensor var_21274_to_fp16 = const()[name = string("op_21274_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202496)))]; + tensor var_21275_cast_fp16 = mul(x = obj_613_cast_fp16, y = var_21274_to_fp16)[name = string("op_21275_cast_fp16")]; + tensor var_21276_cast_fp16 = mul(x = current_value_139_cast_fp16, y = var_21268_to_fp16)[name = string("op_21276_cast_fp16")]; + tensor value_139_cast_fp16 = add(x = var_21275_cast_fp16, y = var_21276_cast_fp16)[name = string("value_139_cast_fp16")]; + fp16 var_21283_to_fp16 = const()[name = string("op_21283_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_559_cast_fp16 = mul(x = mh_q_555_cast_fp16, y = var_21283_to_fp16)[name = string("mh_q_559_cast_fp16")]; + tensor var_21285 = const()[name = string("op_21285"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_557_cast_fp16 = reshape(shape = var_21285, x = key_279_cast_fp16)[name = string("mh_k_557_cast_fp16")]; + tensor var_21287 = const()[name = string("op_21287"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_277_cast_fp16 = reshape(shape = var_21287, x = value_139_cast_fp16)[name = string("mh_v_277_cast_fp16")]; + tensor transpose_276_perm_0 = const()[name = string("transpose_276_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_138_reps_0 = const()[name = string("tile_138_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_276_cast_fp16 = transpose(perm = transpose_276_perm_0, x = mh_k_557_cast_fp16)[name = string("transpose_65")]; + tensor tile_138_cast_fp16 = tile(reps = tile_138_reps_0, x = transpose_276_cast_fp16)[name = string("tile_138_cast_fp16")]; + tensor concat_344 = const()[name = string("concat_344"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_276_cast_fp16 = reshape(shape = concat_344, x = tile_138_cast_fp16)[name = string("reshape_276_cast_fp16")]; + tensor transpose_277_perm_0 = const()[name = string("transpose_277_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_345 = const()[name = string("concat_345"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_277_cast_fp16 = transpose(perm = transpose_277_perm_0, x = reshape_276_cast_fp16)[name = string("transpose_64")]; + tensor reshape_277_cast_fp16 = reshape(shape = concat_345, x = transpose_277_cast_fp16)[name = string("reshape_277_cast_fp16")]; + tensor transpose_278_perm_0 = const()[name = string("transpose_278_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_139_reps_0 = const()[name = string("tile_139_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_278_cast_fp16 = transpose(perm = transpose_278_perm_0, x = mh_v_277_cast_fp16)[name = string("transpose_63")]; + tensor tile_139_cast_fp16 = tile(reps = tile_139_reps_0, x = transpose_278_cast_fp16)[name = string("tile_139_cast_fp16")]; + tensor concat_346 = const()[name = string("concat_346"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_278_cast_fp16 = reshape(shape = concat_346, x = tile_139_cast_fp16)[name = string("reshape_278_cast_fp16")]; + tensor transpose_279_perm_0 = const()[name = string("transpose_279_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_347 = const()[name = string("concat_347"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_279_cast_fp16 = transpose(perm = transpose_279_perm_0, x = reshape_278_cast_fp16)[name = string("transpose_62")]; + tensor reshape_279_cast_fp16 = reshape(shape = concat_347, x = transpose_279_cast_fp16)[name = string("reshape_279_cast_fp16")]; + tensor transpose_593_perm_0 = const()[name = string("transpose_593_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_415_transpose_x_1 = const()[name = string("mh_w_415_transpose_x_1"), val = bool(true)]; + bool mh_w_415_transpose_y_1 = const()[name = string("mh_w_415_transpose_y_1"), val = bool(false)]; + tensor transpose_593_cast_fp16 = transpose(perm = transpose_593_perm_0, x = reshape_277_cast_fp16)[name = string("transpose_61")]; + tensor mh_w_415_cast_fp16 = matmul(transpose_x = mh_w_415_transpose_x_1, transpose_y = mh_w_415_transpose_y_1, x = mh_q_559_cast_fp16, y = transpose_593_cast_fp16)[name = string("mh_w_415_cast_fp16")]; + tensor var_21295_to_fp16 = const()[name = string("op_21295_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202752)))]; + tensor mh_w_417_cast_fp16 = add(x = mh_w_415_cast_fp16, y = var_21295_to_fp16)[name = string("mh_w_417_cast_fp16")]; + tensor mh_w_419_cast_fp16 = softmax(axis = var_21115, x = mh_w_417_cast_fp16)[name = string("mh_w_419_cast_fp16")]; + tensor transpose_594_perm_0 = const()[name = string("transpose_594_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_139_transpose_x_1 = const()[name = string("attn_139_transpose_x_1"), val = bool(false)]; + bool attn_139_transpose_y_1 = const()[name = string("attn_139_transpose_y_1"), val = bool(true)]; + tensor transpose_594_cast_fp16 = transpose(perm = transpose_594_perm_0, x = reshape_279_cast_fp16)[name = string("transpose_60")]; + tensor attn_139_cast_fp16 = matmul(transpose_x = attn_139_transpose_x_1, transpose_y = attn_139_transpose_y_1, x = transpose_594_cast_fp16, y = mh_w_419_cast_fp16)[name = string("attn_139_cast_fp16")]; + tensor var_21301 = const()[name = string("op_21301"), val = tensor([1, 2048, 1, 1])]; + tensor input_601_cast_fp16 = reshape(shape = var_21301, x = attn_139_cast_fp16)[name = string("input_601_cast_fp16")]; + string obj_615_pad_type_0 = const()[name = string("obj_615_pad_type_0"), val = string("valid")]; + tensor obj_615_strides_0 = const()[name = string("obj_615_strides_0"), val = tensor([1, 1])]; + tensor obj_615_pad_0 = const()[name = string("obj_615_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_615_dilations_0 = const()[name = string("obj_615_dilations_0"), val = tensor([1, 1])]; + int32 obj_615_groups_0 = const()[name = string("obj_615_groups_0"), val = int32(1)]; + tensor obj_615_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_615_dilations_0, groups = obj_615_groups_0, pad = obj_615_pad_0, pad_type = obj_615_pad_type_0, strides = obj_615_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_601_cast_fp16)[name = string("obj_615_cast_fp16")]; + tensor inputs_583_cast_fp16 = add(x = inputs_577_cast_fp16, y = obj_615_cast_fp16)[name = string("inputs_583_cast_fp16")]; + tensor inputs_sq_583_cast_fp16 = mul(x = inputs_583_cast_fp16, y = inputs_583_cast_fp16)[name = string("inputs_sq_583_cast_fp16")]; + tensor variance_583_axes_0 = const()[name = string("variance_583_axes_0"), val = tensor([1])]; + bool variance_583_keep_dims_0 = const()[name = string("variance_583_keep_dims_0"), val = bool(true)]; + tensor variance_583_cast_fp16 = reduce_mean(axes = variance_583_axes_0, keep_dims = variance_583_keep_dims_0, x = inputs_sq_583_cast_fp16)[name = string("variance_583_cast_fp16")]; + fp16 var_21319_to_fp16 = const()[name = string("op_21319_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21320_cast_fp16 = add(x = variance_583_cast_fp16, y = var_21319_to_fp16)[name = string("op_21320_cast_fp16")]; + fp32 var_21321_epsilon_0 = const()[name = string("op_21321_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21321_cast_fp16 = rsqrt(epsilon = var_21321_epsilon_0, x = var_21320_cast_fp16)[name = string("op_21321_cast_fp16")]; + tensor hidden_states_721_cast_fp16 = mul(x = inputs_583_cast_fp16, y = var_21321_cast_fp16)[name = string("hidden_states_721_cast_fp16")]; + tensor input_603_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_721_cast_fp16)[name = string("input_603_cast_fp16")]; + string input_605_pad_type_0 = const()[name = string("input_605_pad_type_0"), val = string("valid")]; + tensor input_605_strides_0 = const()[name = string("input_605_strides_0"), val = tensor([1, 1])]; + tensor input_605_pad_0 = const()[name = string("input_605_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_605_dilations_0 = const()[name = string("input_605_dilations_0"), val = tensor([1, 1])]; + int32 input_605_groups_0 = const()[name = string("input_605_groups_0"), val = int32(1)]; + tensor input_605_cast_fp16 = conv(dilations = input_605_dilations_0, groups = input_605_groups_0, pad = input_605_pad_0, pad_type = input_605_pad_type_0, strides = input_605_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_603_cast_fp16)[name = string("input_605_cast_fp16")]; + tensor var_21335_cast_fp16 = silu(x = input_605_cast_fp16)[name = string("op_21335_cast_fp16")]; + string var_21341_pad_type_0 = const()[name = string("op_21341_pad_type_0"), val = string("valid")]; + tensor var_21341_strides_0 = const()[name = string("op_21341_strides_0"), val = tensor([1, 1])]; + tensor var_21341_pad_0 = const()[name = string("op_21341_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_21341_dilations_0 = const()[name = string("op_21341_dilations_0"), val = tensor([1, 1])]; + int32 var_21341_groups_0 = const()[name = string("op_21341_groups_0"), val = int32(1)]; + tensor var_21341_cast_fp16 = conv(dilations = var_21341_dilations_0, groups = var_21341_groups_0, pad = var_21341_pad_0, pad_type = var_21341_pad_type_0, strides = var_21341_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_603_cast_fp16)[name = string("op_21341_cast_fp16")]; + tensor input_607_cast_fp16 = mul(x = var_21335_cast_fp16, y = var_21341_cast_fp16)[name = string("input_607_cast_fp16")]; + string hidden_states_723_pad_type_0 = const()[name = string("hidden_states_723_pad_type_0"), val = string("valid")]; + tensor hidden_states_723_strides_0 = const()[name = string("hidden_states_723_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_723_pad_0 = const()[name = string("hidden_states_723_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_723_dilations_0 = const()[name = string("hidden_states_723_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_723_groups_0 = const()[name = string("hidden_states_723_groups_0"), val = int32(1)]; + tensor hidden_states_723_cast_fp16 = conv(dilations = hidden_states_723_dilations_0, groups = hidden_states_723_groups_0, pad = hidden_states_723_pad_0, pad_type = hidden_states_723_pad_type_0, strides = hidden_states_723_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_607_cast_fp16)[name = string("hidden_states_723_cast_fp16")]; + tensor inputs_585_cast_fp16 = add(x = inputs_583_cast_fp16, y = hidden_states_723_cast_fp16)[name = string("inputs_585_cast_fp16")]; + int32 var_21369 = const()[name = string("op_21369"), val = int32(1)]; + bool key_caches_29_interleave_0 = const()[name = string("key_caches_29_interleave_0"), val = bool(false)]; + tensor key_caches_29_cast_fp16 = concat(axis = var_21369, interleave = key_caches_29_interleave_0, values = (key_263_cast_fp16, key_267_cast_fp16, key_271_cast_fp16, key_275_cast_fp16, key_279_cast_fp16))[name = string("key_caches_29_cast_fp16")]; + int32 var_21372 = const()[name = string("op_21372"), val = int32(1)]; + bool value_caches_29_interleave_0 = const()[name = string("value_caches_29_interleave_0"), val = bool(false)]; + tensor value_caches_29_cast_fp16 = concat(axis = var_21372, interleave = value_caches_29_interleave_0, values = (value_131_cast_fp16, value_133_cast_fp16, value_135_cast_fp16, value_137_cast_fp16, value_139_cast_fp16))[name = string("value_caches_29_cast_fp16")]; + tensor inputs_sq_585_cast_fp16 = mul(x = inputs_585_cast_fp16, y = inputs_585_cast_fp16)[name = string("inputs_sq_585_cast_fp16")]; + tensor variance_585_axes_0 = const()[name = string("variance_585_axes_0"), val = tensor([1])]; + bool variance_585_keep_dims_0 = const()[name = string("variance_585_keep_dims_0"), val = bool(true)]; + tensor variance_585_cast_fp16 = reduce_mean(axes = variance_585_axes_0, keep_dims = variance_585_keep_dims_0, x = inputs_sq_585_cast_fp16)[name = string("variance_585_cast_fp16")]; + fp16 var_21382_to_fp16 = const()[name = string("op_21382_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21383_cast_fp16 = add(x = variance_585_cast_fp16, y = var_21382_to_fp16)[name = string("op_21383_cast_fp16")]; + fp32 var_21384_epsilon_0 = const()[name = string("op_21384_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21384_cast_fp16 = rsqrt(epsilon = var_21384_epsilon_0, x = var_21383_cast_fp16)[name = string("op_21384_cast_fp16")]; + tensor hidden_states_725_cast_fp16 = mul(x = inputs_585_cast_fp16, y = var_21384_cast_fp16)[name = string("hidden_states_725_cast_fp16")]; + tensor input_609_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_725_cast_fp16)[name = string("input_609_cast_fp16")]; + string logits_49_pad_type_0 = const()[name = string("logits_49_pad_type_0"), val = string("valid")]; + tensor logits_49_strides_0 = const()[name = string("logits_49_strides_0"), val = tensor([1, 1])]; + tensor logits_49_pad_0 = const()[name = string("logits_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_49_dilations_0 = const()[name = string("logits_49_dilations_0"), val = tensor([1, 1])]; + int32 logits_49_groups_0 = const()[name = string("logits_49_groups_0"), val = int32(1)]; + tensor lm_heads_12_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105980096))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108077312))))[name = string("lm_heads_12_weight_to_fp16_palettized")]; + tensor logits_49_cast_fp16 = conv(dilations = logits_49_dilations_0, groups = logits_49_groups_0, pad = logits_49_pad_0, pad_type = logits_49_pad_type_0, strides = logits_49_strides_0, weight = lm_heads_12_weight_to_fp16_palettized, x = input_609_cast_fp16)[name = string("logits_49_cast_fp16")]; + tensor var_21402 = const()[name = string("op_21402"), val = tensor([1, 2048])]; + tensor logits_51_cast_fp16 = reshape(shape = var_21402, x = logits_49_cast_fp16)[name = string("logits_51_cast_fp16")]; + tensor scaled_logits_25_cast_fp16 = real_div(x = logits_51_cast_fp16, y = temperature)[name = string("scaled_logits_25_cast_fp16")]; + int32 var_21412 = const()[name = string("op_21412"), val = int32(100)]; + int32 top_values_25_axis_0 = const()[name = string("top_values_25_axis_0"), val = int32(1)]; + bool top_values_25_ascending_0 = const()[name = string("top_values_25_ascending_0"), val = bool(false)]; + bool top_values_25_sort_0 = const()[name = string("top_values_25_sort_0"), val = bool(true)]; + bool top_values_25_return_indices_0 = const()[name = string("top_values_25_return_indices_0"), val = bool(true)]; + string top_values_25_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_25_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_25_cast_fp16_cast_uint16_0, tensor top_values_25_cast_fp16_cast_uint16_1 = topk(ascending = top_values_25_ascending_0, axis = top_values_25_axis_0, k = var_21412, output_indices_dtype = top_values_25_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_25_return_indices_0, sort = top_values_25_sort_0, x = scaled_logits_25_cast_fp16)[name = string("top_values_25_cast_fp16_cast_uint16")]; + tensor var_21418_cast_fp16 = mul(x = top_values_25_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_21418_cast_fp16")]; + tensor var_21422_cast_fp16 = add(x = var_21418_cast_fp16, y = var_3001_cast_fp16)[name = string("op_21422_cast_fp16")]; + tensor reduce_min_12_axes_0 = const()[name = string("reduce_min_12_axes_0"), val = tensor([1])]; + bool reduce_min_12_keep_dims_0 = const()[name = string("reduce_min_12_keep_dims_0"), val = bool(true)]; + tensor reduce_min_12_cast_fp16 = reduce_min(axes = reduce_min_12_axes_0, keep_dims = reduce_min_12_keep_dims_0, x = var_21422_cast_fp16)[name = string("reduce_min_12_cast_fp16")]; + tensor var_21425_cast_fp16 = greater_equal(x = scaled_logits_25_cast_fp16, y = reduce_min_12_cast_fp16)[name = string("op_21425_cast_fp16")]; + fp16 var_21426_value_0_to_fp16 = const()[name = string("op_21426_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_21426_cast_fp16 = fill_like(ref_tensor = scaled_logits_25_cast_fp16, value = var_21426_value_0_to_fp16)[name = string("op_21426_cast_fp16")]; + tensor masked_logits_25_cast_fp16 = select(a = scaled_logits_25_cast_fp16, b = var_21426_cast_fp16, cond = var_21425_cast_fp16)[name = string("masked_logits_25_cast_fp16")]; + tensor var_21430_begin_0 = const()[name = string("op_21430_begin_0"), val = tensor([12, 0])]; + tensor var_21430_end_0 = const()[name = string("op_21430_end_0"), val = tensor([13, 2048])]; + tensor var_21430_end_mask_0 = const()[name = string("op_21430_end_mask_0"), val = tensor([false, true])]; + tensor var_21430_squeeze_mask_0 = const()[name = string("op_21430_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_21430_cast_fp16 = slice_by_index(begin = var_21430_begin_0, end = var_21430_end_0, end_mask = var_21430_end_mask_0, squeeze_mask = var_21430_squeeze_mask_0, x = gumbel)[name = string("op_21430_cast_fp16")]; + tensor var_21433 = const()[name = string("op_21433"), val = tensor([1, 2048])]; + tensor var_21434_cast_fp16 = reshape(shape = var_21433, x = var_21430_cast_fp16)[name = string("op_21434_cast_fp16")]; + tensor noisy_logits_25_cast_fp16 = add(x = masked_logits_25_cast_fp16, y = var_21434_cast_fp16)[name = string("noisy_logits_25_cast_fp16")]; + int32 code_25_axis_0 = const()[name = string("code_25_axis_0"), val = int32(1)]; + bool code_25_keep_dims_0 = const()[name = string("code_25_keep_dims_0"), val = bool(false)]; + string code_25_output_dtype_0 = const()[name = string("code_25_output_dtype_0"), val = string("int32")]; + tensor code_25_cast_fp16 = reduce_argmax(axis = code_25_axis_0, keep_dims = code_25_keep_dims_0, output_dtype = code_25_output_dtype_0, x = noisy_logits_25_cast_fp16)[name = string("code_25_cast_fp16")]; + int32 var_21445 = const()[name = string("op_21445"), val = int32(24576)]; + tensor input_611 = add(x = code_25_cast_fp16, y = var_21445)[name = string("input_611")]; + int32 code_embed_49_axis_0 = const()[name = string("code_embed_49_axis_0"), val = int32(0)]; + int32 code_embed_49_batch_dims_0 = const()[name = string("code_embed_49_batch_dims_0"), val = int32(0)]; + bool code_embed_49_validate_indices_0 = const()[name = string("code_embed_49_validate_indices_0"), val = bool(false)]; + string input_611_to_uint16_dtype_0 = const()[name = string("input_611_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_611_to_uint16 = cast(dtype = input_611_to_uint16_dtype_0, x = input_611)[name = string("cast_2")]; + tensor code_embed_49_cast_fp16_cast_uint16 = gather(axis = code_embed_49_axis_0, batch_dims = code_embed_49_batch_dims_0, indices = input_611_to_uint16, validate_indices = code_embed_49_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_49_cast_fp16_cast_uint16")]; + tensor var_21449 = const()[name = string("op_21449"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_51_cast_fp16 = reshape(shape = var_21449, x = code_embed_49_cast_fp16_cast_uint16)[name = string("code_embed_51_cast_fp16")]; + tensor embed_sum_27_cast_fp16 = add(x = embed_sum_25_cast_fp16, y = code_embed_51_cast_fp16)[name = string("embed_sum_27_cast_fp16")]; + string inputs_587_pad_type_0 = const()[name = string("inputs_587_pad_type_0"), val = string("valid")]; + tensor inputs_587_strides_0 = const()[name = string("inputs_587_strides_0"), val = tensor([1, 1])]; + tensor inputs_587_pad_0 = const()[name = string("inputs_587_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_587_dilations_0 = const()[name = string("inputs_587_dilations_0"), val = tensor([1, 1])]; + int32 inputs_587_groups_0 = const()[name = string("inputs_587_groups_0"), val = int32(1)]; + tensor inputs_587_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_587_dilations_0, groups = inputs_587_groups_0, pad = inputs_587_pad_0, pad_type = inputs_587_pad_type_0, strides = inputs_587_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_51_cast_fp16)[name = string("inputs_587_cast_fp16")]; + tensor obj_619_begin_0 = const()[name = string("obj_619_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_619_end_0 = const()[name = string("obj_619_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_619_end_mask_0 = const()[name = string("obj_619_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_619_cast_fp16 = slice_by_index(begin = obj_619_begin_0, end = obj_619_end_0, end_mask = obj_619_end_mask_0, x = key_caches_29_cast_fp16)[name = string("obj_619_cast_fp16")]; + tensor obj_621_begin_0 = const()[name = string("obj_621_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_621_end_0 = const()[name = string("obj_621_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_621_end_mask_0 = const()[name = string("obj_621_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_621_cast_fp16 = slice_by_index(begin = obj_621_begin_0, end = obj_621_end_0, end_mask = obj_621_end_mask_0, x = value_caches_29_cast_fp16)[name = string("obj_621_cast_fp16")]; + int32 var_21554 = const()[name = string("op_21554"), val = int32(3)]; + int32 var_21564 = const()[name = string("op_21564"), val = int32(-2)]; + tensor inputs_sq_587_cast_fp16 = mul(x = inputs_587_cast_fp16, y = inputs_587_cast_fp16)[name = string("inputs_sq_587_cast_fp16")]; + tensor variance_587_axes_0 = const()[name = string("variance_587_axes_0"), val = tensor([1])]; + bool variance_587_keep_dims_0 = const()[name = string("variance_587_keep_dims_0"), val = bool(true)]; + tensor variance_587_cast_fp16 = reduce_mean(axes = variance_587_axes_0, keep_dims = variance_587_keep_dims_0, x = inputs_sq_587_cast_fp16)[name = string("variance_587_cast_fp16")]; + fp16 var_21578_to_fp16 = const()[name = string("op_21578_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21579_cast_fp16 = add(x = variance_587_cast_fp16, y = var_21578_to_fp16)[name = string("op_21579_cast_fp16")]; + fp32 var_21580_epsilon_0 = const()[name = string("op_21580_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21580_cast_fp16 = rsqrt(epsilon = var_21580_epsilon_0, x = var_21579_cast_fp16)[name = string("op_21580_cast_fp16")]; + tensor hidden_states_727_cast_fp16 = mul(x = inputs_587_cast_fp16, y = var_21580_cast_fp16)[name = string("hidden_states_727_cast_fp16")]; + tensor obj_617_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_727_cast_fp16)[name = string("obj_617_cast_fp16")]; + string query_421_pad_type_0 = const()[name = string("query_421_pad_type_0"), val = string("valid")]; + tensor query_421_strides_0 = const()[name = string("query_421_strides_0"), val = tensor([1, 1])]; + tensor query_421_pad_0 = const()[name = string("query_421_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_421_dilations_0 = const()[name = string("query_421_dilations_0"), val = tensor([1, 1])]; + int32 query_421_groups_0 = const()[name = string("query_421_groups_0"), val = int32(1)]; + tensor query_421_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_421_dilations_0, groups = query_421_groups_0, pad = query_421_pad_0, pad_type = query_421_pad_type_0, strides = query_421_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_617_cast_fp16)[name = string("query_421_cast_fp16")]; + string current_key_281_pad_type_0 = const()[name = string("current_key_281_pad_type_0"), val = string("valid")]; + tensor current_key_281_strides_0 = const()[name = string("current_key_281_strides_0"), val = tensor([1, 1])]; + tensor current_key_281_pad_0 = const()[name = string("current_key_281_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_281_dilations_0 = const()[name = string("current_key_281_dilations_0"), val = tensor([1, 1])]; + int32 current_key_281_groups_0 = const()[name = string("current_key_281_groups_0"), val = int32(1)]; + tensor current_key_281_cast_fp16 = conv(dilations = current_key_281_dilations_0, groups = current_key_281_groups_0, pad = current_key_281_pad_0, pad_type = current_key_281_pad_type_0, strides = current_key_281_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_617_cast_fp16)[name = string("current_key_281_cast_fp16")]; + string current_value_141_pad_type_0 = const()[name = string("current_value_141_pad_type_0"), val = string("valid")]; + tensor current_value_141_strides_0 = const()[name = string("current_value_141_strides_0"), val = tensor([1, 1])]; + tensor current_value_141_pad_0 = const()[name = string("current_value_141_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_141_dilations_0 = const()[name = string("current_value_141_dilations_0"), val = tensor([1, 1])]; + int32 current_value_141_groups_0 = const()[name = string("current_value_141_groups_0"), val = int32(1)]; + tensor current_value_141_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_141_dilations_0, groups = current_value_141_groups_0, pad = current_value_141_pad_0, pad_type = current_value_141_pad_type_0, strides = current_value_141_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_617_cast_fp16)[name = string("current_value_141_cast_fp16")]; + tensor var_21617 = const()[name = string("op_21617"), val = tensor([16, 128, 1, 1])]; + tensor inputs_589_cast_fp16 = reshape(shape = var_21617, x = query_421_cast_fp16)[name = string("inputs_589_cast_fp16")]; + tensor inputs_sq_589_cast_fp16 = mul(x = inputs_589_cast_fp16, y = inputs_589_cast_fp16)[name = string("inputs_sq_589_cast_fp16")]; + tensor variance_589_axes_0 = const()[name = string("variance_589_axes_0"), val = tensor([1])]; + bool variance_589_keep_dims_0 = const()[name = string("variance_589_keep_dims_0"), val = bool(true)]; + tensor variance_589_cast_fp16 = reduce_mean(axes = variance_589_axes_0, keep_dims = variance_589_keep_dims_0, x = inputs_sq_589_cast_fp16)[name = string("variance_589_cast_fp16")]; + fp16 var_21623_to_fp16 = const()[name = string("op_21623_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21624_cast_fp16 = add(x = variance_589_cast_fp16, y = var_21623_to_fp16)[name = string("op_21624_cast_fp16")]; + fp32 var_21625_epsilon_0 = const()[name = string("op_21625_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21625_cast_fp16 = rsqrt(epsilon = var_21625_epsilon_0, x = var_21624_cast_fp16)[name = string("op_21625_cast_fp16")]; + tensor hidden_states_729_cast_fp16 = mul(x = inputs_589_cast_fp16, y = var_21625_cast_fp16)[name = string("hidden_states_729_cast_fp16")]; + tensor query_normed_141_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_729_cast_fp16)[name = string("query_normed_141_cast_fp16")]; + tensor var_21633 = const()[name = string("op_21633"), val = tensor([8, 128, 1, 1])]; + tensor inputs_591_cast_fp16 = reshape(shape = var_21633, x = current_key_281_cast_fp16)[name = string("inputs_591_cast_fp16")]; + tensor inputs_sq_591_cast_fp16 = mul(x = inputs_591_cast_fp16, y = inputs_591_cast_fp16)[name = string("inputs_sq_591_cast_fp16")]; + tensor variance_591_axes_0 = const()[name = string("variance_591_axes_0"), val = tensor([1])]; + bool variance_591_keep_dims_0 = const()[name = string("variance_591_keep_dims_0"), val = bool(true)]; + tensor variance_591_cast_fp16 = reduce_mean(axes = variance_591_axes_0, keep_dims = variance_591_keep_dims_0, x = inputs_sq_591_cast_fp16)[name = string("variance_591_cast_fp16")]; + fp16 var_21639_to_fp16 = const()[name = string("op_21639_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21640_cast_fp16 = add(x = variance_591_cast_fp16, y = var_21639_to_fp16)[name = string("op_21640_cast_fp16")]; + fp32 var_21641_epsilon_0 = const()[name = string("op_21641_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21641_cast_fp16 = rsqrt(epsilon = var_21641_epsilon_0, x = var_21640_cast_fp16)[name = string("op_21641_cast_fp16")]; + tensor hidden_states_731_cast_fp16 = mul(x = inputs_591_cast_fp16, y = var_21641_cast_fp16)[name = string("hidden_states_731_cast_fp16")]; + tensor current_key_normed_141_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_731_cast_fp16)[name = string("current_key_normed_141_cast_fp16")]; + tensor var_21659 = const()[name = string("op_21659"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_561_cast_fp16 = reshape(shape = var_21659, x = query_normed_141_cast_fp16)[name = string("mh_q_561_cast_fp16")]; + tensor var_21661 = const()[name = string("op_21661"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_561_cast_fp16 = reshape(shape = var_21661, x = current_key_normed_141_cast_fp16)[name = string("mh_k_561_cast_fp16")]; + tensor cos_141_to_fp16 = const()[name = string("cos_141_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175202880)))]; + tensor var_21665_cast_fp16 = mul(x = mh_q_561_cast_fp16, y = cos_141_to_fp16)[name = string("op_21665_cast_fp16")]; + tensor var_21670_begin_0 = const()[name = string("op_21670_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_21670_end_0 = const()[name = string("op_21670_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_21670_end_mask_0 = const()[name = string("op_21670_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_21670_cast_fp16 = slice_by_index(begin = var_21670_begin_0, end = var_21670_end_0, end_mask = var_21670_end_mask_0, x = mh_q_561_cast_fp16)[name = string("op_21670_cast_fp16")]; + tensor var_21676_begin_0 = const()[name = string("op_21676_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_21676_end_0 = const()[name = string("op_21676_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_21676_end_mask_0 = const()[name = string("op_21676_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_21676_cast_fp16 = slice_by_index(begin = var_21676_begin_0, end = var_21676_end_0, end_mask = var_21676_end_mask_0, x = mh_q_561_cast_fp16)[name = string("op_21676_cast_fp16")]; + fp16 const_1428_promoted_to_fp16 = const()[name = string("const_1428_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_21678_cast_fp16 = mul(x = var_21676_cast_fp16, y = const_1428_promoted_to_fp16)[name = string("op_21678_cast_fp16")]; + bool var_21680_interleave_0 = const()[name = string("op_21680_interleave_0"), val = bool(false)]; + tensor var_21680_cast_fp16 = concat(axis = var_21564, interleave = var_21680_interleave_0, values = (var_21678_cast_fp16, var_21670_cast_fp16))[name = string("op_21680_cast_fp16")]; + tensor sin_141_to_fp16 = const()[name = string("sin_141_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203200)))]; + tensor var_21681_cast_fp16 = mul(x = var_21680_cast_fp16, y = sin_141_to_fp16)[name = string("op_21681_cast_fp16")]; + tensor mh_q_563_cast_fp16 = add(x = var_21665_cast_fp16, y = var_21681_cast_fp16)[name = string("mh_q_563_cast_fp16")]; + tensor var_21683_cast_fp16 = mul(x = mh_k_561_cast_fp16, y = cos_141_to_fp16)[name = string("op_21683_cast_fp16")]; + tensor var_21688_begin_0 = const()[name = string("op_21688_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_21688_end_0 = const()[name = string("op_21688_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_21688_end_mask_0 = const()[name = string("op_21688_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_21688_cast_fp16 = slice_by_index(begin = var_21688_begin_0, end = var_21688_end_0, end_mask = var_21688_end_mask_0, x = mh_k_561_cast_fp16)[name = string("op_21688_cast_fp16")]; + tensor var_21694_begin_0 = const()[name = string("op_21694_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_21694_end_0 = const()[name = string("op_21694_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_21694_end_mask_0 = const()[name = string("op_21694_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_21694_cast_fp16 = slice_by_index(begin = var_21694_begin_0, end = var_21694_end_0, end_mask = var_21694_end_mask_0, x = mh_k_561_cast_fp16)[name = string("op_21694_cast_fp16")]; + fp16 const_1431_promoted_to_fp16 = const()[name = string("const_1431_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_21696_cast_fp16 = mul(x = var_21694_cast_fp16, y = const_1431_promoted_to_fp16)[name = string("op_21696_cast_fp16")]; + bool var_21698_interleave_0 = const()[name = string("op_21698_interleave_0"), val = bool(false)]; + tensor var_21698_cast_fp16 = concat(axis = var_21564, interleave = var_21698_interleave_0, values = (var_21696_cast_fp16, var_21688_cast_fp16))[name = string("op_21698_cast_fp16")]; + tensor var_21699_cast_fp16 = mul(x = var_21698_cast_fp16, y = sin_141_to_fp16)[name = string("op_21699_cast_fp16")]; + tensor mh_k_563_cast_fp16 = add(x = var_21683_cast_fp16, y = var_21699_cast_fp16)[name = string("mh_k_563_cast_fp16")]; + tensor var_21703 = const()[name = string("op_21703"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_283_cast_fp16 = reshape(shape = var_21703, x = mh_k_563_cast_fp16)[name = string("current_key_283_cast_fp16")]; + tensor var_21709_to_fp16 = const()[name = string("op_21709_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203520)))]; + tensor var_21710_cast_fp16 = mul(x = obj_619_cast_fp16, y = var_21709_to_fp16)[name = string("op_21710_cast_fp16")]; + tensor var_21707_to_fp16 = const()[name = string("op_21707_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203648)))]; + tensor var_21711_cast_fp16 = mul(x = current_key_283_cast_fp16, y = var_21707_to_fp16)[name = string("op_21711_cast_fp16")]; + tensor key_283_cast_fp16 = add(x = var_21710_cast_fp16, y = var_21711_cast_fp16)[name = string("key_283_cast_fp16")]; + tensor var_21713_to_fp16 = const()[name = string("op_21713_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203520)))]; + tensor var_21714_cast_fp16 = mul(x = obj_621_cast_fp16, y = var_21713_to_fp16)[name = string("op_21714_cast_fp16")]; + tensor var_21715_cast_fp16 = mul(x = current_value_141_cast_fp16, y = var_21707_to_fp16)[name = string("op_21715_cast_fp16")]; + tensor value_141_cast_fp16 = add(x = var_21714_cast_fp16, y = var_21715_cast_fp16)[name = string("value_141_cast_fp16")]; + fp16 var_21722_to_fp16 = const()[name = string("op_21722_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_567_cast_fp16 = mul(x = mh_q_563_cast_fp16, y = var_21722_to_fp16)[name = string("mh_q_567_cast_fp16")]; + tensor var_21724 = const()[name = string("op_21724"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_565_cast_fp16 = reshape(shape = var_21724, x = key_283_cast_fp16)[name = string("mh_k_565_cast_fp16")]; + tensor var_21726 = const()[name = string("op_21726"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_281_cast_fp16 = reshape(shape = var_21726, x = value_141_cast_fp16)[name = string("mh_v_281_cast_fp16")]; + tensor transpose_280_perm_0 = const()[name = string("transpose_280_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_140_reps_0 = const()[name = string("tile_140_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_280_cast_fp16 = transpose(perm = transpose_280_perm_0, x = mh_k_565_cast_fp16)[name = string("transpose_59")]; + tensor tile_140_cast_fp16 = tile(reps = tile_140_reps_0, x = transpose_280_cast_fp16)[name = string("tile_140_cast_fp16")]; + tensor concat_353 = const()[name = string("concat_353"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_280_cast_fp16 = reshape(shape = concat_353, x = tile_140_cast_fp16)[name = string("reshape_280_cast_fp16")]; + tensor transpose_281_perm_0 = const()[name = string("transpose_281_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_354 = const()[name = string("concat_354"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_281_cast_fp16 = transpose(perm = transpose_281_perm_0, x = reshape_280_cast_fp16)[name = string("transpose_58")]; + tensor reshape_281_cast_fp16 = reshape(shape = concat_354, x = transpose_281_cast_fp16)[name = string("reshape_281_cast_fp16")]; + tensor transpose_282_perm_0 = const()[name = string("transpose_282_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_141_reps_0 = const()[name = string("tile_141_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_282_cast_fp16 = transpose(perm = transpose_282_perm_0, x = mh_v_281_cast_fp16)[name = string("transpose_57")]; + tensor tile_141_cast_fp16 = tile(reps = tile_141_reps_0, x = transpose_282_cast_fp16)[name = string("tile_141_cast_fp16")]; + tensor concat_355 = const()[name = string("concat_355"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_282_cast_fp16 = reshape(shape = concat_355, x = tile_141_cast_fp16)[name = string("reshape_282_cast_fp16")]; + tensor transpose_283_perm_0 = const()[name = string("transpose_283_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_356 = const()[name = string("concat_356"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_283_cast_fp16 = transpose(perm = transpose_283_perm_0, x = reshape_282_cast_fp16)[name = string("transpose_56")]; + tensor reshape_283_cast_fp16 = reshape(shape = concat_356, x = transpose_283_cast_fp16)[name = string("reshape_283_cast_fp16")]; + tensor transpose_597_perm_0 = const()[name = string("transpose_597_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_421_transpose_x_1 = const()[name = string("mh_w_421_transpose_x_1"), val = bool(true)]; + bool mh_w_421_transpose_y_1 = const()[name = string("mh_w_421_transpose_y_1"), val = bool(false)]; + tensor transpose_597_cast_fp16 = transpose(perm = transpose_597_perm_0, x = reshape_281_cast_fp16)[name = string("transpose_55")]; + tensor mh_w_421_cast_fp16 = matmul(transpose_x = mh_w_421_transpose_x_1, transpose_y = mh_w_421_transpose_y_1, x = mh_q_567_cast_fp16, y = transpose_597_cast_fp16)[name = string("mh_w_421_cast_fp16")]; + tensor var_21734_to_fp16 = const()[name = string("op_21734_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203776)))]; + tensor mh_w_423_cast_fp16 = add(x = mh_w_421_cast_fp16, y = var_21734_to_fp16)[name = string("mh_w_423_cast_fp16")]; + tensor mh_w_425_cast_fp16 = softmax(axis = var_21554, x = mh_w_423_cast_fp16)[name = string("mh_w_425_cast_fp16")]; + tensor transpose_598_perm_0 = const()[name = string("transpose_598_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_141_transpose_x_1 = const()[name = string("attn_141_transpose_x_1"), val = bool(false)]; + bool attn_141_transpose_y_1 = const()[name = string("attn_141_transpose_y_1"), val = bool(true)]; + tensor transpose_598_cast_fp16 = transpose(perm = transpose_598_perm_0, x = reshape_283_cast_fp16)[name = string("transpose_54")]; + tensor attn_141_cast_fp16 = matmul(transpose_x = attn_141_transpose_x_1, transpose_y = attn_141_transpose_y_1, x = transpose_598_cast_fp16, y = mh_w_425_cast_fp16)[name = string("attn_141_cast_fp16")]; + tensor var_21740 = const()[name = string("op_21740"), val = tensor([1, 2048, 1, 1])]; + tensor input_613_cast_fp16 = reshape(shape = var_21740, x = attn_141_cast_fp16)[name = string("input_613_cast_fp16")]; + string obj_627_pad_type_0 = const()[name = string("obj_627_pad_type_0"), val = string("valid")]; + tensor obj_627_strides_0 = const()[name = string("obj_627_strides_0"), val = tensor([1, 1])]; + tensor obj_627_pad_0 = const()[name = string("obj_627_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_627_dilations_0 = const()[name = string("obj_627_dilations_0"), val = tensor([1, 1])]; + int32 obj_627_groups_0 = const()[name = string("obj_627_groups_0"), val = int32(1)]; + tensor obj_627_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_627_dilations_0, groups = obj_627_groups_0, pad = obj_627_pad_0, pad_type = obj_627_pad_type_0, strides = obj_627_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_613_cast_fp16)[name = string("obj_627_cast_fp16")]; + tensor inputs_593_cast_fp16 = add(x = inputs_587_cast_fp16, y = obj_627_cast_fp16)[name = string("inputs_593_cast_fp16")]; + tensor inputs_sq_593_cast_fp16 = mul(x = inputs_593_cast_fp16, y = inputs_593_cast_fp16)[name = string("inputs_sq_593_cast_fp16")]; + tensor variance_593_axes_0 = const()[name = string("variance_593_axes_0"), val = tensor([1])]; + bool variance_593_keep_dims_0 = const()[name = string("variance_593_keep_dims_0"), val = bool(true)]; + tensor variance_593_cast_fp16 = reduce_mean(axes = variance_593_axes_0, keep_dims = variance_593_keep_dims_0, x = inputs_sq_593_cast_fp16)[name = string("variance_593_cast_fp16")]; + fp16 var_21758_to_fp16 = const()[name = string("op_21758_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21759_cast_fp16 = add(x = variance_593_cast_fp16, y = var_21758_to_fp16)[name = string("op_21759_cast_fp16")]; + fp32 var_21760_epsilon_0 = const()[name = string("op_21760_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21760_cast_fp16 = rsqrt(epsilon = var_21760_epsilon_0, x = var_21759_cast_fp16)[name = string("op_21760_cast_fp16")]; + tensor hidden_states_733_cast_fp16 = mul(x = inputs_593_cast_fp16, y = var_21760_cast_fp16)[name = string("hidden_states_733_cast_fp16")]; + tensor input_615_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_733_cast_fp16)[name = string("input_615_cast_fp16")]; + string input_617_pad_type_0 = const()[name = string("input_617_pad_type_0"), val = string("valid")]; + tensor input_617_strides_0 = const()[name = string("input_617_strides_0"), val = tensor([1, 1])]; + tensor input_617_pad_0 = const()[name = string("input_617_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_617_dilations_0 = const()[name = string("input_617_dilations_0"), val = tensor([1, 1])]; + int32 input_617_groups_0 = const()[name = string("input_617_groups_0"), val = int32(1)]; + tensor input_617_cast_fp16 = conv(dilations = input_617_dilations_0, groups = input_617_groups_0, pad = input_617_pad_0, pad_type = input_617_pad_type_0, strides = input_617_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_615_cast_fp16)[name = string("input_617_cast_fp16")]; + tensor var_21774_cast_fp16 = silu(x = input_617_cast_fp16)[name = string("op_21774_cast_fp16")]; + string var_21780_pad_type_0 = const()[name = string("op_21780_pad_type_0"), val = string("valid")]; + tensor var_21780_strides_0 = const()[name = string("op_21780_strides_0"), val = tensor([1, 1])]; + tensor var_21780_pad_0 = const()[name = string("op_21780_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_21780_dilations_0 = const()[name = string("op_21780_dilations_0"), val = tensor([1, 1])]; + int32 var_21780_groups_0 = const()[name = string("op_21780_groups_0"), val = int32(1)]; + tensor var_21780_cast_fp16 = conv(dilations = var_21780_dilations_0, groups = var_21780_groups_0, pad = var_21780_pad_0, pad_type = var_21780_pad_type_0, strides = var_21780_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_615_cast_fp16)[name = string("op_21780_cast_fp16")]; + tensor input_619_cast_fp16 = mul(x = var_21774_cast_fp16, y = var_21780_cast_fp16)[name = string("input_619_cast_fp16")]; + string hidden_states_735_pad_type_0 = const()[name = string("hidden_states_735_pad_type_0"), val = string("valid")]; + tensor hidden_states_735_strides_0 = const()[name = string("hidden_states_735_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_735_pad_0 = const()[name = string("hidden_states_735_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_735_dilations_0 = const()[name = string("hidden_states_735_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_735_groups_0 = const()[name = string("hidden_states_735_groups_0"), val = int32(1)]; + tensor hidden_states_735_cast_fp16 = conv(dilations = hidden_states_735_dilations_0, groups = hidden_states_735_groups_0, pad = hidden_states_735_pad_0, pad_type = hidden_states_735_pad_type_0, strides = hidden_states_735_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_619_cast_fp16)[name = string("hidden_states_735_cast_fp16")]; + tensor inputs_595_cast_fp16 = add(x = inputs_593_cast_fp16, y = hidden_states_735_cast_fp16)[name = string("inputs_595_cast_fp16")]; + tensor obj_631_begin_0 = const()[name = string("obj_631_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_631_end_0 = const()[name = string("obj_631_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_631_end_mask_0 = const()[name = string("obj_631_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_631_cast_fp16 = slice_by_index(begin = obj_631_begin_0, end = obj_631_end_0, end_mask = obj_631_end_mask_0, x = key_caches_29_cast_fp16)[name = string("obj_631_cast_fp16")]; + tensor obj_633_begin_0 = const()[name = string("obj_633_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_633_end_0 = const()[name = string("obj_633_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_633_end_mask_0 = const()[name = string("obj_633_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_633_cast_fp16 = slice_by_index(begin = obj_633_begin_0, end = obj_633_end_0, end_mask = obj_633_end_mask_0, x = value_caches_29_cast_fp16)[name = string("obj_633_cast_fp16")]; + int32 var_21828 = const()[name = string("op_21828"), val = int32(3)]; + int32 var_21838 = const()[name = string("op_21838"), val = int32(-2)]; + tensor inputs_sq_595_cast_fp16 = mul(x = inputs_595_cast_fp16, y = inputs_595_cast_fp16)[name = string("inputs_sq_595_cast_fp16")]; + tensor variance_595_axes_0 = const()[name = string("variance_595_axes_0"), val = tensor([1])]; + bool variance_595_keep_dims_0 = const()[name = string("variance_595_keep_dims_0"), val = bool(true)]; + tensor variance_595_cast_fp16 = reduce_mean(axes = variance_595_axes_0, keep_dims = variance_595_keep_dims_0, x = inputs_sq_595_cast_fp16)[name = string("variance_595_cast_fp16")]; + fp16 var_21852_to_fp16 = const()[name = string("op_21852_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21853_cast_fp16 = add(x = variance_595_cast_fp16, y = var_21852_to_fp16)[name = string("op_21853_cast_fp16")]; + fp32 var_21854_epsilon_0 = const()[name = string("op_21854_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21854_cast_fp16 = rsqrt(epsilon = var_21854_epsilon_0, x = var_21853_cast_fp16)[name = string("op_21854_cast_fp16")]; + tensor hidden_states_737_cast_fp16 = mul(x = inputs_595_cast_fp16, y = var_21854_cast_fp16)[name = string("hidden_states_737_cast_fp16")]; + tensor obj_629_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_737_cast_fp16)[name = string("obj_629_cast_fp16")]; + string query_427_pad_type_0 = const()[name = string("query_427_pad_type_0"), val = string("valid")]; + tensor query_427_strides_0 = const()[name = string("query_427_strides_0"), val = tensor([1, 1])]; + tensor query_427_pad_0 = const()[name = string("query_427_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_427_dilations_0 = const()[name = string("query_427_dilations_0"), val = tensor([1, 1])]; + int32 query_427_groups_0 = const()[name = string("query_427_groups_0"), val = int32(1)]; + tensor query_427_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_427_dilations_0, groups = query_427_groups_0, pad = query_427_pad_0, pad_type = query_427_pad_type_0, strides = query_427_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_629_cast_fp16)[name = string("query_427_cast_fp16")]; + string current_key_285_pad_type_0 = const()[name = string("current_key_285_pad_type_0"), val = string("valid")]; + tensor current_key_285_strides_0 = const()[name = string("current_key_285_strides_0"), val = tensor([1, 1])]; + tensor current_key_285_pad_0 = const()[name = string("current_key_285_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_285_dilations_0 = const()[name = string("current_key_285_dilations_0"), val = tensor([1, 1])]; + int32 current_key_285_groups_0 = const()[name = string("current_key_285_groups_0"), val = int32(1)]; + tensor current_key_285_cast_fp16 = conv(dilations = current_key_285_dilations_0, groups = current_key_285_groups_0, pad = current_key_285_pad_0, pad_type = current_key_285_pad_type_0, strides = current_key_285_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_629_cast_fp16)[name = string("current_key_285_cast_fp16")]; + string current_value_143_pad_type_0 = const()[name = string("current_value_143_pad_type_0"), val = string("valid")]; + tensor current_value_143_strides_0 = const()[name = string("current_value_143_strides_0"), val = tensor([1, 1])]; + tensor current_value_143_pad_0 = const()[name = string("current_value_143_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_143_dilations_0 = const()[name = string("current_value_143_dilations_0"), val = tensor([1, 1])]; + int32 current_value_143_groups_0 = const()[name = string("current_value_143_groups_0"), val = int32(1)]; + tensor current_value_143_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_143_dilations_0, groups = current_value_143_groups_0, pad = current_value_143_pad_0, pad_type = current_value_143_pad_type_0, strides = current_value_143_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_629_cast_fp16)[name = string("current_value_143_cast_fp16")]; + tensor var_21891 = const()[name = string("op_21891"), val = tensor([16, 128, 1, 1])]; + tensor inputs_597_cast_fp16 = reshape(shape = var_21891, x = query_427_cast_fp16)[name = string("inputs_597_cast_fp16")]; + tensor inputs_sq_597_cast_fp16 = mul(x = inputs_597_cast_fp16, y = inputs_597_cast_fp16)[name = string("inputs_sq_597_cast_fp16")]; + tensor variance_597_axes_0 = const()[name = string("variance_597_axes_0"), val = tensor([1])]; + bool variance_597_keep_dims_0 = const()[name = string("variance_597_keep_dims_0"), val = bool(true)]; + tensor variance_597_cast_fp16 = reduce_mean(axes = variance_597_axes_0, keep_dims = variance_597_keep_dims_0, x = inputs_sq_597_cast_fp16)[name = string("variance_597_cast_fp16")]; + fp16 var_21897_to_fp16 = const()[name = string("op_21897_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21898_cast_fp16 = add(x = variance_597_cast_fp16, y = var_21897_to_fp16)[name = string("op_21898_cast_fp16")]; + fp32 var_21899_epsilon_0 = const()[name = string("op_21899_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21899_cast_fp16 = rsqrt(epsilon = var_21899_epsilon_0, x = var_21898_cast_fp16)[name = string("op_21899_cast_fp16")]; + tensor hidden_states_739_cast_fp16 = mul(x = inputs_597_cast_fp16, y = var_21899_cast_fp16)[name = string("hidden_states_739_cast_fp16")]; + tensor query_normed_143_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_739_cast_fp16)[name = string("query_normed_143_cast_fp16")]; + tensor var_21907 = const()[name = string("op_21907"), val = tensor([8, 128, 1, 1])]; + tensor inputs_599_cast_fp16 = reshape(shape = var_21907, x = current_key_285_cast_fp16)[name = string("inputs_599_cast_fp16")]; + tensor inputs_sq_599_cast_fp16 = mul(x = inputs_599_cast_fp16, y = inputs_599_cast_fp16)[name = string("inputs_sq_599_cast_fp16")]; + tensor variance_599_axes_0 = const()[name = string("variance_599_axes_0"), val = tensor([1])]; + bool variance_599_keep_dims_0 = const()[name = string("variance_599_keep_dims_0"), val = bool(true)]; + tensor variance_599_cast_fp16 = reduce_mean(axes = variance_599_axes_0, keep_dims = variance_599_keep_dims_0, x = inputs_sq_599_cast_fp16)[name = string("variance_599_cast_fp16")]; + fp16 var_21913_to_fp16 = const()[name = string("op_21913_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_21914_cast_fp16 = add(x = variance_599_cast_fp16, y = var_21913_to_fp16)[name = string("op_21914_cast_fp16")]; + fp32 var_21915_epsilon_0 = const()[name = string("op_21915_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_21915_cast_fp16 = rsqrt(epsilon = var_21915_epsilon_0, x = var_21914_cast_fp16)[name = string("op_21915_cast_fp16")]; + tensor hidden_states_741_cast_fp16 = mul(x = inputs_599_cast_fp16, y = var_21915_cast_fp16)[name = string("hidden_states_741_cast_fp16")]; + tensor current_key_normed_143_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_741_cast_fp16)[name = string("current_key_normed_143_cast_fp16")]; + tensor var_21933 = const()[name = string("op_21933"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_569_cast_fp16 = reshape(shape = var_21933, x = query_normed_143_cast_fp16)[name = string("mh_q_569_cast_fp16")]; + tensor var_21935 = const()[name = string("op_21935"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_569_cast_fp16 = reshape(shape = var_21935, x = current_key_normed_143_cast_fp16)[name = string("mh_k_569_cast_fp16")]; + tensor var_21939_cast_fp16 = mul(x = mh_q_569_cast_fp16, y = cos_141_to_fp16)[name = string("op_21939_cast_fp16")]; + tensor var_21944_begin_0 = const()[name = string("op_21944_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_21944_end_0 = const()[name = string("op_21944_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_21944_end_mask_0 = const()[name = string("op_21944_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_21944_cast_fp16 = slice_by_index(begin = var_21944_begin_0, end = var_21944_end_0, end_mask = var_21944_end_mask_0, x = mh_q_569_cast_fp16)[name = string("op_21944_cast_fp16")]; + tensor var_21950_begin_0 = const()[name = string("op_21950_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_21950_end_0 = const()[name = string("op_21950_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_21950_end_mask_0 = const()[name = string("op_21950_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_21950_cast_fp16 = slice_by_index(begin = var_21950_begin_0, end = var_21950_end_0, end_mask = var_21950_end_mask_0, x = mh_q_569_cast_fp16)[name = string("op_21950_cast_fp16")]; + fp16 const_1448_promoted_to_fp16 = const()[name = string("const_1448_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_21952_cast_fp16 = mul(x = var_21950_cast_fp16, y = const_1448_promoted_to_fp16)[name = string("op_21952_cast_fp16")]; + bool var_21954_interleave_0 = const()[name = string("op_21954_interleave_0"), val = bool(false)]; + tensor var_21954_cast_fp16 = concat(axis = var_21838, interleave = var_21954_interleave_0, values = (var_21952_cast_fp16, var_21944_cast_fp16))[name = string("op_21954_cast_fp16")]; + tensor var_21955_cast_fp16 = mul(x = var_21954_cast_fp16, y = sin_141_to_fp16)[name = string("op_21955_cast_fp16")]; + tensor mh_q_571_cast_fp16 = add(x = var_21939_cast_fp16, y = var_21955_cast_fp16)[name = string("mh_q_571_cast_fp16")]; + tensor var_21957_cast_fp16 = mul(x = mh_k_569_cast_fp16, y = cos_141_to_fp16)[name = string("op_21957_cast_fp16")]; + tensor var_21962_begin_0 = const()[name = string("op_21962_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_21962_end_0 = const()[name = string("op_21962_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_21962_end_mask_0 = const()[name = string("op_21962_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_21962_cast_fp16 = slice_by_index(begin = var_21962_begin_0, end = var_21962_end_0, end_mask = var_21962_end_mask_0, x = mh_k_569_cast_fp16)[name = string("op_21962_cast_fp16")]; + tensor var_21968_begin_0 = const()[name = string("op_21968_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_21968_end_0 = const()[name = string("op_21968_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_21968_end_mask_0 = const()[name = string("op_21968_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_21968_cast_fp16 = slice_by_index(begin = var_21968_begin_0, end = var_21968_end_0, end_mask = var_21968_end_mask_0, x = mh_k_569_cast_fp16)[name = string("op_21968_cast_fp16")]; + fp16 const_1451_promoted_to_fp16 = const()[name = string("const_1451_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_21970_cast_fp16 = mul(x = var_21968_cast_fp16, y = const_1451_promoted_to_fp16)[name = string("op_21970_cast_fp16")]; + bool var_21972_interleave_0 = const()[name = string("op_21972_interleave_0"), val = bool(false)]; + tensor var_21972_cast_fp16 = concat(axis = var_21838, interleave = var_21972_interleave_0, values = (var_21970_cast_fp16, var_21962_cast_fp16))[name = string("op_21972_cast_fp16")]; + tensor var_21973_cast_fp16 = mul(x = var_21972_cast_fp16, y = sin_141_to_fp16)[name = string("op_21973_cast_fp16")]; + tensor mh_k_571_cast_fp16 = add(x = var_21957_cast_fp16, y = var_21973_cast_fp16)[name = string("mh_k_571_cast_fp16")]; + tensor var_21977 = const()[name = string("op_21977"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_287_cast_fp16 = reshape(shape = var_21977, x = mh_k_571_cast_fp16)[name = string("current_key_287_cast_fp16")]; + tensor var_21983_to_fp16 = const()[name = string("op_21983_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203520)))]; + tensor var_21984_cast_fp16 = mul(x = obj_631_cast_fp16, y = var_21983_to_fp16)[name = string("op_21984_cast_fp16")]; + tensor var_21981_to_fp16 = const()[name = string("op_21981_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203648)))]; + tensor var_21985_cast_fp16 = mul(x = current_key_287_cast_fp16, y = var_21981_to_fp16)[name = string("op_21985_cast_fp16")]; + tensor key_287_cast_fp16 = add(x = var_21984_cast_fp16, y = var_21985_cast_fp16)[name = string("key_287_cast_fp16")]; + tensor var_21987_to_fp16 = const()[name = string("op_21987_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203520)))]; + tensor var_21988_cast_fp16 = mul(x = obj_633_cast_fp16, y = var_21987_to_fp16)[name = string("op_21988_cast_fp16")]; + tensor var_21989_cast_fp16 = mul(x = current_value_143_cast_fp16, y = var_21981_to_fp16)[name = string("op_21989_cast_fp16")]; + tensor value_143_cast_fp16 = add(x = var_21988_cast_fp16, y = var_21989_cast_fp16)[name = string("value_143_cast_fp16")]; + fp16 var_21996_to_fp16 = const()[name = string("op_21996_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_575_cast_fp16 = mul(x = mh_q_571_cast_fp16, y = var_21996_to_fp16)[name = string("mh_q_575_cast_fp16")]; + tensor var_21998 = const()[name = string("op_21998"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_573_cast_fp16 = reshape(shape = var_21998, x = key_287_cast_fp16)[name = string("mh_k_573_cast_fp16")]; + tensor var_22000 = const()[name = string("op_22000"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_285_cast_fp16 = reshape(shape = var_22000, x = value_143_cast_fp16)[name = string("mh_v_285_cast_fp16")]; + tensor transpose_284_perm_0 = const()[name = string("transpose_284_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_142_reps_0 = const()[name = string("tile_142_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_284_cast_fp16 = transpose(perm = transpose_284_perm_0, x = mh_k_573_cast_fp16)[name = string("transpose_53")]; + tensor tile_142_cast_fp16 = tile(reps = tile_142_reps_0, x = transpose_284_cast_fp16)[name = string("tile_142_cast_fp16")]; + tensor concat_357 = const()[name = string("concat_357"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_284_cast_fp16 = reshape(shape = concat_357, x = tile_142_cast_fp16)[name = string("reshape_284_cast_fp16")]; + tensor transpose_285_perm_0 = const()[name = string("transpose_285_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_358 = const()[name = string("concat_358"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_285_cast_fp16 = transpose(perm = transpose_285_perm_0, x = reshape_284_cast_fp16)[name = string("transpose_52")]; + tensor reshape_285_cast_fp16 = reshape(shape = concat_358, x = transpose_285_cast_fp16)[name = string("reshape_285_cast_fp16")]; + tensor transpose_286_perm_0 = const()[name = string("transpose_286_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_143_reps_0 = const()[name = string("tile_143_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_286_cast_fp16 = transpose(perm = transpose_286_perm_0, x = mh_v_285_cast_fp16)[name = string("transpose_51")]; + tensor tile_143_cast_fp16 = tile(reps = tile_143_reps_0, x = transpose_286_cast_fp16)[name = string("tile_143_cast_fp16")]; + tensor concat_359 = const()[name = string("concat_359"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_286_cast_fp16 = reshape(shape = concat_359, x = tile_143_cast_fp16)[name = string("reshape_286_cast_fp16")]; + tensor transpose_287_perm_0 = const()[name = string("transpose_287_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_360 = const()[name = string("concat_360"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_287_cast_fp16 = transpose(perm = transpose_287_perm_0, x = reshape_286_cast_fp16)[name = string("transpose_50")]; + tensor reshape_287_cast_fp16 = reshape(shape = concat_360, x = transpose_287_cast_fp16)[name = string("reshape_287_cast_fp16")]; + tensor transpose_601_perm_0 = const()[name = string("transpose_601_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_427_transpose_x_1 = const()[name = string("mh_w_427_transpose_x_1"), val = bool(true)]; + bool mh_w_427_transpose_y_1 = const()[name = string("mh_w_427_transpose_y_1"), val = bool(false)]; + tensor transpose_601_cast_fp16 = transpose(perm = transpose_601_perm_0, x = reshape_285_cast_fp16)[name = string("transpose_49")]; + tensor mh_w_427_cast_fp16 = matmul(transpose_x = mh_w_427_transpose_x_1, transpose_y = mh_w_427_transpose_y_1, x = mh_q_575_cast_fp16, y = transpose_601_cast_fp16)[name = string("mh_w_427_cast_fp16")]; + tensor var_22008_to_fp16 = const()[name = string("op_22008_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203776)))]; + tensor mh_w_429_cast_fp16 = add(x = mh_w_427_cast_fp16, y = var_22008_to_fp16)[name = string("mh_w_429_cast_fp16")]; + tensor mh_w_431_cast_fp16 = softmax(axis = var_21828, x = mh_w_429_cast_fp16)[name = string("mh_w_431_cast_fp16")]; + tensor transpose_602_perm_0 = const()[name = string("transpose_602_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_143_transpose_x_1 = const()[name = string("attn_143_transpose_x_1"), val = bool(false)]; + bool attn_143_transpose_y_1 = const()[name = string("attn_143_transpose_y_1"), val = bool(true)]; + tensor transpose_602_cast_fp16 = transpose(perm = transpose_602_perm_0, x = reshape_287_cast_fp16)[name = string("transpose_48")]; + tensor attn_143_cast_fp16 = matmul(transpose_x = attn_143_transpose_x_1, transpose_y = attn_143_transpose_y_1, x = transpose_602_cast_fp16, y = mh_w_431_cast_fp16)[name = string("attn_143_cast_fp16")]; + tensor var_22014 = const()[name = string("op_22014"), val = tensor([1, 2048, 1, 1])]; + tensor input_621_cast_fp16 = reshape(shape = var_22014, x = attn_143_cast_fp16)[name = string("input_621_cast_fp16")]; + string obj_635_pad_type_0 = const()[name = string("obj_635_pad_type_0"), val = string("valid")]; + tensor obj_635_strides_0 = const()[name = string("obj_635_strides_0"), val = tensor([1, 1])]; + tensor obj_635_pad_0 = const()[name = string("obj_635_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_635_dilations_0 = const()[name = string("obj_635_dilations_0"), val = tensor([1, 1])]; + int32 obj_635_groups_0 = const()[name = string("obj_635_groups_0"), val = int32(1)]; + tensor obj_635_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_635_dilations_0, groups = obj_635_groups_0, pad = obj_635_pad_0, pad_type = obj_635_pad_type_0, strides = obj_635_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_621_cast_fp16)[name = string("obj_635_cast_fp16")]; + tensor inputs_601_cast_fp16 = add(x = inputs_595_cast_fp16, y = obj_635_cast_fp16)[name = string("inputs_601_cast_fp16")]; + tensor inputs_sq_601_cast_fp16 = mul(x = inputs_601_cast_fp16, y = inputs_601_cast_fp16)[name = string("inputs_sq_601_cast_fp16")]; + tensor variance_601_axes_0 = const()[name = string("variance_601_axes_0"), val = tensor([1])]; + bool variance_601_keep_dims_0 = const()[name = string("variance_601_keep_dims_0"), val = bool(true)]; + tensor variance_601_cast_fp16 = reduce_mean(axes = variance_601_axes_0, keep_dims = variance_601_keep_dims_0, x = inputs_sq_601_cast_fp16)[name = string("variance_601_cast_fp16")]; + fp16 var_22032_to_fp16 = const()[name = string("op_22032_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22033_cast_fp16 = add(x = variance_601_cast_fp16, y = var_22032_to_fp16)[name = string("op_22033_cast_fp16")]; + fp32 var_22034_epsilon_0 = const()[name = string("op_22034_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22034_cast_fp16 = rsqrt(epsilon = var_22034_epsilon_0, x = var_22033_cast_fp16)[name = string("op_22034_cast_fp16")]; + tensor hidden_states_743_cast_fp16 = mul(x = inputs_601_cast_fp16, y = var_22034_cast_fp16)[name = string("hidden_states_743_cast_fp16")]; + tensor input_623_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_743_cast_fp16)[name = string("input_623_cast_fp16")]; + string input_625_pad_type_0 = const()[name = string("input_625_pad_type_0"), val = string("valid")]; + tensor input_625_strides_0 = const()[name = string("input_625_strides_0"), val = tensor([1, 1])]; + tensor input_625_pad_0 = const()[name = string("input_625_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_625_dilations_0 = const()[name = string("input_625_dilations_0"), val = tensor([1, 1])]; + int32 input_625_groups_0 = const()[name = string("input_625_groups_0"), val = int32(1)]; + tensor input_625_cast_fp16 = conv(dilations = input_625_dilations_0, groups = input_625_groups_0, pad = input_625_pad_0, pad_type = input_625_pad_type_0, strides = input_625_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_623_cast_fp16)[name = string("input_625_cast_fp16")]; + tensor var_22048_cast_fp16 = silu(x = input_625_cast_fp16)[name = string("op_22048_cast_fp16")]; + string var_22054_pad_type_0 = const()[name = string("op_22054_pad_type_0"), val = string("valid")]; + tensor var_22054_strides_0 = const()[name = string("op_22054_strides_0"), val = tensor([1, 1])]; + tensor var_22054_pad_0 = const()[name = string("op_22054_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_22054_dilations_0 = const()[name = string("op_22054_dilations_0"), val = tensor([1, 1])]; + int32 var_22054_groups_0 = const()[name = string("op_22054_groups_0"), val = int32(1)]; + tensor var_22054_cast_fp16 = conv(dilations = var_22054_dilations_0, groups = var_22054_groups_0, pad = var_22054_pad_0, pad_type = var_22054_pad_type_0, strides = var_22054_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_623_cast_fp16)[name = string("op_22054_cast_fp16")]; + tensor input_627_cast_fp16 = mul(x = var_22048_cast_fp16, y = var_22054_cast_fp16)[name = string("input_627_cast_fp16")]; + string hidden_states_745_pad_type_0 = const()[name = string("hidden_states_745_pad_type_0"), val = string("valid")]; + tensor hidden_states_745_strides_0 = const()[name = string("hidden_states_745_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_745_pad_0 = const()[name = string("hidden_states_745_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_745_dilations_0 = const()[name = string("hidden_states_745_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_745_groups_0 = const()[name = string("hidden_states_745_groups_0"), val = int32(1)]; + tensor hidden_states_745_cast_fp16 = conv(dilations = hidden_states_745_dilations_0, groups = hidden_states_745_groups_0, pad = hidden_states_745_pad_0, pad_type = hidden_states_745_pad_type_0, strides = hidden_states_745_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_627_cast_fp16)[name = string("hidden_states_745_cast_fp16")]; + tensor inputs_603_cast_fp16 = add(x = inputs_601_cast_fp16, y = hidden_states_745_cast_fp16)[name = string("inputs_603_cast_fp16")]; + tensor obj_639_begin_0 = const()[name = string("obj_639_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_639_end_0 = const()[name = string("obj_639_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_639_end_mask_0 = const()[name = string("obj_639_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_639_cast_fp16 = slice_by_index(begin = obj_639_begin_0, end = obj_639_end_0, end_mask = obj_639_end_mask_0, x = key_caches_29_cast_fp16)[name = string("obj_639_cast_fp16")]; + tensor obj_641_begin_0 = const()[name = string("obj_641_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_641_end_0 = const()[name = string("obj_641_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_641_end_mask_0 = const()[name = string("obj_641_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_641_cast_fp16 = slice_by_index(begin = obj_641_begin_0, end = obj_641_end_0, end_mask = obj_641_end_mask_0, x = value_caches_29_cast_fp16)[name = string("obj_641_cast_fp16")]; + int32 var_22102 = const()[name = string("op_22102"), val = int32(3)]; + int32 var_22112 = const()[name = string("op_22112"), val = int32(-2)]; + tensor inputs_sq_603_cast_fp16 = mul(x = inputs_603_cast_fp16, y = inputs_603_cast_fp16)[name = string("inputs_sq_603_cast_fp16")]; + tensor variance_603_axes_0 = const()[name = string("variance_603_axes_0"), val = tensor([1])]; + bool variance_603_keep_dims_0 = const()[name = string("variance_603_keep_dims_0"), val = bool(true)]; + tensor variance_603_cast_fp16 = reduce_mean(axes = variance_603_axes_0, keep_dims = variance_603_keep_dims_0, x = inputs_sq_603_cast_fp16)[name = string("variance_603_cast_fp16")]; + fp16 var_22126_to_fp16 = const()[name = string("op_22126_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22127_cast_fp16 = add(x = variance_603_cast_fp16, y = var_22126_to_fp16)[name = string("op_22127_cast_fp16")]; + fp32 var_22128_epsilon_0 = const()[name = string("op_22128_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22128_cast_fp16 = rsqrt(epsilon = var_22128_epsilon_0, x = var_22127_cast_fp16)[name = string("op_22128_cast_fp16")]; + tensor hidden_states_747_cast_fp16 = mul(x = inputs_603_cast_fp16, y = var_22128_cast_fp16)[name = string("hidden_states_747_cast_fp16")]; + tensor obj_637_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_747_cast_fp16)[name = string("obj_637_cast_fp16")]; + string query_433_pad_type_0 = const()[name = string("query_433_pad_type_0"), val = string("valid")]; + tensor query_433_strides_0 = const()[name = string("query_433_strides_0"), val = tensor([1, 1])]; + tensor query_433_pad_0 = const()[name = string("query_433_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_433_dilations_0 = const()[name = string("query_433_dilations_0"), val = tensor([1, 1])]; + int32 query_433_groups_0 = const()[name = string("query_433_groups_0"), val = int32(1)]; + tensor query_433_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_433_dilations_0, groups = query_433_groups_0, pad = query_433_pad_0, pad_type = query_433_pad_type_0, strides = query_433_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_637_cast_fp16)[name = string("query_433_cast_fp16")]; + string current_key_289_pad_type_0 = const()[name = string("current_key_289_pad_type_0"), val = string("valid")]; + tensor current_key_289_strides_0 = const()[name = string("current_key_289_strides_0"), val = tensor([1, 1])]; + tensor current_key_289_pad_0 = const()[name = string("current_key_289_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_289_dilations_0 = const()[name = string("current_key_289_dilations_0"), val = tensor([1, 1])]; + int32 current_key_289_groups_0 = const()[name = string("current_key_289_groups_0"), val = int32(1)]; + tensor current_key_289_cast_fp16 = conv(dilations = current_key_289_dilations_0, groups = current_key_289_groups_0, pad = current_key_289_pad_0, pad_type = current_key_289_pad_type_0, strides = current_key_289_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_637_cast_fp16)[name = string("current_key_289_cast_fp16")]; + string current_value_145_pad_type_0 = const()[name = string("current_value_145_pad_type_0"), val = string("valid")]; + tensor current_value_145_strides_0 = const()[name = string("current_value_145_strides_0"), val = tensor([1, 1])]; + tensor current_value_145_pad_0 = const()[name = string("current_value_145_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_145_dilations_0 = const()[name = string("current_value_145_dilations_0"), val = tensor([1, 1])]; + int32 current_value_145_groups_0 = const()[name = string("current_value_145_groups_0"), val = int32(1)]; + tensor current_value_145_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_145_dilations_0, groups = current_value_145_groups_0, pad = current_value_145_pad_0, pad_type = current_value_145_pad_type_0, strides = current_value_145_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_637_cast_fp16)[name = string("current_value_145_cast_fp16")]; + tensor var_22165 = const()[name = string("op_22165"), val = tensor([16, 128, 1, 1])]; + tensor inputs_605_cast_fp16 = reshape(shape = var_22165, x = query_433_cast_fp16)[name = string("inputs_605_cast_fp16")]; + tensor inputs_sq_605_cast_fp16 = mul(x = inputs_605_cast_fp16, y = inputs_605_cast_fp16)[name = string("inputs_sq_605_cast_fp16")]; + tensor variance_605_axes_0 = const()[name = string("variance_605_axes_0"), val = tensor([1])]; + bool variance_605_keep_dims_0 = const()[name = string("variance_605_keep_dims_0"), val = bool(true)]; + tensor variance_605_cast_fp16 = reduce_mean(axes = variance_605_axes_0, keep_dims = variance_605_keep_dims_0, x = inputs_sq_605_cast_fp16)[name = string("variance_605_cast_fp16")]; + fp16 var_22171_to_fp16 = const()[name = string("op_22171_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22172_cast_fp16 = add(x = variance_605_cast_fp16, y = var_22171_to_fp16)[name = string("op_22172_cast_fp16")]; + fp32 var_22173_epsilon_0 = const()[name = string("op_22173_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22173_cast_fp16 = rsqrt(epsilon = var_22173_epsilon_0, x = var_22172_cast_fp16)[name = string("op_22173_cast_fp16")]; + tensor hidden_states_749_cast_fp16 = mul(x = inputs_605_cast_fp16, y = var_22173_cast_fp16)[name = string("hidden_states_749_cast_fp16")]; + tensor query_normed_145_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_749_cast_fp16)[name = string("query_normed_145_cast_fp16")]; + tensor var_22181 = const()[name = string("op_22181"), val = tensor([8, 128, 1, 1])]; + tensor inputs_607_cast_fp16 = reshape(shape = var_22181, x = current_key_289_cast_fp16)[name = string("inputs_607_cast_fp16")]; + tensor inputs_sq_607_cast_fp16 = mul(x = inputs_607_cast_fp16, y = inputs_607_cast_fp16)[name = string("inputs_sq_607_cast_fp16")]; + tensor variance_607_axes_0 = const()[name = string("variance_607_axes_0"), val = tensor([1])]; + bool variance_607_keep_dims_0 = const()[name = string("variance_607_keep_dims_0"), val = bool(true)]; + tensor variance_607_cast_fp16 = reduce_mean(axes = variance_607_axes_0, keep_dims = variance_607_keep_dims_0, x = inputs_sq_607_cast_fp16)[name = string("variance_607_cast_fp16")]; + fp16 var_22187_to_fp16 = const()[name = string("op_22187_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22188_cast_fp16 = add(x = variance_607_cast_fp16, y = var_22187_to_fp16)[name = string("op_22188_cast_fp16")]; + fp32 var_22189_epsilon_0 = const()[name = string("op_22189_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22189_cast_fp16 = rsqrt(epsilon = var_22189_epsilon_0, x = var_22188_cast_fp16)[name = string("op_22189_cast_fp16")]; + tensor hidden_states_751_cast_fp16 = mul(x = inputs_607_cast_fp16, y = var_22189_cast_fp16)[name = string("hidden_states_751_cast_fp16")]; + tensor current_key_normed_145_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_751_cast_fp16)[name = string("current_key_normed_145_cast_fp16")]; + tensor var_22207 = const()[name = string("op_22207"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_577_cast_fp16 = reshape(shape = var_22207, x = query_normed_145_cast_fp16)[name = string("mh_q_577_cast_fp16")]; + tensor var_22209 = const()[name = string("op_22209"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_577_cast_fp16 = reshape(shape = var_22209, x = current_key_normed_145_cast_fp16)[name = string("mh_k_577_cast_fp16")]; + tensor var_22213_cast_fp16 = mul(x = mh_q_577_cast_fp16, y = cos_141_to_fp16)[name = string("op_22213_cast_fp16")]; + tensor var_22218_begin_0 = const()[name = string("op_22218_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_22218_end_0 = const()[name = string("op_22218_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_22218_end_mask_0 = const()[name = string("op_22218_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_22218_cast_fp16 = slice_by_index(begin = var_22218_begin_0, end = var_22218_end_0, end_mask = var_22218_end_mask_0, x = mh_q_577_cast_fp16)[name = string("op_22218_cast_fp16")]; + tensor var_22224_begin_0 = const()[name = string("op_22224_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_22224_end_0 = const()[name = string("op_22224_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_22224_end_mask_0 = const()[name = string("op_22224_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_22224_cast_fp16 = slice_by_index(begin = var_22224_begin_0, end = var_22224_end_0, end_mask = var_22224_end_mask_0, x = mh_q_577_cast_fp16)[name = string("op_22224_cast_fp16")]; + fp16 const_1468_promoted_to_fp16 = const()[name = string("const_1468_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_22226_cast_fp16 = mul(x = var_22224_cast_fp16, y = const_1468_promoted_to_fp16)[name = string("op_22226_cast_fp16")]; + bool var_22228_interleave_0 = const()[name = string("op_22228_interleave_0"), val = bool(false)]; + tensor var_22228_cast_fp16 = concat(axis = var_22112, interleave = var_22228_interleave_0, values = (var_22226_cast_fp16, var_22218_cast_fp16))[name = string("op_22228_cast_fp16")]; + tensor var_22229_cast_fp16 = mul(x = var_22228_cast_fp16, y = sin_141_to_fp16)[name = string("op_22229_cast_fp16")]; + tensor mh_q_579_cast_fp16 = add(x = var_22213_cast_fp16, y = var_22229_cast_fp16)[name = string("mh_q_579_cast_fp16")]; + tensor var_22231_cast_fp16 = mul(x = mh_k_577_cast_fp16, y = cos_141_to_fp16)[name = string("op_22231_cast_fp16")]; + tensor var_22236_begin_0 = const()[name = string("op_22236_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_22236_end_0 = const()[name = string("op_22236_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_22236_end_mask_0 = const()[name = string("op_22236_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_22236_cast_fp16 = slice_by_index(begin = var_22236_begin_0, end = var_22236_end_0, end_mask = var_22236_end_mask_0, x = mh_k_577_cast_fp16)[name = string("op_22236_cast_fp16")]; + tensor var_22242_begin_0 = const()[name = string("op_22242_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_22242_end_0 = const()[name = string("op_22242_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_22242_end_mask_0 = const()[name = string("op_22242_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_22242_cast_fp16 = slice_by_index(begin = var_22242_begin_0, end = var_22242_end_0, end_mask = var_22242_end_mask_0, x = mh_k_577_cast_fp16)[name = string("op_22242_cast_fp16")]; + fp16 const_1471_promoted_to_fp16 = const()[name = string("const_1471_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_22244_cast_fp16 = mul(x = var_22242_cast_fp16, y = const_1471_promoted_to_fp16)[name = string("op_22244_cast_fp16")]; + bool var_22246_interleave_0 = const()[name = string("op_22246_interleave_0"), val = bool(false)]; + tensor var_22246_cast_fp16 = concat(axis = var_22112, interleave = var_22246_interleave_0, values = (var_22244_cast_fp16, var_22236_cast_fp16))[name = string("op_22246_cast_fp16")]; + tensor var_22247_cast_fp16 = mul(x = var_22246_cast_fp16, y = sin_141_to_fp16)[name = string("op_22247_cast_fp16")]; + tensor mh_k_579_cast_fp16 = add(x = var_22231_cast_fp16, y = var_22247_cast_fp16)[name = string("mh_k_579_cast_fp16")]; + tensor var_22251 = const()[name = string("op_22251"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_291_cast_fp16 = reshape(shape = var_22251, x = mh_k_579_cast_fp16)[name = string("current_key_291_cast_fp16")]; + tensor var_22257_to_fp16 = const()[name = string("op_22257_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203520)))]; + tensor var_22258_cast_fp16 = mul(x = obj_639_cast_fp16, y = var_22257_to_fp16)[name = string("op_22258_cast_fp16")]; + tensor var_22255_to_fp16 = const()[name = string("op_22255_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203648)))]; + tensor var_22259_cast_fp16 = mul(x = current_key_291_cast_fp16, y = var_22255_to_fp16)[name = string("op_22259_cast_fp16")]; + tensor key_291_cast_fp16 = add(x = var_22258_cast_fp16, y = var_22259_cast_fp16)[name = string("key_291_cast_fp16")]; + tensor var_22261_to_fp16 = const()[name = string("op_22261_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203520)))]; + tensor var_22262_cast_fp16 = mul(x = obj_641_cast_fp16, y = var_22261_to_fp16)[name = string("op_22262_cast_fp16")]; + tensor var_22263_cast_fp16 = mul(x = current_value_145_cast_fp16, y = var_22255_to_fp16)[name = string("op_22263_cast_fp16")]; + tensor value_145_cast_fp16 = add(x = var_22262_cast_fp16, y = var_22263_cast_fp16)[name = string("value_145_cast_fp16")]; + fp16 var_22270_to_fp16 = const()[name = string("op_22270_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_583_cast_fp16 = mul(x = mh_q_579_cast_fp16, y = var_22270_to_fp16)[name = string("mh_q_583_cast_fp16")]; + tensor var_22272 = const()[name = string("op_22272"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_581_cast_fp16 = reshape(shape = var_22272, x = key_291_cast_fp16)[name = string("mh_k_581_cast_fp16")]; + tensor var_22274 = const()[name = string("op_22274"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_289_cast_fp16 = reshape(shape = var_22274, x = value_145_cast_fp16)[name = string("mh_v_289_cast_fp16")]; + tensor transpose_288_perm_0 = const()[name = string("transpose_288_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_144_reps_0 = const()[name = string("tile_144_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_288_cast_fp16 = transpose(perm = transpose_288_perm_0, x = mh_k_581_cast_fp16)[name = string("transpose_47")]; + tensor tile_144_cast_fp16 = tile(reps = tile_144_reps_0, x = transpose_288_cast_fp16)[name = string("tile_144_cast_fp16")]; + tensor concat_361 = const()[name = string("concat_361"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_288_cast_fp16 = reshape(shape = concat_361, x = tile_144_cast_fp16)[name = string("reshape_288_cast_fp16")]; + tensor transpose_289_perm_0 = const()[name = string("transpose_289_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_362 = const()[name = string("concat_362"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_289_cast_fp16 = transpose(perm = transpose_289_perm_0, x = reshape_288_cast_fp16)[name = string("transpose_46")]; + tensor reshape_289_cast_fp16 = reshape(shape = concat_362, x = transpose_289_cast_fp16)[name = string("reshape_289_cast_fp16")]; + tensor transpose_290_perm_0 = const()[name = string("transpose_290_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_145_reps_0 = const()[name = string("tile_145_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_290_cast_fp16 = transpose(perm = transpose_290_perm_0, x = mh_v_289_cast_fp16)[name = string("transpose_45")]; + tensor tile_145_cast_fp16 = tile(reps = tile_145_reps_0, x = transpose_290_cast_fp16)[name = string("tile_145_cast_fp16")]; + tensor concat_363 = const()[name = string("concat_363"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_290_cast_fp16 = reshape(shape = concat_363, x = tile_145_cast_fp16)[name = string("reshape_290_cast_fp16")]; + tensor transpose_291_perm_0 = const()[name = string("transpose_291_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_364 = const()[name = string("concat_364"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_291_cast_fp16 = transpose(perm = transpose_291_perm_0, x = reshape_290_cast_fp16)[name = string("transpose_44")]; + tensor reshape_291_cast_fp16 = reshape(shape = concat_364, x = transpose_291_cast_fp16)[name = string("reshape_291_cast_fp16")]; + tensor transpose_605_perm_0 = const()[name = string("transpose_605_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_433_transpose_x_1 = const()[name = string("mh_w_433_transpose_x_1"), val = bool(true)]; + bool mh_w_433_transpose_y_1 = const()[name = string("mh_w_433_transpose_y_1"), val = bool(false)]; + tensor transpose_605_cast_fp16 = transpose(perm = transpose_605_perm_0, x = reshape_289_cast_fp16)[name = string("transpose_43")]; + tensor mh_w_433_cast_fp16 = matmul(transpose_x = mh_w_433_transpose_x_1, transpose_y = mh_w_433_transpose_y_1, x = mh_q_583_cast_fp16, y = transpose_605_cast_fp16)[name = string("mh_w_433_cast_fp16")]; + tensor var_22282_to_fp16 = const()[name = string("op_22282_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203776)))]; + tensor mh_w_435_cast_fp16 = add(x = mh_w_433_cast_fp16, y = var_22282_to_fp16)[name = string("mh_w_435_cast_fp16")]; + tensor mh_w_437_cast_fp16 = softmax(axis = var_22102, x = mh_w_435_cast_fp16)[name = string("mh_w_437_cast_fp16")]; + tensor transpose_606_perm_0 = const()[name = string("transpose_606_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_145_transpose_x_1 = const()[name = string("attn_145_transpose_x_1"), val = bool(false)]; + bool attn_145_transpose_y_1 = const()[name = string("attn_145_transpose_y_1"), val = bool(true)]; + tensor transpose_606_cast_fp16 = transpose(perm = transpose_606_perm_0, x = reshape_291_cast_fp16)[name = string("transpose_42")]; + tensor attn_145_cast_fp16 = matmul(transpose_x = attn_145_transpose_x_1, transpose_y = attn_145_transpose_y_1, x = transpose_606_cast_fp16, y = mh_w_437_cast_fp16)[name = string("attn_145_cast_fp16")]; + tensor var_22288 = const()[name = string("op_22288"), val = tensor([1, 2048, 1, 1])]; + tensor input_629_cast_fp16 = reshape(shape = var_22288, x = attn_145_cast_fp16)[name = string("input_629_cast_fp16")]; + string obj_643_pad_type_0 = const()[name = string("obj_643_pad_type_0"), val = string("valid")]; + tensor obj_643_strides_0 = const()[name = string("obj_643_strides_0"), val = tensor([1, 1])]; + tensor obj_643_pad_0 = const()[name = string("obj_643_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_643_dilations_0 = const()[name = string("obj_643_dilations_0"), val = tensor([1, 1])]; + int32 obj_643_groups_0 = const()[name = string("obj_643_groups_0"), val = int32(1)]; + tensor obj_643_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_643_dilations_0, groups = obj_643_groups_0, pad = obj_643_pad_0, pad_type = obj_643_pad_type_0, strides = obj_643_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_629_cast_fp16)[name = string("obj_643_cast_fp16")]; + tensor inputs_609_cast_fp16 = add(x = inputs_603_cast_fp16, y = obj_643_cast_fp16)[name = string("inputs_609_cast_fp16")]; + tensor inputs_sq_609_cast_fp16 = mul(x = inputs_609_cast_fp16, y = inputs_609_cast_fp16)[name = string("inputs_sq_609_cast_fp16")]; + tensor variance_609_axes_0 = const()[name = string("variance_609_axes_0"), val = tensor([1])]; + bool variance_609_keep_dims_0 = const()[name = string("variance_609_keep_dims_0"), val = bool(true)]; + tensor variance_609_cast_fp16 = reduce_mean(axes = variance_609_axes_0, keep_dims = variance_609_keep_dims_0, x = inputs_sq_609_cast_fp16)[name = string("variance_609_cast_fp16")]; + fp16 var_22306_to_fp16 = const()[name = string("op_22306_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22307_cast_fp16 = add(x = variance_609_cast_fp16, y = var_22306_to_fp16)[name = string("op_22307_cast_fp16")]; + fp32 var_22308_epsilon_0 = const()[name = string("op_22308_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22308_cast_fp16 = rsqrt(epsilon = var_22308_epsilon_0, x = var_22307_cast_fp16)[name = string("op_22308_cast_fp16")]; + tensor hidden_states_753_cast_fp16 = mul(x = inputs_609_cast_fp16, y = var_22308_cast_fp16)[name = string("hidden_states_753_cast_fp16")]; + tensor input_631_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_753_cast_fp16)[name = string("input_631_cast_fp16")]; + string input_633_pad_type_0 = const()[name = string("input_633_pad_type_0"), val = string("valid")]; + tensor input_633_strides_0 = const()[name = string("input_633_strides_0"), val = tensor([1, 1])]; + tensor input_633_pad_0 = const()[name = string("input_633_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_633_dilations_0 = const()[name = string("input_633_dilations_0"), val = tensor([1, 1])]; + int32 input_633_groups_0 = const()[name = string("input_633_groups_0"), val = int32(1)]; + tensor input_633_cast_fp16 = conv(dilations = input_633_dilations_0, groups = input_633_groups_0, pad = input_633_pad_0, pad_type = input_633_pad_type_0, strides = input_633_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_631_cast_fp16)[name = string("input_633_cast_fp16")]; + tensor var_22322_cast_fp16 = silu(x = input_633_cast_fp16)[name = string("op_22322_cast_fp16")]; + string var_22328_pad_type_0 = const()[name = string("op_22328_pad_type_0"), val = string("valid")]; + tensor var_22328_strides_0 = const()[name = string("op_22328_strides_0"), val = tensor([1, 1])]; + tensor var_22328_pad_0 = const()[name = string("op_22328_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_22328_dilations_0 = const()[name = string("op_22328_dilations_0"), val = tensor([1, 1])]; + int32 var_22328_groups_0 = const()[name = string("op_22328_groups_0"), val = int32(1)]; + tensor var_22328_cast_fp16 = conv(dilations = var_22328_dilations_0, groups = var_22328_groups_0, pad = var_22328_pad_0, pad_type = var_22328_pad_type_0, strides = var_22328_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_631_cast_fp16)[name = string("op_22328_cast_fp16")]; + tensor input_635_cast_fp16 = mul(x = var_22322_cast_fp16, y = var_22328_cast_fp16)[name = string("input_635_cast_fp16")]; + string hidden_states_755_pad_type_0 = const()[name = string("hidden_states_755_pad_type_0"), val = string("valid")]; + tensor hidden_states_755_strides_0 = const()[name = string("hidden_states_755_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_755_pad_0 = const()[name = string("hidden_states_755_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_755_dilations_0 = const()[name = string("hidden_states_755_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_755_groups_0 = const()[name = string("hidden_states_755_groups_0"), val = int32(1)]; + tensor hidden_states_755_cast_fp16 = conv(dilations = hidden_states_755_dilations_0, groups = hidden_states_755_groups_0, pad = hidden_states_755_pad_0, pad_type = hidden_states_755_pad_type_0, strides = hidden_states_755_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_635_cast_fp16)[name = string("hidden_states_755_cast_fp16")]; + tensor inputs_611_cast_fp16 = add(x = inputs_609_cast_fp16, y = hidden_states_755_cast_fp16)[name = string("inputs_611_cast_fp16")]; + tensor obj_647_begin_0 = const()[name = string("obj_647_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_647_end_0 = const()[name = string("obj_647_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_647_end_mask_0 = const()[name = string("obj_647_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_647_cast_fp16 = slice_by_index(begin = obj_647_begin_0, end = obj_647_end_0, end_mask = obj_647_end_mask_0, x = key_caches_29_cast_fp16)[name = string("obj_647_cast_fp16")]; + tensor obj_649_begin_0 = const()[name = string("obj_649_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_649_end_0 = const()[name = string("obj_649_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_649_end_mask_0 = const()[name = string("obj_649_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_649_cast_fp16 = slice_by_index(begin = obj_649_begin_0, end = obj_649_end_0, end_mask = obj_649_end_mask_0, x = value_caches_29_cast_fp16)[name = string("obj_649_cast_fp16")]; + int32 var_22376 = const()[name = string("op_22376"), val = int32(3)]; + int32 var_22386 = const()[name = string("op_22386"), val = int32(-2)]; + tensor inputs_sq_611_cast_fp16 = mul(x = inputs_611_cast_fp16, y = inputs_611_cast_fp16)[name = string("inputs_sq_611_cast_fp16")]; + tensor variance_611_axes_0 = const()[name = string("variance_611_axes_0"), val = tensor([1])]; + bool variance_611_keep_dims_0 = const()[name = string("variance_611_keep_dims_0"), val = bool(true)]; + tensor variance_611_cast_fp16 = reduce_mean(axes = variance_611_axes_0, keep_dims = variance_611_keep_dims_0, x = inputs_sq_611_cast_fp16)[name = string("variance_611_cast_fp16")]; + fp16 var_22400_to_fp16 = const()[name = string("op_22400_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22401_cast_fp16 = add(x = variance_611_cast_fp16, y = var_22400_to_fp16)[name = string("op_22401_cast_fp16")]; + fp32 var_22402_epsilon_0 = const()[name = string("op_22402_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22402_cast_fp16 = rsqrt(epsilon = var_22402_epsilon_0, x = var_22401_cast_fp16)[name = string("op_22402_cast_fp16")]; + tensor hidden_states_757_cast_fp16 = mul(x = inputs_611_cast_fp16, y = var_22402_cast_fp16)[name = string("hidden_states_757_cast_fp16")]; + tensor obj_645_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_757_cast_fp16)[name = string("obj_645_cast_fp16")]; + string query_439_pad_type_0 = const()[name = string("query_439_pad_type_0"), val = string("valid")]; + tensor query_439_strides_0 = const()[name = string("query_439_strides_0"), val = tensor([1, 1])]; + tensor query_439_pad_0 = const()[name = string("query_439_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_439_dilations_0 = const()[name = string("query_439_dilations_0"), val = tensor([1, 1])]; + int32 query_439_groups_0 = const()[name = string("query_439_groups_0"), val = int32(1)]; + tensor query_439_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_439_dilations_0, groups = query_439_groups_0, pad = query_439_pad_0, pad_type = query_439_pad_type_0, strides = query_439_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_645_cast_fp16)[name = string("query_439_cast_fp16")]; + string current_key_293_pad_type_0 = const()[name = string("current_key_293_pad_type_0"), val = string("valid")]; + tensor current_key_293_strides_0 = const()[name = string("current_key_293_strides_0"), val = tensor([1, 1])]; + tensor current_key_293_pad_0 = const()[name = string("current_key_293_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_293_dilations_0 = const()[name = string("current_key_293_dilations_0"), val = tensor([1, 1])]; + int32 current_key_293_groups_0 = const()[name = string("current_key_293_groups_0"), val = int32(1)]; + tensor current_key_293_cast_fp16 = conv(dilations = current_key_293_dilations_0, groups = current_key_293_groups_0, pad = current_key_293_pad_0, pad_type = current_key_293_pad_type_0, strides = current_key_293_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_645_cast_fp16)[name = string("current_key_293_cast_fp16")]; + string current_value_147_pad_type_0 = const()[name = string("current_value_147_pad_type_0"), val = string("valid")]; + tensor current_value_147_strides_0 = const()[name = string("current_value_147_strides_0"), val = tensor([1, 1])]; + tensor current_value_147_pad_0 = const()[name = string("current_value_147_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_147_dilations_0 = const()[name = string("current_value_147_dilations_0"), val = tensor([1, 1])]; + int32 current_value_147_groups_0 = const()[name = string("current_value_147_groups_0"), val = int32(1)]; + tensor current_value_147_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_147_dilations_0, groups = current_value_147_groups_0, pad = current_value_147_pad_0, pad_type = current_value_147_pad_type_0, strides = current_value_147_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_645_cast_fp16)[name = string("current_value_147_cast_fp16")]; + tensor var_22439 = const()[name = string("op_22439"), val = tensor([16, 128, 1, 1])]; + tensor inputs_613_cast_fp16 = reshape(shape = var_22439, x = query_439_cast_fp16)[name = string("inputs_613_cast_fp16")]; + tensor inputs_sq_613_cast_fp16 = mul(x = inputs_613_cast_fp16, y = inputs_613_cast_fp16)[name = string("inputs_sq_613_cast_fp16")]; + tensor variance_613_axes_0 = const()[name = string("variance_613_axes_0"), val = tensor([1])]; + bool variance_613_keep_dims_0 = const()[name = string("variance_613_keep_dims_0"), val = bool(true)]; + tensor variance_613_cast_fp16 = reduce_mean(axes = variance_613_axes_0, keep_dims = variance_613_keep_dims_0, x = inputs_sq_613_cast_fp16)[name = string("variance_613_cast_fp16")]; + fp16 var_22445_to_fp16 = const()[name = string("op_22445_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22446_cast_fp16 = add(x = variance_613_cast_fp16, y = var_22445_to_fp16)[name = string("op_22446_cast_fp16")]; + fp32 var_22447_epsilon_0 = const()[name = string("op_22447_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22447_cast_fp16 = rsqrt(epsilon = var_22447_epsilon_0, x = var_22446_cast_fp16)[name = string("op_22447_cast_fp16")]; + tensor hidden_states_759_cast_fp16 = mul(x = inputs_613_cast_fp16, y = var_22447_cast_fp16)[name = string("hidden_states_759_cast_fp16")]; + tensor query_normed_147_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_759_cast_fp16)[name = string("query_normed_147_cast_fp16")]; + tensor var_22455 = const()[name = string("op_22455"), val = tensor([8, 128, 1, 1])]; + tensor inputs_615_cast_fp16 = reshape(shape = var_22455, x = current_key_293_cast_fp16)[name = string("inputs_615_cast_fp16")]; + tensor inputs_sq_615_cast_fp16 = mul(x = inputs_615_cast_fp16, y = inputs_615_cast_fp16)[name = string("inputs_sq_615_cast_fp16")]; + tensor variance_615_axes_0 = const()[name = string("variance_615_axes_0"), val = tensor([1])]; + bool variance_615_keep_dims_0 = const()[name = string("variance_615_keep_dims_0"), val = bool(true)]; + tensor variance_615_cast_fp16 = reduce_mean(axes = variance_615_axes_0, keep_dims = variance_615_keep_dims_0, x = inputs_sq_615_cast_fp16)[name = string("variance_615_cast_fp16")]; + fp16 var_22461_to_fp16 = const()[name = string("op_22461_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22462_cast_fp16 = add(x = variance_615_cast_fp16, y = var_22461_to_fp16)[name = string("op_22462_cast_fp16")]; + fp32 var_22463_epsilon_0 = const()[name = string("op_22463_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22463_cast_fp16 = rsqrt(epsilon = var_22463_epsilon_0, x = var_22462_cast_fp16)[name = string("op_22463_cast_fp16")]; + tensor hidden_states_761_cast_fp16 = mul(x = inputs_615_cast_fp16, y = var_22463_cast_fp16)[name = string("hidden_states_761_cast_fp16")]; + tensor current_key_normed_147_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_761_cast_fp16)[name = string("current_key_normed_147_cast_fp16")]; + tensor var_22481 = const()[name = string("op_22481"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_585_cast_fp16 = reshape(shape = var_22481, x = query_normed_147_cast_fp16)[name = string("mh_q_585_cast_fp16")]; + tensor var_22483 = const()[name = string("op_22483"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_585_cast_fp16 = reshape(shape = var_22483, x = current_key_normed_147_cast_fp16)[name = string("mh_k_585_cast_fp16")]; + tensor var_22487_cast_fp16 = mul(x = mh_q_585_cast_fp16, y = cos_141_to_fp16)[name = string("op_22487_cast_fp16")]; + tensor var_22492_begin_0 = const()[name = string("op_22492_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_22492_end_0 = const()[name = string("op_22492_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_22492_end_mask_0 = const()[name = string("op_22492_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_22492_cast_fp16 = slice_by_index(begin = var_22492_begin_0, end = var_22492_end_0, end_mask = var_22492_end_mask_0, x = mh_q_585_cast_fp16)[name = string("op_22492_cast_fp16")]; + tensor var_22498_begin_0 = const()[name = string("op_22498_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_22498_end_0 = const()[name = string("op_22498_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_22498_end_mask_0 = const()[name = string("op_22498_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_22498_cast_fp16 = slice_by_index(begin = var_22498_begin_0, end = var_22498_end_0, end_mask = var_22498_end_mask_0, x = mh_q_585_cast_fp16)[name = string("op_22498_cast_fp16")]; + fp16 const_1488_promoted_to_fp16 = const()[name = string("const_1488_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_22500_cast_fp16 = mul(x = var_22498_cast_fp16, y = const_1488_promoted_to_fp16)[name = string("op_22500_cast_fp16")]; + bool var_22502_interleave_0 = const()[name = string("op_22502_interleave_0"), val = bool(false)]; + tensor var_22502_cast_fp16 = concat(axis = var_22386, interleave = var_22502_interleave_0, values = (var_22500_cast_fp16, var_22492_cast_fp16))[name = string("op_22502_cast_fp16")]; + tensor var_22503_cast_fp16 = mul(x = var_22502_cast_fp16, y = sin_141_to_fp16)[name = string("op_22503_cast_fp16")]; + tensor mh_q_587_cast_fp16 = add(x = var_22487_cast_fp16, y = var_22503_cast_fp16)[name = string("mh_q_587_cast_fp16")]; + tensor var_22505_cast_fp16 = mul(x = mh_k_585_cast_fp16, y = cos_141_to_fp16)[name = string("op_22505_cast_fp16")]; + tensor var_22510_begin_0 = const()[name = string("op_22510_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_22510_end_0 = const()[name = string("op_22510_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_22510_end_mask_0 = const()[name = string("op_22510_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_22510_cast_fp16 = slice_by_index(begin = var_22510_begin_0, end = var_22510_end_0, end_mask = var_22510_end_mask_0, x = mh_k_585_cast_fp16)[name = string("op_22510_cast_fp16")]; + tensor var_22516_begin_0 = const()[name = string("op_22516_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_22516_end_0 = const()[name = string("op_22516_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_22516_end_mask_0 = const()[name = string("op_22516_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_22516_cast_fp16 = slice_by_index(begin = var_22516_begin_0, end = var_22516_end_0, end_mask = var_22516_end_mask_0, x = mh_k_585_cast_fp16)[name = string("op_22516_cast_fp16")]; + fp16 const_1491_promoted_to_fp16 = const()[name = string("const_1491_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_22518_cast_fp16 = mul(x = var_22516_cast_fp16, y = const_1491_promoted_to_fp16)[name = string("op_22518_cast_fp16")]; + bool var_22520_interleave_0 = const()[name = string("op_22520_interleave_0"), val = bool(false)]; + tensor var_22520_cast_fp16 = concat(axis = var_22386, interleave = var_22520_interleave_0, values = (var_22518_cast_fp16, var_22510_cast_fp16))[name = string("op_22520_cast_fp16")]; + tensor var_22521_cast_fp16 = mul(x = var_22520_cast_fp16, y = sin_141_to_fp16)[name = string("op_22521_cast_fp16")]; + tensor mh_k_587_cast_fp16 = add(x = var_22505_cast_fp16, y = var_22521_cast_fp16)[name = string("mh_k_587_cast_fp16")]; + tensor var_22525 = const()[name = string("op_22525"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_295_cast_fp16 = reshape(shape = var_22525, x = mh_k_587_cast_fp16)[name = string("current_key_295_cast_fp16")]; + tensor var_22531_to_fp16 = const()[name = string("op_22531_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203520)))]; + tensor var_22532_cast_fp16 = mul(x = obj_647_cast_fp16, y = var_22531_to_fp16)[name = string("op_22532_cast_fp16")]; + tensor var_22529_to_fp16 = const()[name = string("op_22529_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203648)))]; + tensor var_22533_cast_fp16 = mul(x = current_key_295_cast_fp16, y = var_22529_to_fp16)[name = string("op_22533_cast_fp16")]; + tensor key_295_cast_fp16 = add(x = var_22532_cast_fp16, y = var_22533_cast_fp16)[name = string("key_295_cast_fp16")]; + tensor var_22535_to_fp16 = const()[name = string("op_22535_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203520)))]; + tensor var_22536_cast_fp16 = mul(x = obj_649_cast_fp16, y = var_22535_to_fp16)[name = string("op_22536_cast_fp16")]; + tensor var_22537_cast_fp16 = mul(x = current_value_147_cast_fp16, y = var_22529_to_fp16)[name = string("op_22537_cast_fp16")]; + tensor value_147_cast_fp16 = add(x = var_22536_cast_fp16, y = var_22537_cast_fp16)[name = string("value_147_cast_fp16")]; + fp16 var_22544_to_fp16 = const()[name = string("op_22544_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_591_cast_fp16 = mul(x = mh_q_587_cast_fp16, y = var_22544_to_fp16)[name = string("mh_q_591_cast_fp16")]; + tensor var_22546 = const()[name = string("op_22546"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_589_cast_fp16 = reshape(shape = var_22546, x = key_295_cast_fp16)[name = string("mh_k_589_cast_fp16")]; + tensor var_22548 = const()[name = string("op_22548"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_293_cast_fp16 = reshape(shape = var_22548, x = value_147_cast_fp16)[name = string("mh_v_293_cast_fp16")]; + tensor transpose_292_perm_0 = const()[name = string("transpose_292_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_146_reps_0 = const()[name = string("tile_146_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_292_cast_fp16 = transpose(perm = transpose_292_perm_0, x = mh_k_589_cast_fp16)[name = string("transpose_41")]; + tensor tile_146_cast_fp16 = tile(reps = tile_146_reps_0, x = transpose_292_cast_fp16)[name = string("tile_146_cast_fp16")]; + tensor concat_365 = const()[name = string("concat_365"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_292_cast_fp16 = reshape(shape = concat_365, x = tile_146_cast_fp16)[name = string("reshape_292_cast_fp16")]; + tensor transpose_293_perm_0 = const()[name = string("transpose_293_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_366 = const()[name = string("concat_366"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_293_cast_fp16 = transpose(perm = transpose_293_perm_0, x = reshape_292_cast_fp16)[name = string("transpose_40")]; + tensor reshape_293_cast_fp16 = reshape(shape = concat_366, x = transpose_293_cast_fp16)[name = string("reshape_293_cast_fp16")]; + tensor transpose_294_perm_0 = const()[name = string("transpose_294_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_147_reps_0 = const()[name = string("tile_147_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_294_cast_fp16 = transpose(perm = transpose_294_perm_0, x = mh_v_293_cast_fp16)[name = string("transpose_39")]; + tensor tile_147_cast_fp16 = tile(reps = tile_147_reps_0, x = transpose_294_cast_fp16)[name = string("tile_147_cast_fp16")]; + tensor concat_367 = const()[name = string("concat_367"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_294_cast_fp16 = reshape(shape = concat_367, x = tile_147_cast_fp16)[name = string("reshape_294_cast_fp16")]; + tensor transpose_295_perm_0 = const()[name = string("transpose_295_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_368 = const()[name = string("concat_368"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_295_cast_fp16 = transpose(perm = transpose_295_perm_0, x = reshape_294_cast_fp16)[name = string("transpose_38")]; + tensor reshape_295_cast_fp16 = reshape(shape = concat_368, x = transpose_295_cast_fp16)[name = string("reshape_295_cast_fp16")]; + tensor transpose_609_perm_0 = const()[name = string("transpose_609_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_439_transpose_x_1 = const()[name = string("mh_w_439_transpose_x_1"), val = bool(true)]; + bool mh_w_439_transpose_y_1 = const()[name = string("mh_w_439_transpose_y_1"), val = bool(false)]; + tensor transpose_609_cast_fp16 = transpose(perm = transpose_609_perm_0, x = reshape_293_cast_fp16)[name = string("transpose_37")]; + tensor mh_w_439_cast_fp16 = matmul(transpose_x = mh_w_439_transpose_x_1, transpose_y = mh_w_439_transpose_y_1, x = mh_q_591_cast_fp16, y = transpose_609_cast_fp16)[name = string("mh_w_439_cast_fp16")]; + tensor var_22556_to_fp16 = const()[name = string("op_22556_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203776)))]; + tensor mh_w_441_cast_fp16 = add(x = mh_w_439_cast_fp16, y = var_22556_to_fp16)[name = string("mh_w_441_cast_fp16")]; + tensor mh_w_443_cast_fp16 = softmax(axis = var_22376, x = mh_w_441_cast_fp16)[name = string("mh_w_443_cast_fp16")]; + tensor transpose_610_perm_0 = const()[name = string("transpose_610_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_147_transpose_x_1 = const()[name = string("attn_147_transpose_x_1"), val = bool(false)]; + bool attn_147_transpose_y_1 = const()[name = string("attn_147_transpose_y_1"), val = bool(true)]; + tensor transpose_610_cast_fp16 = transpose(perm = transpose_610_perm_0, x = reshape_295_cast_fp16)[name = string("transpose_36")]; + tensor attn_147_cast_fp16 = matmul(transpose_x = attn_147_transpose_x_1, transpose_y = attn_147_transpose_y_1, x = transpose_610_cast_fp16, y = mh_w_443_cast_fp16)[name = string("attn_147_cast_fp16")]; + tensor var_22562 = const()[name = string("op_22562"), val = tensor([1, 2048, 1, 1])]; + tensor input_637_cast_fp16 = reshape(shape = var_22562, x = attn_147_cast_fp16)[name = string("input_637_cast_fp16")]; + string obj_651_pad_type_0 = const()[name = string("obj_651_pad_type_0"), val = string("valid")]; + tensor obj_651_strides_0 = const()[name = string("obj_651_strides_0"), val = tensor([1, 1])]; + tensor obj_651_pad_0 = const()[name = string("obj_651_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_651_dilations_0 = const()[name = string("obj_651_dilations_0"), val = tensor([1, 1])]; + int32 obj_651_groups_0 = const()[name = string("obj_651_groups_0"), val = int32(1)]; + tensor obj_651_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_651_dilations_0, groups = obj_651_groups_0, pad = obj_651_pad_0, pad_type = obj_651_pad_type_0, strides = obj_651_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_637_cast_fp16)[name = string("obj_651_cast_fp16")]; + tensor inputs_617_cast_fp16 = add(x = inputs_611_cast_fp16, y = obj_651_cast_fp16)[name = string("inputs_617_cast_fp16")]; + tensor inputs_sq_617_cast_fp16 = mul(x = inputs_617_cast_fp16, y = inputs_617_cast_fp16)[name = string("inputs_sq_617_cast_fp16")]; + tensor variance_617_axes_0 = const()[name = string("variance_617_axes_0"), val = tensor([1])]; + bool variance_617_keep_dims_0 = const()[name = string("variance_617_keep_dims_0"), val = bool(true)]; + tensor variance_617_cast_fp16 = reduce_mean(axes = variance_617_axes_0, keep_dims = variance_617_keep_dims_0, x = inputs_sq_617_cast_fp16)[name = string("variance_617_cast_fp16")]; + fp16 var_22580_to_fp16 = const()[name = string("op_22580_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22581_cast_fp16 = add(x = variance_617_cast_fp16, y = var_22580_to_fp16)[name = string("op_22581_cast_fp16")]; + fp32 var_22582_epsilon_0 = const()[name = string("op_22582_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22582_cast_fp16 = rsqrt(epsilon = var_22582_epsilon_0, x = var_22581_cast_fp16)[name = string("op_22582_cast_fp16")]; + tensor hidden_states_763_cast_fp16 = mul(x = inputs_617_cast_fp16, y = var_22582_cast_fp16)[name = string("hidden_states_763_cast_fp16")]; + tensor input_639_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_763_cast_fp16)[name = string("input_639_cast_fp16")]; + string input_641_pad_type_0 = const()[name = string("input_641_pad_type_0"), val = string("valid")]; + tensor input_641_strides_0 = const()[name = string("input_641_strides_0"), val = tensor([1, 1])]; + tensor input_641_pad_0 = const()[name = string("input_641_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_641_dilations_0 = const()[name = string("input_641_dilations_0"), val = tensor([1, 1])]; + int32 input_641_groups_0 = const()[name = string("input_641_groups_0"), val = int32(1)]; + tensor input_641_cast_fp16 = conv(dilations = input_641_dilations_0, groups = input_641_groups_0, pad = input_641_pad_0, pad_type = input_641_pad_type_0, strides = input_641_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_639_cast_fp16)[name = string("input_641_cast_fp16")]; + tensor var_22596_cast_fp16 = silu(x = input_641_cast_fp16)[name = string("op_22596_cast_fp16")]; + string var_22602_pad_type_0 = const()[name = string("op_22602_pad_type_0"), val = string("valid")]; + tensor var_22602_strides_0 = const()[name = string("op_22602_strides_0"), val = tensor([1, 1])]; + tensor var_22602_pad_0 = const()[name = string("op_22602_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_22602_dilations_0 = const()[name = string("op_22602_dilations_0"), val = tensor([1, 1])]; + int32 var_22602_groups_0 = const()[name = string("op_22602_groups_0"), val = int32(1)]; + tensor var_22602_cast_fp16 = conv(dilations = var_22602_dilations_0, groups = var_22602_groups_0, pad = var_22602_pad_0, pad_type = var_22602_pad_type_0, strides = var_22602_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_639_cast_fp16)[name = string("op_22602_cast_fp16")]; + tensor input_643_cast_fp16 = mul(x = var_22596_cast_fp16, y = var_22602_cast_fp16)[name = string("input_643_cast_fp16")]; + string hidden_states_765_pad_type_0 = const()[name = string("hidden_states_765_pad_type_0"), val = string("valid")]; + tensor hidden_states_765_strides_0 = const()[name = string("hidden_states_765_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_765_pad_0 = const()[name = string("hidden_states_765_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_765_dilations_0 = const()[name = string("hidden_states_765_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_765_groups_0 = const()[name = string("hidden_states_765_groups_0"), val = int32(1)]; + tensor hidden_states_765_cast_fp16 = conv(dilations = hidden_states_765_dilations_0, groups = hidden_states_765_groups_0, pad = hidden_states_765_pad_0, pad_type = hidden_states_765_pad_type_0, strides = hidden_states_765_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_643_cast_fp16)[name = string("hidden_states_765_cast_fp16")]; + tensor inputs_619_cast_fp16 = add(x = inputs_617_cast_fp16, y = hidden_states_765_cast_fp16)[name = string("inputs_619_cast_fp16")]; + tensor obj_655_begin_0 = const()[name = string("obj_655_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_655_end_0 = const()[name = string("obj_655_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_655_end_mask_0 = const()[name = string("obj_655_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_655_cast_fp16 = slice_by_index(begin = obj_655_begin_0, end = obj_655_end_0, end_mask = obj_655_end_mask_0, x = key_caches_29_cast_fp16)[name = string("obj_655_cast_fp16")]; + tensor obj_657_begin_0 = const()[name = string("obj_657_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_657_end_0 = const()[name = string("obj_657_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_657_end_mask_0 = const()[name = string("obj_657_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_657_cast_fp16 = slice_by_index(begin = obj_657_begin_0, end = obj_657_end_0, end_mask = obj_657_end_mask_0, x = value_caches_29_cast_fp16)[name = string("obj_657_cast_fp16")]; + int32 var_22650 = const()[name = string("op_22650"), val = int32(3)]; + int32 var_22660 = const()[name = string("op_22660"), val = int32(-2)]; + tensor inputs_sq_619_cast_fp16 = mul(x = inputs_619_cast_fp16, y = inputs_619_cast_fp16)[name = string("inputs_sq_619_cast_fp16")]; + tensor variance_619_axes_0 = const()[name = string("variance_619_axes_0"), val = tensor([1])]; + bool variance_619_keep_dims_0 = const()[name = string("variance_619_keep_dims_0"), val = bool(true)]; + tensor variance_619_cast_fp16 = reduce_mean(axes = variance_619_axes_0, keep_dims = variance_619_keep_dims_0, x = inputs_sq_619_cast_fp16)[name = string("variance_619_cast_fp16")]; + fp16 var_22674_to_fp16 = const()[name = string("op_22674_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22675_cast_fp16 = add(x = variance_619_cast_fp16, y = var_22674_to_fp16)[name = string("op_22675_cast_fp16")]; + fp32 var_22676_epsilon_0 = const()[name = string("op_22676_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22676_cast_fp16 = rsqrt(epsilon = var_22676_epsilon_0, x = var_22675_cast_fp16)[name = string("op_22676_cast_fp16")]; + tensor hidden_states_767_cast_fp16 = mul(x = inputs_619_cast_fp16, y = var_22676_cast_fp16)[name = string("hidden_states_767_cast_fp16")]; + tensor obj_653_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_767_cast_fp16)[name = string("obj_653_cast_fp16")]; + string query_445_pad_type_0 = const()[name = string("query_445_pad_type_0"), val = string("valid")]; + tensor query_445_strides_0 = const()[name = string("query_445_strides_0"), val = tensor([1, 1])]; + tensor query_445_pad_0 = const()[name = string("query_445_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_445_dilations_0 = const()[name = string("query_445_dilations_0"), val = tensor([1, 1])]; + int32 query_445_groups_0 = const()[name = string("query_445_groups_0"), val = int32(1)]; + tensor query_445_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_445_dilations_0, groups = query_445_groups_0, pad = query_445_pad_0, pad_type = query_445_pad_type_0, strides = query_445_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_653_cast_fp16)[name = string("query_445_cast_fp16")]; + string current_key_297_pad_type_0 = const()[name = string("current_key_297_pad_type_0"), val = string("valid")]; + tensor current_key_297_strides_0 = const()[name = string("current_key_297_strides_0"), val = tensor([1, 1])]; + tensor current_key_297_pad_0 = const()[name = string("current_key_297_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_297_dilations_0 = const()[name = string("current_key_297_dilations_0"), val = tensor([1, 1])]; + int32 current_key_297_groups_0 = const()[name = string("current_key_297_groups_0"), val = int32(1)]; + tensor current_key_297_cast_fp16 = conv(dilations = current_key_297_dilations_0, groups = current_key_297_groups_0, pad = current_key_297_pad_0, pad_type = current_key_297_pad_type_0, strides = current_key_297_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_653_cast_fp16)[name = string("current_key_297_cast_fp16")]; + string current_value_149_pad_type_0 = const()[name = string("current_value_149_pad_type_0"), val = string("valid")]; + tensor current_value_149_strides_0 = const()[name = string("current_value_149_strides_0"), val = tensor([1, 1])]; + tensor current_value_149_pad_0 = const()[name = string("current_value_149_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_149_dilations_0 = const()[name = string("current_value_149_dilations_0"), val = tensor([1, 1])]; + int32 current_value_149_groups_0 = const()[name = string("current_value_149_groups_0"), val = int32(1)]; + tensor current_value_149_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_149_dilations_0, groups = current_value_149_groups_0, pad = current_value_149_pad_0, pad_type = current_value_149_pad_type_0, strides = current_value_149_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_653_cast_fp16)[name = string("current_value_149_cast_fp16")]; + tensor var_22713 = const()[name = string("op_22713"), val = tensor([16, 128, 1, 1])]; + tensor inputs_621_cast_fp16 = reshape(shape = var_22713, x = query_445_cast_fp16)[name = string("inputs_621_cast_fp16")]; + tensor inputs_sq_621_cast_fp16 = mul(x = inputs_621_cast_fp16, y = inputs_621_cast_fp16)[name = string("inputs_sq_621_cast_fp16")]; + tensor variance_621_axes_0 = const()[name = string("variance_621_axes_0"), val = tensor([1])]; + bool variance_621_keep_dims_0 = const()[name = string("variance_621_keep_dims_0"), val = bool(true)]; + tensor variance_621_cast_fp16 = reduce_mean(axes = variance_621_axes_0, keep_dims = variance_621_keep_dims_0, x = inputs_sq_621_cast_fp16)[name = string("variance_621_cast_fp16")]; + fp16 var_22719_to_fp16 = const()[name = string("op_22719_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22720_cast_fp16 = add(x = variance_621_cast_fp16, y = var_22719_to_fp16)[name = string("op_22720_cast_fp16")]; + fp32 var_22721_epsilon_0 = const()[name = string("op_22721_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22721_cast_fp16 = rsqrt(epsilon = var_22721_epsilon_0, x = var_22720_cast_fp16)[name = string("op_22721_cast_fp16")]; + tensor hidden_states_769_cast_fp16 = mul(x = inputs_621_cast_fp16, y = var_22721_cast_fp16)[name = string("hidden_states_769_cast_fp16")]; + tensor query_normed_149_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_769_cast_fp16)[name = string("query_normed_149_cast_fp16")]; + tensor var_22729 = const()[name = string("op_22729"), val = tensor([8, 128, 1, 1])]; + tensor inputs_623_cast_fp16 = reshape(shape = var_22729, x = current_key_297_cast_fp16)[name = string("inputs_623_cast_fp16")]; + tensor inputs_sq_623_cast_fp16 = mul(x = inputs_623_cast_fp16, y = inputs_623_cast_fp16)[name = string("inputs_sq_623_cast_fp16")]; + tensor variance_623_axes_0 = const()[name = string("variance_623_axes_0"), val = tensor([1])]; + bool variance_623_keep_dims_0 = const()[name = string("variance_623_keep_dims_0"), val = bool(true)]; + tensor variance_623_cast_fp16 = reduce_mean(axes = variance_623_axes_0, keep_dims = variance_623_keep_dims_0, x = inputs_sq_623_cast_fp16)[name = string("variance_623_cast_fp16")]; + fp16 var_22735_to_fp16 = const()[name = string("op_22735_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22736_cast_fp16 = add(x = variance_623_cast_fp16, y = var_22735_to_fp16)[name = string("op_22736_cast_fp16")]; + fp32 var_22737_epsilon_0 = const()[name = string("op_22737_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22737_cast_fp16 = rsqrt(epsilon = var_22737_epsilon_0, x = var_22736_cast_fp16)[name = string("op_22737_cast_fp16")]; + tensor hidden_states_771_cast_fp16 = mul(x = inputs_623_cast_fp16, y = var_22737_cast_fp16)[name = string("hidden_states_771_cast_fp16")]; + tensor current_key_normed_149_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_771_cast_fp16)[name = string("current_key_normed_149_cast_fp16")]; + tensor var_22755 = const()[name = string("op_22755"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_593_cast_fp16 = reshape(shape = var_22755, x = query_normed_149_cast_fp16)[name = string("mh_q_593_cast_fp16")]; + tensor var_22757 = const()[name = string("op_22757"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_593_cast_fp16 = reshape(shape = var_22757, x = current_key_normed_149_cast_fp16)[name = string("mh_k_593_cast_fp16")]; + tensor var_22761_cast_fp16 = mul(x = mh_q_593_cast_fp16, y = cos_141_to_fp16)[name = string("op_22761_cast_fp16")]; + tensor var_22766_begin_0 = const()[name = string("op_22766_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_22766_end_0 = const()[name = string("op_22766_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_22766_end_mask_0 = const()[name = string("op_22766_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_22766_cast_fp16 = slice_by_index(begin = var_22766_begin_0, end = var_22766_end_0, end_mask = var_22766_end_mask_0, x = mh_q_593_cast_fp16)[name = string("op_22766_cast_fp16")]; + tensor var_22772_begin_0 = const()[name = string("op_22772_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_22772_end_0 = const()[name = string("op_22772_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_22772_end_mask_0 = const()[name = string("op_22772_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_22772_cast_fp16 = slice_by_index(begin = var_22772_begin_0, end = var_22772_end_0, end_mask = var_22772_end_mask_0, x = mh_q_593_cast_fp16)[name = string("op_22772_cast_fp16")]; + fp16 const_1508_promoted_to_fp16 = const()[name = string("const_1508_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_22774_cast_fp16 = mul(x = var_22772_cast_fp16, y = const_1508_promoted_to_fp16)[name = string("op_22774_cast_fp16")]; + bool var_22776_interleave_0 = const()[name = string("op_22776_interleave_0"), val = bool(false)]; + tensor var_22776_cast_fp16 = concat(axis = var_22660, interleave = var_22776_interleave_0, values = (var_22774_cast_fp16, var_22766_cast_fp16))[name = string("op_22776_cast_fp16")]; + tensor var_22777_cast_fp16 = mul(x = var_22776_cast_fp16, y = sin_141_to_fp16)[name = string("op_22777_cast_fp16")]; + tensor mh_q_595_cast_fp16 = add(x = var_22761_cast_fp16, y = var_22777_cast_fp16)[name = string("mh_q_595_cast_fp16")]; + tensor var_22779_cast_fp16 = mul(x = mh_k_593_cast_fp16, y = cos_141_to_fp16)[name = string("op_22779_cast_fp16")]; + tensor var_22784_begin_0 = const()[name = string("op_22784_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_22784_end_0 = const()[name = string("op_22784_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_22784_end_mask_0 = const()[name = string("op_22784_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_22784_cast_fp16 = slice_by_index(begin = var_22784_begin_0, end = var_22784_end_0, end_mask = var_22784_end_mask_0, x = mh_k_593_cast_fp16)[name = string("op_22784_cast_fp16")]; + tensor var_22790_begin_0 = const()[name = string("op_22790_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_22790_end_0 = const()[name = string("op_22790_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_22790_end_mask_0 = const()[name = string("op_22790_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_22790_cast_fp16 = slice_by_index(begin = var_22790_begin_0, end = var_22790_end_0, end_mask = var_22790_end_mask_0, x = mh_k_593_cast_fp16)[name = string("op_22790_cast_fp16")]; + fp16 const_1511_promoted_to_fp16 = const()[name = string("const_1511_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_22792_cast_fp16 = mul(x = var_22790_cast_fp16, y = const_1511_promoted_to_fp16)[name = string("op_22792_cast_fp16")]; + bool var_22794_interleave_0 = const()[name = string("op_22794_interleave_0"), val = bool(false)]; + tensor var_22794_cast_fp16 = concat(axis = var_22660, interleave = var_22794_interleave_0, values = (var_22792_cast_fp16, var_22784_cast_fp16))[name = string("op_22794_cast_fp16")]; + tensor var_22795_cast_fp16 = mul(x = var_22794_cast_fp16, y = sin_141_to_fp16)[name = string("op_22795_cast_fp16")]; + tensor mh_k_595_cast_fp16 = add(x = var_22779_cast_fp16, y = var_22795_cast_fp16)[name = string("mh_k_595_cast_fp16")]; + tensor var_22799 = const()[name = string("op_22799"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_299_cast_fp16 = reshape(shape = var_22799, x = mh_k_595_cast_fp16)[name = string("current_key_299_cast_fp16")]; + tensor var_22805_to_fp16 = const()[name = string("op_22805_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203520)))]; + tensor var_22806_cast_fp16 = mul(x = obj_655_cast_fp16, y = var_22805_to_fp16)[name = string("op_22806_cast_fp16")]; + tensor var_22803_to_fp16 = const()[name = string("op_22803_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203648)))]; + tensor var_22807_cast_fp16 = mul(x = current_key_299_cast_fp16, y = var_22803_to_fp16)[name = string("op_22807_cast_fp16")]; + tensor key_299_cast_fp16 = add(x = var_22806_cast_fp16, y = var_22807_cast_fp16)[name = string("key_299_cast_fp16")]; + tensor var_22809_to_fp16 = const()[name = string("op_22809_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203520)))]; + tensor var_22810_cast_fp16 = mul(x = obj_657_cast_fp16, y = var_22809_to_fp16)[name = string("op_22810_cast_fp16")]; + tensor var_22811_cast_fp16 = mul(x = current_value_149_cast_fp16, y = var_22803_to_fp16)[name = string("op_22811_cast_fp16")]; + tensor value_149_cast_fp16 = add(x = var_22810_cast_fp16, y = var_22811_cast_fp16)[name = string("value_149_cast_fp16")]; + fp16 var_22818_to_fp16 = const()[name = string("op_22818_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_599_cast_fp16 = mul(x = mh_q_595_cast_fp16, y = var_22818_to_fp16)[name = string("mh_q_599_cast_fp16")]; + tensor var_22820 = const()[name = string("op_22820"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_597_cast_fp16 = reshape(shape = var_22820, x = key_299_cast_fp16)[name = string("mh_k_597_cast_fp16")]; + tensor var_22822 = const()[name = string("op_22822"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_297_cast_fp16 = reshape(shape = var_22822, x = value_149_cast_fp16)[name = string("mh_v_297_cast_fp16")]; + tensor transpose_296_perm_0 = const()[name = string("transpose_296_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_148_reps_0 = const()[name = string("tile_148_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_296_cast_fp16 = transpose(perm = transpose_296_perm_0, x = mh_k_597_cast_fp16)[name = string("transpose_35")]; + tensor tile_148_cast_fp16 = tile(reps = tile_148_reps_0, x = transpose_296_cast_fp16)[name = string("tile_148_cast_fp16")]; + tensor concat_369 = const()[name = string("concat_369"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_296_cast_fp16 = reshape(shape = concat_369, x = tile_148_cast_fp16)[name = string("reshape_296_cast_fp16")]; + tensor transpose_297_perm_0 = const()[name = string("transpose_297_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_370 = const()[name = string("concat_370"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_297_cast_fp16 = transpose(perm = transpose_297_perm_0, x = reshape_296_cast_fp16)[name = string("transpose_34")]; + tensor reshape_297_cast_fp16 = reshape(shape = concat_370, x = transpose_297_cast_fp16)[name = string("reshape_297_cast_fp16")]; + tensor transpose_298_perm_0 = const()[name = string("transpose_298_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_149_reps_0 = const()[name = string("tile_149_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_298_cast_fp16 = transpose(perm = transpose_298_perm_0, x = mh_v_297_cast_fp16)[name = string("transpose_33")]; + tensor tile_149_cast_fp16 = tile(reps = tile_149_reps_0, x = transpose_298_cast_fp16)[name = string("tile_149_cast_fp16")]; + tensor concat_371 = const()[name = string("concat_371"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_298_cast_fp16 = reshape(shape = concat_371, x = tile_149_cast_fp16)[name = string("reshape_298_cast_fp16")]; + tensor transpose_299_perm_0 = const()[name = string("transpose_299_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_372 = const()[name = string("concat_372"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_299_cast_fp16 = transpose(perm = transpose_299_perm_0, x = reshape_298_cast_fp16)[name = string("transpose_32")]; + tensor reshape_299_cast_fp16 = reshape(shape = concat_372, x = transpose_299_cast_fp16)[name = string("reshape_299_cast_fp16")]; + tensor transpose_613_perm_0 = const()[name = string("transpose_613_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_445_transpose_x_1 = const()[name = string("mh_w_445_transpose_x_1"), val = bool(true)]; + bool mh_w_445_transpose_y_1 = const()[name = string("mh_w_445_transpose_y_1"), val = bool(false)]; + tensor transpose_613_cast_fp16 = transpose(perm = transpose_613_perm_0, x = reshape_297_cast_fp16)[name = string("transpose_31")]; + tensor mh_w_445_cast_fp16 = matmul(transpose_x = mh_w_445_transpose_x_1, transpose_y = mh_w_445_transpose_y_1, x = mh_q_599_cast_fp16, y = transpose_613_cast_fp16)[name = string("mh_w_445_cast_fp16")]; + tensor var_22830_to_fp16 = const()[name = string("op_22830_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203776)))]; + tensor mh_w_447_cast_fp16 = add(x = mh_w_445_cast_fp16, y = var_22830_to_fp16)[name = string("mh_w_447_cast_fp16")]; + tensor mh_w_449_cast_fp16 = softmax(axis = var_22650, x = mh_w_447_cast_fp16)[name = string("mh_w_449_cast_fp16")]; + tensor transpose_614_perm_0 = const()[name = string("transpose_614_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_149_transpose_x_1 = const()[name = string("attn_149_transpose_x_1"), val = bool(false)]; + bool attn_149_transpose_y_1 = const()[name = string("attn_149_transpose_y_1"), val = bool(true)]; + tensor transpose_614_cast_fp16 = transpose(perm = transpose_614_perm_0, x = reshape_299_cast_fp16)[name = string("transpose_30")]; + tensor attn_149_cast_fp16 = matmul(transpose_x = attn_149_transpose_x_1, transpose_y = attn_149_transpose_y_1, x = transpose_614_cast_fp16, y = mh_w_449_cast_fp16)[name = string("attn_149_cast_fp16")]; + tensor var_22836 = const()[name = string("op_22836"), val = tensor([1, 2048, 1, 1])]; + tensor input_645_cast_fp16 = reshape(shape = var_22836, x = attn_149_cast_fp16)[name = string("input_645_cast_fp16")]; + string obj_659_pad_type_0 = const()[name = string("obj_659_pad_type_0"), val = string("valid")]; + tensor obj_659_strides_0 = const()[name = string("obj_659_strides_0"), val = tensor([1, 1])]; + tensor obj_659_pad_0 = const()[name = string("obj_659_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_659_dilations_0 = const()[name = string("obj_659_dilations_0"), val = tensor([1, 1])]; + int32 obj_659_groups_0 = const()[name = string("obj_659_groups_0"), val = int32(1)]; + tensor obj_659_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_659_dilations_0, groups = obj_659_groups_0, pad = obj_659_pad_0, pad_type = obj_659_pad_type_0, strides = obj_659_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_645_cast_fp16)[name = string("obj_659_cast_fp16")]; + tensor inputs_625_cast_fp16 = add(x = inputs_619_cast_fp16, y = obj_659_cast_fp16)[name = string("inputs_625_cast_fp16")]; + tensor inputs_sq_625_cast_fp16 = mul(x = inputs_625_cast_fp16, y = inputs_625_cast_fp16)[name = string("inputs_sq_625_cast_fp16")]; + tensor variance_625_axes_0 = const()[name = string("variance_625_axes_0"), val = tensor([1])]; + bool variance_625_keep_dims_0 = const()[name = string("variance_625_keep_dims_0"), val = bool(true)]; + tensor variance_625_cast_fp16 = reduce_mean(axes = variance_625_axes_0, keep_dims = variance_625_keep_dims_0, x = inputs_sq_625_cast_fp16)[name = string("variance_625_cast_fp16")]; + fp16 var_22854_to_fp16 = const()[name = string("op_22854_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22855_cast_fp16 = add(x = variance_625_cast_fp16, y = var_22854_to_fp16)[name = string("op_22855_cast_fp16")]; + fp32 var_22856_epsilon_0 = const()[name = string("op_22856_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22856_cast_fp16 = rsqrt(epsilon = var_22856_epsilon_0, x = var_22855_cast_fp16)[name = string("op_22856_cast_fp16")]; + tensor hidden_states_773_cast_fp16 = mul(x = inputs_625_cast_fp16, y = var_22856_cast_fp16)[name = string("hidden_states_773_cast_fp16")]; + tensor input_647_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_773_cast_fp16)[name = string("input_647_cast_fp16")]; + string input_649_pad_type_0 = const()[name = string("input_649_pad_type_0"), val = string("valid")]; + tensor input_649_strides_0 = const()[name = string("input_649_strides_0"), val = tensor([1, 1])]; + tensor input_649_pad_0 = const()[name = string("input_649_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_649_dilations_0 = const()[name = string("input_649_dilations_0"), val = tensor([1, 1])]; + int32 input_649_groups_0 = const()[name = string("input_649_groups_0"), val = int32(1)]; + tensor input_649_cast_fp16 = conv(dilations = input_649_dilations_0, groups = input_649_groups_0, pad = input_649_pad_0, pad_type = input_649_pad_type_0, strides = input_649_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_647_cast_fp16)[name = string("input_649_cast_fp16")]; + tensor var_22870_cast_fp16 = silu(x = input_649_cast_fp16)[name = string("op_22870_cast_fp16")]; + string var_22876_pad_type_0 = const()[name = string("op_22876_pad_type_0"), val = string("valid")]; + tensor var_22876_strides_0 = const()[name = string("op_22876_strides_0"), val = tensor([1, 1])]; + tensor var_22876_pad_0 = const()[name = string("op_22876_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_22876_dilations_0 = const()[name = string("op_22876_dilations_0"), val = tensor([1, 1])]; + int32 var_22876_groups_0 = const()[name = string("op_22876_groups_0"), val = int32(1)]; + tensor var_22876_cast_fp16 = conv(dilations = var_22876_dilations_0, groups = var_22876_groups_0, pad = var_22876_pad_0, pad_type = var_22876_pad_type_0, strides = var_22876_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_647_cast_fp16)[name = string("op_22876_cast_fp16")]; + tensor input_651_cast_fp16 = mul(x = var_22870_cast_fp16, y = var_22876_cast_fp16)[name = string("input_651_cast_fp16")]; + string hidden_states_775_pad_type_0 = const()[name = string("hidden_states_775_pad_type_0"), val = string("valid")]; + tensor hidden_states_775_strides_0 = const()[name = string("hidden_states_775_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_775_pad_0 = const()[name = string("hidden_states_775_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_775_dilations_0 = const()[name = string("hidden_states_775_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_775_groups_0 = const()[name = string("hidden_states_775_groups_0"), val = int32(1)]; + tensor hidden_states_775_cast_fp16 = conv(dilations = hidden_states_775_dilations_0, groups = hidden_states_775_groups_0, pad = hidden_states_775_pad_0, pad_type = hidden_states_775_pad_type_0, strides = hidden_states_775_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_651_cast_fp16)[name = string("hidden_states_775_cast_fp16")]; + tensor inputs_627_cast_fp16 = add(x = inputs_625_cast_fp16, y = hidden_states_775_cast_fp16)[name = string("inputs_627_cast_fp16")]; + int32 var_22904 = const()[name = string("op_22904"), val = int32(1)]; + bool key_caches_interleave_0 = const()[name = string("key_caches_interleave_0"), val = bool(false)]; + tensor key_caches_cast_fp16 = concat(axis = var_22904, interleave = key_caches_interleave_0, values = (key_283_cast_fp16, key_287_cast_fp16, key_291_cast_fp16, key_295_cast_fp16, key_299_cast_fp16))[name = string("key_caches_cast_fp16")]; + int32 var_22907 = const()[name = string("op_22907"), val = int32(1)]; + bool value_caches_interleave_0 = const()[name = string("value_caches_interleave_0"), val = bool(false)]; + tensor value_caches_cast_fp16 = concat(axis = var_22907, interleave = value_caches_interleave_0, values = (value_141_cast_fp16, value_143_cast_fp16, value_145_cast_fp16, value_147_cast_fp16, value_149_cast_fp16))[name = string("value_caches_cast_fp16")]; + tensor inputs_sq_627_cast_fp16 = mul(x = inputs_627_cast_fp16, y = inputs_627_cast_fp16)[name = string("inputs_sq_627_cast_fp16")]; + tensor variance_627_axes_0 = const()[name = string("variance_627_axes_0"), val = tensor([1])]; + bool variance_627_keep_dims_0 = const()[name = string("variance_627_keep_dims_0"), val = bool(true)]; + tensor variance_627_cast_fp16 = reduce_mean(axes = variance_627_axes_0, keep_dims = variance_627_keep_dims_0, x = inputs_sq_627_cast_fp16)[name = string("variance_627_cast_fp16")]; + fp16 var_22917_to_fp16 = const()[name = string("op_22917_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_22918_cast_fp16 = add(x = variance_627_cast_fp16, y = var_22917_to_fp16)[name = string("op_22918_cast_fp16")]; + fp32 var_22919_epsilon_0 = const()[name = string("op_22919_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_22919_cast_fp16 = rsqrt(epsilon = var_22919_epsilon_0, x = var_22918_cast_fp16)[name = string("op_22919_cast_fp16")]; + tensor hidden_states_777_cast_fp16 = mul(x = inputs_627_cast_fp16, y = var_22919_cast_fp16)[name = string("hidden_states_777_cast_fp16")]; + tensor input_653_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_777_cast_fp16)[name = string("input_653_cast_fp16")]; + string logits_53_pad_type_0 = const()[name = string("logits_53_pad_type_0"), val = string("valid")]; + tensor logits_53_strides_0 = const()[name = string("logits_53_strides_0"), val = tensor([1, 1])]; + tensor logits_53_pad_0 = const()[name = string("logits_53_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_53_dilations_0 = const()[name = string("logits_53_dilations_0"), val = tensor([1, 1])]; + int32 logits_53_groups_0 = const()[name = string("logits_53_groups_0"), val = int32(1)]; + tensor lm_heads_13_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108077888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110175104))))[name = string("lm_heads_13_weight_to_fp16_palettized")]; + tensor logits_53_cast_fp16 = conv(dilations = logits_53_dilations_0, groups = logits_53_groups_0, pad = logits_53_pad_0, pad_type = logits_53_pad_type_0, strides = logits_53_strides_0, weight = lm_heads_13_weight_to_fp16_palettized, x = input_653_cast_fp16)[name = string("logits_53_cast_fp16")]; + tensor var_22937 = const()[name = string("op_22937"), val = tensor([1, 2048])]; + tensor logits_55_cast_fp16 = reshape(shape = var_22937, x = logits_53_cast_fp16)[name = string("logits_55_cast_fp16")]; + tensor scaled_logits_27_cast_fp16 = real_div(x = logits_55_cast_fp16, y = temperature)[name = string("scaled_logits_27_cast_fp16")]; + int32 var_22947 = const()[name = string("op_22947"), val = int32(100)]; + int32 top_values_27_axis_0 = const()[name = string("top_values_27_axis_0"), val = int32(1)]; + bool top_values_27_ascending_0 = const()[name = string("top_values_27_ascending_0"), val = bool(false)]; + bool top_values_27_sort_0 = const()[name = string("top_values_27_sort_0"), val = bool(true)]; + bool top_values_27_return_indices_0 = const()[name = string("top_values_27_return_indices_0"), val = bool(true)]; + string top_values_27_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_27_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_27_cast_fp16_cast_uint16_0, tensor top_values_27_cast_fp16_cast_uint16_1 = topk(ascending = top_values_27_ascending_0, axis = top_values_27_axis_0, k = var_22947, output_indices_dtype = top_values_27_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_27_return_indices_0, sort = top_values_27_sort_0, x = scaled_logits_27_cast_fp16)[name = string("top_values_27_cast_fp16_cast_uint16")]; + tensor var_22953_cast_fp16 = mul(x = top_values_27_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_22953_cast_fp16")]; + tensor var_22957_cast_fp16 = add(x = var_22953_cast_fp16, y = var_3001_cast_fp16)[name = string("op_22957_cast_fp16")]; + tensor reduce_min_13_axes_0 = const()[name = string("reduce_min_13_axes_0"), val = tensor([1])]; + bool reduce_min_13_keep_dims_0 = const()[name = string("reduce_min_13_keep_dims_0"), val = bool(true)]; + tensor reduce_min_13_cast_fp16 = reduce_min(axes = reduce_min_13_axes_0, keep_dims = reduce_min_13_keep_dims_0, x = var_22957_cast_fp16)[name = string("reduce_min_13_cast_fp16")]; + tensor var_22960_cast_fp16 = greater_equal(x = scaled_logits_27_cast_fp16, y = reduce_min_13_cast_fp16)[name = string("op_22960_cast_fp16")]; + fp16 var_22961_value_0_to_fp16 = const()[name = string("op_22961_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_22961_cast_fp16 = fill_like(ref_tensor = scaled_logits_27_cast_fp16, value = var_22961_value_0_to_fp16)[name = string("op_22961_cast_fp16")]; + tensor masked_logits_27_cast_fp16 = select(a = scaled_logits_27_cast_fp16, b = var_22961_cast_fp16, cond = var_22960_cast_fp16)[name = string("masked_logits_27_cast_fp16")]; + tensor var_22965_begin_0 = const()[name = string("op_22965_begin_0"), val = tensor([13, 0])]; + tensor var_22965_end_0 = const()[name = string("op_22965_end_0"), val = tensor([14, 2048])]; + tensor var_22965_end_mask_0 = const()[name = string("op_22965_end_mask_0"), val = tensor([false, true])]; + tensor var_22965_squeeze_mask_0 = const()[name = string("op_22965_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_22965_cast_fp16 = slice_by_index(begin = var_22965_begin_0, end = var_22965_end_0, end_mask = var_22965_end_mask_0, squeeze_mask = var_22965_squeeze_mask_0, x = gumbel)[name = string("op_22965_cast_fp16")]; + tensor var_22968 = const()[name = string("op_22968"), val = tensor([1, 2048])]; + tensor var_22969_cast_fp16 = reshape(shape = var_22968, x = var_22965_cast_fp16)[name = string("op_22969_cast_fp16")]; + tensor noisy_logits_27_cast_fp16 = add(x = masked_logits_27_cast_fp16, y = var_22969_cast_fp16)[name = string("noisy_logits_27_cast_fp16")]; + int32 code_27_axis_0 = const()[name = string("code_27_axis_0"), val = int32(1)]; + bool code_27_keep_dims_0 = const()[name = string("code_27_keep_dims_0"), val = bool(false)]; + string code_27_output_dtype_0 = const()[name = string("code_27_output_dtype_0"), val = string("int32")]; + tensor code_27_cast_fp16 = reduce_argmax(axis = code_27_axis_0, keep_dims = code_27_keep_dims_0, output_dtype = code_27_output_dtype_0, x = noisy_logits_27_cast_fp16)[name = string("code_27_cast_fp16")]; + int32 var_22980 = const()[name = string("op_22980"), val = int32(26624)]; + tensor input_655 = add(x = code_27_cast_fp16, y = var_22980)[name = string("input_655")]; + int32 code_embed_53_axis_0 = const()[name = string("code_embed_53_axis_0"), val = int32(0)]; + int32 code_embed_53_batch_dims_0 = const()[name = string("code_embed_53_batch_dims_0"), val = int32(0)]; + bool code_embed_53_validate_indices_0 = const()[name = string("code_embed_53_validate_indices_0"), val = bool(false)]; + string input_655_to_uint16_dtype_0 = const()[name = string("input_655_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_655_to_uint16 = cast(dtype = input_655_to_uint16_dtype_0, x = input_655)[name = string("cast_1")]; + tensor code_embed_53_cast_fp16_cast_uint16 = gather(axis = code_embed_53_axis_0, batch_dims = code_embed_53_batch_dims_0, indices = input_655_to_uint16, validate_indices = code_embed_53_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_53_cast_fp16_cast_uint16")]; + tensor var_22984 = const()[name = string("op_22984"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_55_cast_fp16 = reshape(shape = var_22984, x = code_embed_53_cast_fp16_cast_uint16)[name = string("code_embed_55_cast_fp16")]; + tensor embed_sum_cast_fp16 = add(x = embed_sum_27_cast_fp16, y = code_embed_55_cast_fp16)[name = string("embed_sum_cast_fp16")]; + string inputs_629_pad_type_0 = const()[name = string("inputs_629_pad_type_0"), val = string("valid")]; + tensor inputs_629_strides_0 = const()[name = string("inputs_629_strides_0"), val = tensor([1, 1])]; + tensor inputs_629_pad_0 = const()[name = string("inputs_629_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_629_dilations_0 = const()[name = string("inputs_629_dilations_0"), val = tensor([1, 1])]; + int32 inputs_629_groups_0 = const()[name = string("inputs_629_groups_0"), val = int32(1)]; + tensor inputs_629_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_629_dilations_0, groups = inputs_629_groups_0, pad = inputs_629_pad_0, pad_type = inputs_629_pad_type_0, strides = inputs_629_strides_0, weight = input_projection_weight_to_fp16_palettized, x = code_embed_55_cast_fp16)[name = string("inputs_629_cast_fp16")]; + tensor obj_663_begin_0 = const()[name = string("obj_663_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_663_end_0 = const()[name = string("obj_663_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_663_end_mask_0 = const()[name = string("obj_663_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_663_cast_fp16 = slice_by_index(begin = obj_663_begin_0, end = obj_663_end_0, end_mask = obj_663_end_mask_0, x = key_caches_cast_fp16)[name = string("obj_663_cast_fp16")]; + tensor obj_665_begin_0 = const()[name = string("obj_665_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_665_end_0 = const()[name = string("obj_665_end_0"), val = tensor([1, 1024, 1, 16])]; + tensor obj_665_end_mask_0 = const()[name = string("obj_665_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_665_cast_fp16 = slice_by_index(begin = obj_665_begin_0, end = obj_665_end_0, end_mask = obj_665_end_mask_0, x = value_caches_cast_fp16)[name = string("obj_665_cast_fp16")]; + int32 var_23075 = const()[name = string("op_23075"), val = int32(3)]; + int32 var_23085 = const()[name = string("op_23085"), val = int32(-2)]; + tensor inputs_sq_629_cast_fp16 = mul(x = inputs_629_cast_fp16, y = inputs_629_cast_fp16)[name = string("inputs_sq_629_cast_fp16")]; + tensor variance_629_axes_0 = const()[name = string("variance_629_axes_0"), val = tensor([1])]; + bool variance_629_keep_dims_0 = const()[name = string("variance_629_keep_dims_0"), val = bool(true)]; + tensor variance_629_cast_fp16 = reduce_mean(axes = variance_629_axes_0, keep_dims = variance_629_keep_dims_0, x = inputs_sq_629_cast_fp16)[name = string("variance_629_cast_fp16")]; + fp16 var_23099_to_fp16 = const()[name = string("op_23099_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23100_cast_fp16 = add(x = variance_629_cast_fp16, y = var_23099_to_fp16)[name = string("op_23100_cast_fp16")]; + fp32 var_23101_epsilon_0 = const()[name = string("op_23101_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23101_cast_fp16 = rsqrt(epsilon = var_23101_epsilon_0, x = var_23100_cast_fp16)[name = string("op_23101_cast_fp16")]; + tensor hidden_states_779_cast_fp16 = mul(x = inputs_629_cast_fp16, y = var_23101_cast_fp16)[name = string("hidden_states_779_cast_fp16")]; + tensor obj_661_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_779_cast_fp16)[name = string("obj_661_cast_fp16")]; + string query_451_pad_type_0 = const()[name = string("query_451_pad_type_0"), val = string("valid")]; + tensor query_451_strides_0 = const()[name = string("query_451_strides_0"), val = tensor([1, 1])]; + tensor query_451_pad_0 = const()[name = string("query_451_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_451_dilations_0 = const()[name = string("query_451_dilations_0"), val = tensor([1, 1])]; + int32 query_451_groups_0 = const()[name = string("query_451_groups_0"), val = int32(1)]; + tensor query_451_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_451_dilations_0, groups = query_451_groups_0, pad = query_451_pad_0, pad_type = query_451_pad_type_0, strides = query_451_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_661_cast_fp16)[name = string("query_451_cast_fp16")]; + string current_key_301_pad_type_0 = const()[name = string("current_key_301_pad_type_0"), val = string("valid")]; + tensor current_key_301_strides_0 = const()[name = string("current_key_301_strides_0"), val = tensor([1, 1])]; + tensor current_key_301_pad_0 = const()[name = string("current_key_301_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_301_dilations_0 = const()[name = string("current_key_301_dilations_0"), val = tensor([1, 1])]; + int32 current_key_301_groups_0 = const()[name = string("current_key_301_groups_0"), val = int32(1)]; + tensor current_key_301_cast_fp16 = conv(dilations = current_key_301_dilations_0, groups = current_key_301_groups_0, pad = current_key_301_pad_0, pad_type = current_key_301_pad_type_0, strides = current_key_301_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_661_cast_fp16)[name = string("current_key_301_cast_fp16")]; + string current_value_151_pad_type_0 = const()[name = string("current_value_151_pad_type_0"), val = string("valid")]; + tensor current_value_151_strides_0 = const()[name = string("current_value_151_strides_0"), val = tensor([1, 1])]; + tensor current_value_151_pad_0 = const()[name = string("current_value_151_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_151_dilations_0 = const()[name = string("current_value_151_dilations_0"), val = tensor([1, 1])]; + int32 current_value_151_groups_0 = const()[name = string("current_value_151_groups_0"), val = int32(1)]; + tensor current_value_151_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_151_dilations_0, groups = current_value_151_groups_0, pad = current_value_151_pad_0, pad_type = current_value_151_pad_type_0, strides = current_value_151_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_661_cast_fp16)[name = string("current_value_151_cast_fp16")]; + tensor var_23138 = const()[name = string("op_23138"), val = tensor([16, 128, 1, 1])]; + tensor inputs_631_cast_fp16 = reshape(shape = var_23138, x = query_451_cast_fp16)[name = string("inputs_631_cast_fp16")]; + tensor inputs_sq_631_cast_fp16 = mul(x = inputs_631_cast_fp16, y = inputs_631_cast_fp16)[name = string("inputs_sq_631_cast_fp16")]; + tensor variance_631_axes_0 = const()[name = string("variance_631_axes_0"), val = tensor([1])]; + bool variance_631_keep_dims_0 = const()[name = string("variance_631_keep_dims_0"), val = bool(true)]; + tensor variance_631_cast_fp16 = reduce_mean(axes = variance_631_axes_0, keep_dims = variance_631_keep_dims_0, x = inputs_sq_631_cast_fp16)[name = string("variance_631_cast_fp16")]; + fp16 var_23144_to_fp16 = const()[name = string("op_23144_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23145_cast_fp16 = add(x = variance_631_cast_fp16, y = var_23144_to_fp16)[name = string("op_23145_cast_fp16")]; + fp32 var_23146_epsilon_0 = const()[name = string("op_23146_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23146_cast_fp16 = rsqrt(epsilon = var_23146_epsilon_0, x = var_23145_cast_fp16)[name = string("op_23146_cast_fp16")]; + tensor hidden_states_781_cast_fp16 = mul(x = inputs_631_cast_fp16, y = var_23146_cast_fp16)[name = string("hidden_states_781_cast_fp16")]; + tensor query_normed_151_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_781_cast_fp16)[name = string("query_normed_151_cast_fp16")]; + tensor var_23154 = const()[name = string("op_23154"), val = tensor([8, 128, 1, 1])]; + tensor inputs_633_cast_fp16 = reshape(shape = var_23154, x = current_key_301_cast_fp16)[name = string("inputs_633_cast_fp16")]; + tensor inputs_sq_633_cast_fp16 = mul(x = inputs_633_cast_fp16, y = inputs_633_cast_fp16)[name = string("inputs_sq_633_cast_fp16")]; + tensor variance_633_axes_0 = const()[name = string("variance_633_axes_0"), val = tensor([1])]; + bool variance_633_keep_dims_0 = const()[name = string("variance_633_keep_dims_0"), val = bool(true)]; + tensor variance_633_cast_fp16 = reduce_mean(axes = variance_633_axes_0, keep_dims = variance_633_keep_dims_0, x = inputs_sq_633_cast_fp16)[name = string("variance_633_cast_fp16")]; + fp16 var_23160_to_fp16 = const()[name = string("op_23160_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23161_cast_fp16 = add(x = variance_633_cast_fp16, y = var_23160_to_fp16)[name = string("op_23161_cast_fp16")]; + fp32 var_23162_epsilon_0 = const()[name = string("op_23162_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23162_cast_fp16 = rsqrt(epsilon = var_23162_epsilon_0, x = var_23161_cast_fp16)[name = string("op_23162_cast_fp16")]; + tensor hidden_states_783_cast_fp16 = mul(x = inputs_633_cast_fp16, y = var_23162_cast_fp16)[name = string("hidden_states_783_cast_fp16")]; + tensor current_key_normed_151_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_783_cast_fp16)[name = string("current_key_normed_151_cast_fp16")]; + tensor var_23180 = const()[name = string("op_23180"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_601_cast_fp16 = reshape(shape = var_23180, x = query_normed_151_cast_fp16)[name = string("mh_q_601_cast_fp16")]; + tensor var_23182 = const()[name = string("op_23182"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_601_cast_fp16 = reshape(shape = var_23182, x = current_key_normed_151_cast_fp16)[name = string("mh_k_601_cast_fp16")]; + tensor cos_151_to_fp16 = const()[name = string("cos_151_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175203904)))]; + tensor var_23186_cast_fp16 = mul(x = mh_q_601_cast_fp16, y = cos_151_to_fp16)[name = string("op_23186_cast_fp16")]; + tensor var_23191_begin_0 = const()[name = string("op_23191_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_23191_end_0 = const()[name = string("op_23191_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_23191_end_mask_0 = const()[name = string("op_23191_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_23191_cast_fp16 = slice_by_index(begin = var_23191_begin_0, end = var_23191_end_0, end_mask = var_23191_end_mask_0, x = mh_q_601_cast_fp16)[name = string("op_23191_cast_fp16")]; + tensor var_23197_begin_0 = const()[name = string("op_23197_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_23197_end_0 = const()[name = string("op_23197_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_23197_end_mask_0 = const()[name = string("op_23197_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_23197_cast_fp16 = slice_by_index(begin = var_23197_begin_0, end = var_23197_end_0, end_mask = var_23197_end_mask_0, x = mh_q_601_cast_fp16)[name = string("op_23197_cast_fp16")]; + fp16 const_1529_promoted_to_fp16 = const()[name = string("const_1529_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_23199_cast_fp16 = mul(x = var_23197_cast_fp16, y = const_1529_promoted_to_fp16)[name = string("op_23199_cast_fp16")]; + bool var_23201_interleave_0 = const()[name = string("op_23201_interleave_0"), val = bool(false)]; + tensor var_23201_cast_fp16 = concat(axis = var_23085, interleave = var_23201_interleave_0, values = (var_23199_cast_fp16, var_23191_cast_fp16))[name = string("op_23201_cast_fp16")]; + tensor sin_151_to_fp16 = const()[name = string("sin_151_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204224)))]; + tensor var_23202_cast_fp16 = mul(x = var_23201_cast_fp16, y = sin_151_to_fp16)[name = string("op_23202_cast_fp16")]; + tensor mh_q_603_cast_fp16 = add(x = var_23186_cast_fp16, y = var_23202_cast_fp16)[name = string("mh_q_603_cast_fp16")]; + tensor var_23204_cast_fp16 = mul(x = mh_k_601_cast_fp16, y = cos_151_to_fp16)[name = string("op_23204_cast_fp16")]; + tensor var_23209_begin_0 = const()[name = string("op_23209_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_23209_end_0 = const()[name = string("op_23209_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_23209_end_mask_0 = const()[name = string("op_23209_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_23209_cast_fp16 = slice_by_index(begin = var_23209_begin_0, end = var_23209_end_0, end_mask = var_23209_end_mask_0, x = mh_k_601_cast_fp16)[name = string("op_23209_cast_fp16")]; + tensor var_23215_begin_0 = const()[name = string("op_23215_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_23215_end_0 = const()[name = string("op_23215_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_23215_end_mask_0 = const()[name = string("op_23215_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_23215_cast_fp16 = slice_by_index(begin = var_23215_begin_0, end = var_23215_end_0, end_mask = var_23215_end_mask_0, x = mh_k_601_cast_fp16)[name = string("op_23215_cast_fp16")]; + fp16 const_1532_promoted_to_fp16 = const()[name = string("const_1532_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_23217_cast_fp16 = mul(x = var_23215_cast_fp16, y = const_1532_promoted_to_fp16)[name = string("op_23217_cast_fp16")]; + bool var_23219_interleave_0 = const()[name = string("op_23219_interleave_0"), val = bool(false)]; + tensor var_23219_cast_fp16 = concat(axis = var_23085, interleave = var_23219_interleave_0, values = (var_23217_cast_fp16, var_23209_cast_fp16))[name = string("op_23219_cast_fp16")]; + tensor var_23220_cast_fp16 = mul(x = var_23219_cast_fp16, y = sin_151_to_fp16)[name = string("op_23220_cast_fp16")]; + tensor mh_k_603_cast_fp16 = add(x = var_23204_cast_fp16, y = var_23220_cast_fp16)[name = string("mh_k_603_cast_fp16")]; + tensor var_23224 = const()[name = string("op_23224"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_303_cast_fp16 = reshape(shape = var_23224, x = mh_k_603_cast_fp16)[name = string("current_key_303_cast_fp16")]; + tensor var_23230_to_fp16 = const()[name = string("op_23230_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204544)))]; + tensor var_23231_cast_fp16 = mul(x = obj_663_cast_fp16, y = var_23230_to_fp16)[name = string("op_23231_cast_fp16")]; + tensor var_23228_to_fp16 = const()[name = string("op_23228_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204672)))]; + tensor var_23232_cast_fp16 = mul(x = current_key_303_cast_fp16, y = var_23228_to_fp16)[name = string("op_23232_cast_fp16")]; + tensor key_303_cast_fp16 = add(x = var_23231_cast_fp16, y = var_23232_cast_fp16)[name = string("key_303_cast_fp16")]; + tensor var_23234_to_fp16 = const()[name = string("op_23234_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204544)))]; + tensor var_23235_cast_fp16 = mul(x = obj_665_cast_fp16, y = var_23234_to_fp16)[name = string("op_23235_cast_fp16")]; + tensor var_23236_cast_fp16 = mul(x = current_value_151_cast_fp16, y = var_23228_to_fp16)[name = string("op_23236_cast_fp16")]; + tensor value_151_cast_fp16 = add(x = var_23235_cast_fp16, y = var_23236_cast_fp16)[name = string("value_151_cast_fp16")]; + fp16 var_23243_to_fp16 = const()[name = string("op_23243_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_607_cast_fp16 = mul(x = mh_q_603_cast_fp16, y = var_23243_to_fp16)[name = string("mh_q_607_cast_fp16")]; + tensor var_23245 = const()[name = string("op_23245"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_605_cast_fp16 = reshape(shape = var_23245, x = key_303_cast_fp16)[name = string("mh_k_605_cast_fp16")]; + tensor var_23247 = const()[name = string("op_23247"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_301_cast_fp16 = reshape(shape = var_23247, x = value_151_cast_fp16)[name = string("mh_v_301_cast_fp16")]; + tensor transpose_300_perm_0 = const()[name = string("transpose_300_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_150_reps_0 = const()[name = string("tile_150_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_300_cast_fp16 = transpose(perm = transpose_300_perm_0, x = mh_k_605_cast_fp16)[name = string("transpose_29")]; + tensor tile_150_cast_fp16 = tile(reps = tile_150_reps_0, x = transpose_300_cast_fp16)[name = string("tile_150_cast_fp16")]; + tensor concat_378 = const()[name = string("concat_378"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_300_cast_fp16 = reshape(shape = concat_378, x = tile_150_cast_fp16)[name = string("reshape_300_cast_fp16")]; + tensor transpose_301_perm_0 = const()[name = string("transpose_301_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_379 = const()[name = string("concat_379"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_301_cast_fp16 = transpose(perm = transpose_301_perm_0, x = reshape_300_cast_fp16)[name = string("transpose_28")]; + tensor reshape_301_cast_fp16 = reshape(shape = concat_379, x = transpose_301_cast_fp16)[name = string("reshape_301_cast_fp16")]; + tensor transpose_302_perm_0 = const()[name = string("transpose_302_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_151_reps_0 = const()[name = string("tile_151_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_302_cast_fp16 = transpose(perm = transpose_302_perm_0, x = mh_v_301_cast_fp16)[name = string("transpose_27")]; + tensor tile_151_cast_fp16 = tile(reps = tile_151_reps_0, x = transpose_302_cast_fp16)[name = string("tile_151_cast_fp16")]; + tensor concat_380 = const()[name = string("concat_380"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_302_cast_fp16 = reshape(shape = concat_380, x = tile_151_cast_fp16)[name = string("reshape_302_cast_fp16")]; + tensor transpose_303_perm_0 = const()[name = string("transpose_303_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_381 = const()[name = string("concat_381"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_303_cast_fp16 = transpose(perm = transpose_303_perm_0, x = reshape_302_cast_fp16)[name = string("transpose_26")]; + tensor reshape_303_cast_fp16 = reshape(shape = concat_381, x = transpose_303_cast_fp16)[name = string("reshape_303_cast_fp16")]; + tensor transpose_617_perm_0 = const()[name = string("transpose_617_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_451_transpose_x_1 = const()[name = string("mh_w_451_transpose_x_1"), val = bool(true)]; + bool mh_w_451_transpose_y_1 = const()[name = string("mh_w_451_transpose_y_1"), val = bool(false)]; + tensor transpose_617_cast_fp16 = transpose(perm = transpose_617_perm_0, x = reshape_301_cast_fp16)[name = string("transpose_25")]; + tensor mh_w_451_cast_fp16 = matmul(transpose_x = mh_w_451_transpose_x_1, transpose_y = mh_w_451_transpose_y_1, x = mh_q_607_cast_fp16, y = transpose_617_cast_fp16)[name = string("mh_w_451_cast_fp16")]; + tensor mh_w_455_cast_fp16 = softmax(axis = var_23075, x = mh_w_451_cast_fp16)[name = string("mh_w_455_cast_fp16")]; + tensor transpose_618_perm_0 = const()[name = string("transpose_618_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_151_transpose_x_1 = const()[name = string("attn_151_transpose_x_1"), val = bool(false)]; + bool attn_151_transpose_y_1 = const()[name = string("attn_151_transpose_y_1"), val = bool(true)]; + tensor transpose_618_cast_fp16 = transpose(perm = transpose_618_perm_0, x = reshape_303_cast_fp16)[name = string("transpose_24")]; + tensor attn_151_cast_fp16 = matmul(transpose_x = attn_151_transpose_x_1, transpose_y = attn_151_transpose_y_1, x = transpose_618_cast_fp16, y = mh_w_455_cast_fp16)[name = string("attn_151_cast_fp16")]; + tensor var_23261 = const()[name = string("op_23261"), val = tensor([1, 2048, 1, 1])]; + tensor input_657_cast_fp16 = reshape(shape = var_23261, x = attn_151_cast_fp16)[name = string("input_657_cast_fp16")]; + string obj_671_pad_type_0 = const()[name = string("obj_671_pad_type_0"), val = string("valid")]; + tensor obj_671_strides_0 = const()[name = string("obj_671_strides_0"), val = tensor([1, 1])]; + tensor obj_671_pad_0 = const()[name = string("obj_671_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_671_dilations_0 = const()[name = string("obj_671_dilations_0"), val = tensor([1, 1])]; + int32 obj_671_groups_0 = const()[name = string("obj_671_groups_0"), val = int32(1)]; + tensor obj_671_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_671_dilations_0, groups = obj_671_groups_0, pad = obj_671_pad_0, pad_type = obj_671_pad_type_0, strides = obj_671_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_657_cast_fp16)[name = string("obj_671_cast_fp16")]; + tensor inputs_635_cast_fp16 = add(x = inputs_629_cast_fp16, y = obj_671_cast_fp16)[name = string("inputs_635_cast_fp16")]; + tensor inputs_sq_635_cast_fp16 = mul(x = inputs_635_cast_fp16, y = inputs_635_cast_fp16)[name = string("inputs_sq_635_cast_fp16")]; + tensor variance_635_axes_0 = const()[name = string("variance_635_axes_0"), val = tensor([1])]; + bool variance_635_keep_dims_0 = const()[name = string("variance_635_keep_dims_0"), val = bool(true)]; + tensor variance_635_cast_fp16 = reduce_mean(axes = variance_635_axes_0, keep_dims = variance_635_keep_dims_0, x = inputs_sq_635_cast_fp16)[name = string("variance_635_cast_fp16")]; + fp16 var_23279_to_fp16 = const()[name = string("op_23279_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23280_cast_fp16 = add(x = variance_635_cast_fp16, y = var_23279_to_fp16)[name = string("op_23280_cast_fp16")]; + fp32 var_23281_epsilon_0 = const()[name = string("op_23281_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23281_cast_fp16 = rsqrt(epsilon = var_23281_epsilon_0, x = var_23280_cast_fp16)[name = string("op_23281_cast_fp16")]; + tensor hidden_states_785_cast_fp16 = mul(x = inputs_635_cast_fp16, y = var_23281_cast_fp16)[name = string("hidden_states_785_cast_fp16")]; + tensor input_659_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_785_cast_fp16)[name = string("input_659_cast_fp16")]; + string input_661_pad_type_0 = const()[name = string("input_661_pad_type_0"), val = string("valid")]; + tensor input_661_strides_0 = const()[name = string("input_661_strides_0"), val = tensor([1, 1])]; + tensor input_661_pad_0 = const()[name = string("input_661_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_661_dilations_0 = const()[name = string("input_661_dilations_0"), val = tensor([1, 1])]; + int32 input_661_groups_0 = const()[name = string("input_661_groups_0"), val = int32(1)]; + tensor input_661_cast_fp16 = conv(dilations = input_661_dilations_0, groups = input_661_groups_0, pad = input_661_pad_0, pad_type = input_661_pad_type_0, strides = input_661_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_659_cast_fp16)[name = string("input_661_cast_fp16")]; + tensor var_23295_cast_fp16 = silu(x = input_661_cast_fp16)[name = string("op_23295_cast_fp16")]; + string var_23301_pad_type_0 = const()[name = string("op_23301_pad_type_0"), val = string("valid")]; + tensor var_23301_strides_0 = const()[name = string("op_23301_strides_0"), val = tensor([1, 1])]; + tensor var_23301_pad_0 = const()[name = string("op_23301_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_23301_dilations_0 = const()[name = string("op_23301_dilations_0"), val = tensor([1, 1])]; + int32 var_23301_groups_0 = const()[name = string("op_23301_groups_0"), val = int32(1)]; + tensor var_23301_cast_fp16 = conv(dilations = var_23301_dilations_0, groups = var_23301_groups_0, pad = var_23301_pad_0, pad_type = var_23301_pad_type_0, strides = var_23301_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_659_cast_fp16)[name = string("op_23301_cast_fp16")]; + tensor input_663_cast_fp16 = mul(x = var_23295_cast_fp16, y = var_23301_cast_fp16)[name = string("input_663_cast_fp16")]; + string hidden_states_787_pad_type_0 = const()[name = string("hidden_states_787_pad_type_0"), val = string("valid")]; + tensor hidden_states_787_strides_0 = const()[name = string("hidden_states_787_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_787_pad_0 = const()[name = string("hidden_states_787_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_787_dilations_0 = const()[name = string("hidden_states_787_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_787_groups_0 = const()[name = string("hidden_states_787_groups_0"), val = int32(1)]; + tensor hidden_states_787_cast_fp16 = conv(dilations = hidden_states_787_dilations_0, groups = hidden_states_787_groups_0, pad = hidden_states_787_pad_0, pad_type = hidden_states_787_pad_type_0, strides = hidden_states_787_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_663_cast_fp16)[name = string("hidden_states_787_cast_fp16")]; + tensor inputs_637_cast_fp16 = add(x = inputs_635_cast_fp16, y = hidden_states_787_cast_fp16)[name = string("inputs_637_cast_fp16")]; + tensor obj_675_begin_0 = const()[name = string("obj_675_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_675_end_0 = const()[name = string("obj_675_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_675_end_mask_0 = const()[name = string("obj_675_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_675_cast_fp16 = slice_by_index(begin = obj_675_begin_0, end = obj_675_end_0, end_mask = obj_675_end_mask_0, x = key_caches_cast_fp16)[name = string("obj_675_cast_fp16")]; + tensor obj_677_begin_0 = const()[name = string("obj_677_begin_0"), val = tensor([0, 1024, 0, 0])]; + tensor obj_677_end_0 = const()[name = string("obj_677_end_0"), val = tensor([1, 2048, 1, 16])]; + tensor obj_677_end_mask_0 = const()[name = string("obj_677_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_677_cast_fp16 = slice_by_index(begin = obj_677_begin_0, end = obj_677_end_0, end_mask = obj_677_end_mask_0, x = value_caches_cast_fp16)[name = string("obj_677_cast_fp16")]; + int32 var_23335 = const()[name = string("op_23335"), val = int32(3)]; + int32 var_23345 = const()[name = string("op_23345"), val = int32(-2)]; + tensor inputs_sq_637_cast_fp16 = mul(x = inputs_637_cast_fp16, y = inputs_637_cast_fp16)[name = string("inputs_sq_637_cast_fp16")]; + tensor variance_637_axes_0 = const()[name = string("variance_637_axes_0"), val = tensor([1])]; + bool variance_637_keep_dims_0 = const()[name = string("variance_637_keep_dims_0"), val = bool(true)]; + tensor variance_637_cast_fp16 = reduce_mean(axes = variance_637_axes_0, keep_dims = variance_637_keep_dims_0, x = inputs_sq_637_cast_fp16)[name = string("variance_637_cast_fp16")]; + fp16 var_23359_to_fp16 = const()[name = string("op_23359_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23360_cast_fp16 = add(x = variance_637_cast_fp16, y = var_23359_to_fp16)[name = string("op_23360_cast_fp16")]; + fp32 var_23361_epsilon_0 = const()[name = string("op_23361_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23361_cast_fp16 = rsqrt(epsilon = var_23361_epsilon_0, x = var_23360_cast_fp16)[name = string("op_23361_cast_fp16")]; + tensor hidden_states_789_cast_fp16 = mul(x = inputs_637_cast_fp16, y = var_23361_cast_fp16)[name = string("hidden_states_789_cast_fp16")]; + tensor obj_673_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_789_cast_fp16)[name = string("obj_673_cast_fp16")]; + string query_457_pad_type_0 = const()[name = string("query_457_pad_type_0"), val = string("valid")]; + tensor query_457_strides_0 = const()[name = string("query_457_strides_0"), val = tensor([1, 1])]; + tensor query_457_pad_0 = const()[name = string("query_457_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_457_dilations_0 = const()[name = string("query_457_dilations_0"), val = tensor([1, 1])]; + int32 query_457_groups_0 = const()[name = string("query_457_groups_0"), val = int32(1)]; + tensor query_457_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_457_dilations_0, groups = query_457_groups_0, pad = query_457_pad_0, pad_type = query_457_pad_type_0, strides = query_457_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_673_cast_fp16)[name = string("query_457_cast_fp16")]; + string current_key_305_pad_type_0 = const()[name = string("current_key_305_pad_type_0"), val = string("valid")]; + tensor current_key_305_strides_0 = const()[name = string("current_key_305_strides_0"), val = tensor([1, 1])]; + tensor current_key_305_pad_0 = const()[name = string("current_key_305_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_305_dilations_0 = const()[name = string("current_key_305_dilations_0"), val = tensor([1, 1])]; + int32 current_key_305_groups_0 = const()[name = string("current_key_305_groups_0"), val = int32(1)]; + tensor current_key_305_cast_fp16 = conv(dilations = current_key_305_dilations_0, groups = current_key_305_groups_0, pad = current_key_305_pad_0, pad_type = current_key_305_pad_type_0, strides = current_key_305_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_673_cast_fp16)[name = string("current_key_305_cast_fp16")]; + string current_value_153_pad_type_0 = const()[name = string("current_value_153_pad_type_0"), val = string("valid")]; + tensor current_value_153_strides_0 = const()[name = string("current_value_153_strides_0"), val = tensor([1, 1])]; + tensor current_value_153_pad_0 = const()[name = string("current_value_153_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_153_dilations_0 = const()[name = string("current_value_153_dilations_0"), val = tensor([1, 1])]; + int32 current_value_153_groups_0 = const()[name = string("current_value_153_groups_0"), val = int32(1)]; + tensor current_value_153_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_153_dilations_0, groups = current_value_153_groups_0, pad = current_value_153_pad_0, pad_type = current_value_153_pad_type_0, strides = current_value_153_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_673_cast_fp16)[name = string("current_value_153_cast_fp16")]; + tensor var_23398 = const()[name = string("op_23398"), val = tensor([16, 128, 1, 1])]; + tensor inputs_639_cast_fp16 = reshape(shape = var_23398, x = query_457_cast_fp16)[name = string("inputs_639_cast_fp16")]; + tensor inputs_sq_639_cast_fp16 = mul(x = inputs_639_cast_fp16, y = inputs_639_cast_fp16)[name = string("inputs_sq_639_cast_fp16")]; + tensor variance_639_axes_0 = const()[name = string("variance_639_axes_0"), val = tensor([1])]; + bool variance_639_keep_dims_0 = const()[name = string("variance_639_keep_dims_0"), val = bool(true)]; + tensor variance_639_cast_fp16 = reduce_mean(axes = variance_639_axes_0, keep_dims = variance_639_keep_dims_0, x = inputs_sq_639_cast_fp16)[name = string("variance_639_cast_fp16")]; + fp16 var_23404_to_fp16 = const()[name = string("op_23404_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23405_cast_fp16 = add(x = variance_639_cast_fp16, y = var_23404_to_fp16)[name = string("op_23405_cast_fp16")]; + fp32 var_23406_epsilon_0 = const()[name = string("op_23406_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23406_cast_fp16 = rsqrt(epsilon = var_23406_epsilon_0, x = var_23405_cast_fp16)[name = string("op_23406_cast_fp16")]; + tensor hidden_states_791_cast_fp16 = mul(x = inputs_639_cast_fp16, y = var_23406_cast_fp16)[name = string("hidden_states_791_cast_fp16")]; + tensor query_normed_153_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_791_cast_fp16)[name = string("query_normed_153_cast_fp16")]; + tensor var_23414 = const()[name = string("op_23414"), val = tensor([8, 128, 1, 1])]; + tensor inputs_641_cast_fp16 = reshape(shape = var_23414, x = current_key_305_cast_fp16)[name = string("inputs_641_cast_fp16")]; + tensor inputs_sq_641_cast_fp16 = mul(x = inputs_641_cast_fp16, y = inputs_641_cast_fp16)[name = string("inputs_sq_641_cast_fp16")]; + tensor variance_641_axes_0 = const()[name = string("variance_641_axes_0"), val = tensor([1])]; + bool variance_641_keep_dims_0 = const()[name = string("variance_641_keep_dims_0"), val = bool(true)]; + tensor variance_641_cast_fp16 = reduce_mean(axes = variance_641_axes_0, keep_dims = variance_641_keep_dims_0, x = inputs_sq_641_cast_fp16)[name = string("variance_641_cast_fp16")]; + fp16 var_23420_to_fp16 = const()[name = string("op_23420_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23421_cast_fp16 = add(x = variance_641_cast_fp16, y = var_23420_to_fp16)[name = string("op_23421_cast_fp16")]; + fp32 var_23422_epsilon_0 = const()[name = string("op_23422_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23422_cast_fp16 = rsqrt(epsilon = var_23422_epsilon_0, x = var_23421_cast_fp16)[name = string("op_23422_cast_fp16")]; + tensor hidden_states_793_cast_fp16 = mul(x = inputs_641_cast_fp16, y = var_23422_cast_fp16)[name = string("hidden_states_793_cast_fp16")]; + tensor current_key_normed_153_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_793_cast_fp16)[name = string("current_key_normed_153_cast_fp16")]; + tensor var_23440 = const()[name = string("op_23440"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_609_cast_fp16 = reshape(shape = var_23440, x = query_normed_153_cast_fp16)[name = string("mh_q_609_cast_fp16")]; + tensor var_23442 = const()[name = string("op_23442"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_609_cast_fp16 = reshape(shape = var_23442, x = current_key_normed_153_cast_fp16)[name = string("mh_k_609_cast_fp16")]; + tensor var_23446_cast_fp16 = mul(x = mh_q_609_cast_fp16, y = cos_151_to_fp16)[name = string("op_23446_cast_fp16")]; + tensor var_23451_begin_0 = const()[name = string("op_23451_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_23451_end_0 = const()[name = string("op_23451_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_23451_end_mask_0 = const()[name = string("op_23451_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_23451_cast_fp16 = slice_by_index(begin = var_23451_begin_0, end = var_23451_end_0, end_mask = var_23451_end_mask_0, x = mh_q_609_cast_fp16)[name = string("op_23451_cast_fp16")]; + tensor var_23457_begin_0 = const()[name = string("op_23457_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_23457_end_0 = const()[name = string("op_23457_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_23457_end_mask_0 = const()[name = string("op_23457_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_23457_cast_fp16 = slice_by_index(begin = var_23457_begin_0, end = var_23457_end_0, end_mask = var_23457_end_mask_0, x = mh_q_609_cast_fp16)[name = string("op_23457_cast_fp16")]; + fp16 const_1549_promoted_to_fp16 = const()[name = string("const_1549_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_23459_cast_fp16 = mul(x = var_23457_cast_fp16, y = const_1549_promoted_to_fp16)[name = string("op_23459_cast_fp16")]; + bool var_23461_interleave_0 = const()[name = string("op_23461_interleave_0"), val = bool(false)]; + tensor var_23461_cast_fp16 = concat(axis = var_23345, interleave = var_23461_interleave_0, values = (var_23459_cast_fp16, var_23451_cast_fp16))[name = string("op_23461_cast_fp16")]; + tensor var_23462_cast_fp16 = mul(x = var_23461_cast_fp16, y = sin_151_to_fp16)[name = string("op_23462_cast_fp16")]; + tensor mh_q_611_cast_fp16 = add(x = var_23446_cast_fp16, y = var_23462_cast_fp16)[name = string("mh_q_611_cast_fp16")]; + tensor var_23464_cast_fp16 = mul(x = mh_k_609_cast_fp16, y = cos_151_to_fp16)[name = string("op_23464_cast_fp16")]; + tensor var_23469_begin_0 = const()[name = string("op_23469_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_23469_end_0 = const()[name = string("op_23469_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_23469_end_mask_0 = const()[name = string("op_23469_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_23469_cast_fp16 = slice_by_index(begin = var_23469_begin_0, end = var_23469_end_0, end_mask = var_23469_end_mask_0, x = mh_k_609_cast_fp16)[name = string("op_23469_cast_fp16")]; + tensor var_23475_begin_0 = const()[name = string("op_23475_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_23475_end_0 = const()[name = string("op_23475_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_23475_end_mask_0 = const()[name = string("op_23475_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_23475_cast_fp16 = slice_by_index(begin = var_23475_begin_0, end = var_23475_end_0, end_mask = var_23475_end_mask_0, x = mh_k_609_cast_fp16)[name = string("op_23475_cast_fp16")]; + fp16 const_1552_promoted_to_fp16 = const()[name = string("const_1552_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_23477_cast_fp16 = mul(x = var_23475_cast_fp16, y = const_1552_promoted_to_fp16)[name = string("op_23477_cast_fp16")]; + bool var_23479_interleave_0 = const()[name = string("op_23479_interleave_0"), val = bool(false)]; + tensor var_23479_cast_fp16 = concat(axis = var_23345, interleave = var_23479_interleave_0, values = (var_23477_cast_fp16, var_23469_cast_fp16))[name = string("op_23479_cast_fp16")]; + tensor var_23480_cast_fp16 = mul(x = var_23479_cast_fp16, y = sin_151_to_fp16)[name = string("op_23480_cast_fp16")]; + tensor mh_k_611_cast_fp16 = add(x = var_23464_cast_fp16, y = var_23480_cast_fp16)[name = string("mh_k_611_cast_fp16")]; + tensor var_23484 = const()[name = string("op_23484"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_307_cast_fp16 = reshape(shape = var_23484, x = mh_k_611_cast_fp16)[name = string("current_key_307_cast_fp16")]; + tensor var_23490_to_fp16 = const()[name = string("op_23490_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204544)))]; + tensor var_23491_cast_fp16 = mul(x = obj_675_cast_fp16, y = var_23490_to_fp16)[name = string("op_23491_cast_fp16")]; + tensor var_23488_to_fp16 = const()[name = string("op_23488_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204672)))]; + tensor var_23492_cast_fp16 = mul(x = current_key_307_cast_fp16, y = var_23488_to_fp16)[name = string("op_23492_cast_fp16")]; + tensor key_307_cast_fp16 = add(x = var_23491_cast_fp16, y = var_23492_cast_fp16)[name = string("key_307_cast_fp16")]; + tensor var_23494_to_fp16 = const()[name = string("op_23494_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204544)))]; + tensor var_23495_cast_fp16 = mul(x = obj_677_cast_fp16, y = var_23494_to_fp16)[name = string("op_23495_cast_fp16")]; + tensor var_23496_cast_fp16 = mul(x = current_value_153_cast_fp16, y = var_23488_to_fp16)[name = string("op_23496_cast_fp16")]; + tensor value_153_cast_fp16 = add(x = var_23495_cast_fp16, y = var_23496_cast_fp16)[name = string("value_153_cast_fp16")]; + fp16 var_23503_to_fp16 = const()[name = string("op_23503_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_615_cast_fp16 = mul(x = mh_q_611_cast_fp16, y = var_23503_to_fp16)[name = string("mh_q_615_cast_fp16")]; + tensor var_23505 = const()[name = string("op_23505"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_613_cast_fp16 = reshape(shape = var_23505, x = key_307_cast_fp16)[name = string("mh_k_613_cast_fp16")]; + tensor var_23507 = const()[name = string("op_23507"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_305_cast_fp16 = reshape(shape = var_23507, x = value_153_cast_fp16)[name = string("mh_v_305_cast_fp16")]; + tensor transpose_304_perm_0 = const()[name = string("transpose_304_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_152_reps_0 = const()[name = string("tile_152_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_304_cast_fp16 = transpose(perm = transpose_304_perm_0, x = mh_k_613_cast_fp16)[name = string("transpose_23")]; + tensor tile_152_cast_fp16 = tile(reps = tile_152_reps_0, x = transpose_304_cast_fp16)[name = string("tile_152_cast_fp16")]; + tensor concat_382 = const()[name = string("concat_382"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_304_cast_fp16 = reshape(shape = concat_382, x = tile_152_cast_fp16)[name = string("reshape_304_cast_fp16")]; + tensor transpose_305_perm_0 = const()[name = string("transpose_305_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_383 = const()[name = string("concat_383"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_305_cast_fp16 = transpose(perm = transpose_305_perm_0, x = reshape_304_cast_fp16)[name = string("transpose_22")]; + tensor reshape_305_cast_fp16 = reshape(shape = concat_383, x = transpose_305_cast_fp16)[name = string("reshape_305_cast_fp16")]; + tensor transpose_306_perm_0 = const()[name = string("transpose_306_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_153_reps_0 = const()[name = string("tile_153_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_306_cast_fp16 = transpose(perm = transpose_306_perm_0, x = mh_v_305_cast_fp16)[name = string("transpose_21")]; + tensor tile_153_cast_fp16 = tile(reps = tile_153_reps_0, x = transpose_306_cast_fp16)[name = string("tile_153_cast_fp16")]; + tensor concat_384 = const()[name = string("concat_384"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_306_cast_fp16 = reshape(shape = concat_384, x = tile_153_cast_fp16)[name = string("reshape_306_cast_fp16")]; + tensor transpose_307_perm_0 = const()[name = string("transpose_307_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_385 = const()[name = string("concat_385"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_307_cast_fp16 = transpose(perm = transpose_307_perm_0, x = reshape_306_cast_fp16)[name = string("transpose_20")]; + tensor reshape_307_cast_fp16 = reshape(shape = concat_385, x = transpose_307_cast_fp16)[name = string("reshape_307_cast_fp16")]; + tensor transpose_621_perm_0 = const()[name = string("transpose_621_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_457_transpose_x_1 = const()[name = string("mh_w_457_transpose_x_1"), val = bool(true)]; + bool mh_w_457_transpose_y_1 = const()[name = string("mh_w_457_transpose_y_1"), val = bool(false)]; + tensor transpose_621_cast_fp16 = transpose(perm = transpose_621_perm_0, x = reshape_305_cast_fp16)[name = string("transpose_19")]; + tensor mh_w_457_cast_fp16 = matmul(transpose_x = mh_w_457_transpose_x_1, transpose_y = mh_w_457_transpose_y_1, x = mh_q_615_cast_fp16, y = transpose_621_cast_fp16)[name = string("mh_w_457_cast_fp16")]; + tensor mh_w_461_cast_fp16 = softmax(axis = var_23335, x = mh_w_457_cast_fp16)[name = string("mh_w_461_cast_fp16")]; + tensor transpose_622_perm_0 = const()[name = string("transpose_622_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_153_transpose_x_1 = const()[name = string("attn_153_transpose_x_1"), val = bool(false)]; + bool attn_153_transpose_y_1 = const()[name = string("attn_153_transpose_y_1"), val = bool(true)]; + tensor transpose_622_cast_fp16 = transpose(perm = transpose_622_perm_0, x = reshape_307_cast_fp16)[name = string("transpose_18")]; + tensor attn_153_cast_fp16 = matmul(transpose_x = attn_153_transpose_x_1, transpose_y = attn_153_transpose_y_1, x = transpose_622_cast_fp16, y = mh_w_461_cast_fp16)[name = string("attn_153_cast_fp16")]; + tensor var_23521 = const()[name = string("op_23521"), val = tensor([1, 2048, 1, 1])]; + tensor input_665_cast_fp16 = reshape(shape = var_23521, x = attn_153_cast_fp16)[name = string("input_665_cast_fp16")]; + string obj_679_pad_type_0 = const()[name = string("obj_679_pad_type_0"), val = string("valid")]; + tensor obj_679_strides_0 = const()[name = string("obj_679_strides_0"), val = tensor([1, 1])]; + tensor obj_679_pad_0 = const()[name = string("obj_679_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_679_dilations_0 = const()[name = string("obj_679_dilations_0"), val = tensor([1, 1])]; + int32 obj_679_groups_0 = const()[name = string("obj_679_groups_0"), val = int32(1)]; + tensor obj_679_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_679_dilations_0, groups = obj_679_groups_0, pad = obj_679_pad_0, pad_type = obj_679_pad_type_0, strides = obj_679_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_665_cast_fp16)[name = string("obj_679_cast_fp16")]; + tensor inputs_643_cast_fp16 = add(x = inputs_637_cast_fp16, y = obj_679_cast_fp16)[name = string("inputs_643_cast_fp16")]; + tensor inputs_sq_643_cast_fp16 = mul(x = inputs_643_cast_fp16, y = inputs_643_cast_fp16)[name = string("inputs_sq_643_cast_fp16")]; + tensor variance_643_axes_0 = const()[name = string("variance_643_axes_0"), val = tensor([1])]; + bool variance_643_keep_dims_0 = const()[name = string("variance_643_keep_dims_0"), val = bool(true)]; + tensor variance_643_cast_fp16 = reduce_mean(axes = variance_643_axes_0, keep_dims = variance_643_keep_dims_0, x = inputs_sq_643_cast_fp16)[name = string("variance_643_cast_fp16")]; + fp16 var_23539_to_fp16 = const()[name = string("op_23539_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23540_cast_fp16 = add(x = variance_643_cast_fp16, y = var_23539_to_fp16)[name = string("op_23540_cast_fp16")]; + fp32 var_23541_epsilon_0 = const()[name = string("op_23541_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23541_cast_fp16 = rsqrt(epsilon = var_23541_epsilon_0, x = var_23540_cast_fp16)[name = string("op_23541_cast_fp16")]; + tensor hidden_states_795_cast_fp16 = mul(x = inputs_643_cast_fp16, y = var_23541_cast_fp16)[name = string("hidden_states_795_cast_fp16")]; + tensor input_667_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_795_cast_fp16)[name = string("input_667_cast_fp16")]; + string input_669_pad_type_0 = const()[name = string("input_669_pad_type_0"), val = string("valid")]; + tensor input_669_strides_0 = const()[name = string("input_669_strides_0"), val = tensor([1, 1])]; + tensor input_669_pad_0 = const()[name = string("input_669_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_669_dilations_0 = const()[name = string("input_669_dilations_0"), val = tensor([1, 1])]; + int32 input_669_groups_0 = const()[name = string("input_669_groups_0"), val = int32(1)]; + tensor input_669_cast_fp16 = conv(dilations = input_669_dilations_0, groups = input_669_groups_0, pad = input_669_pad_0, pad_type = input_669_pad_type_0, strides = input_669_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_667_cast_fp16)[name = string("input_669_cast_fp16")]; + tensor var_23555_cast_fp16 = silu(x = input_669_cast_fp16)[name = string("op_23555_cast_fp16")]; + string var_23561_pad_type_0 = const()[name = string("op_23561_pad_type_0"), val = string("valid")]; + tensor var_23561_strides_0 = const()[name = string("op_23561_strides_0"), val = tensor([1, 1])]; + tensor var_23561_pad_0 = const()[name = string("op_23561_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_23561_dilations_0 = const()[name = string("op_23561_dilations_0"), val = tensor([1, 1])]; + int32 var_23561_groups_0 = const()[name = string("op_23561_groups_0"), val = int32(1)]; + tensor var_23561_cast_fp16 = conv(dilations = var_23561_dilations_0, groups = var_23561_groups_0, pad = var_23561_pad_0, pad_type = var_23561_pad_type_0, strides = var_23561_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_667_cast_fp16)[name = string("op_23561_cast_fp16")]; + tensor input_671_cast_fp16 = mul(x = var_23555_cast_fp16, y = var_23561_cast_fp16)[name = string("input_671_cast_fp16")]; + string hidden_states_797_pad_type_0 = const()[name = string("hidden_states_797_pad_type_0"), val = string("valid")]; + tensor hidden_states_797_strides_0 = const()[name = string("hidden_states_797_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_797_pad_0 = const()[name = string("hidden_states_797_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_797_dilations_0 = const()[name = string("hidden_states_797_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_797_groups_0 = const()[name = string("hidden_states_797_groups_0"), val = int32(1)]; + tensor hidden_states_797_cast_fp16 = conv(dilations = hidden_states_797_dilations_0, groups = hidden_states_797_groups_0, pad = hidden_states_797_pad_0, pad_type = hidden_states_797_pad_type_0, strides = hidden_states_797_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_671_cast_fp16)[name = string("hidden_states_797_cast_fp16")]; + tensor inputs_645_cast_fp16 = add(x = inputs_643_cast_fp16, y = hidden_states_797_cast_fp16)[name = string("inputs_645_cast_fp16")]; + tensor obj_683_begin_0 = const()[name = string("obj_683_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_683_end_0 = const()[name = string("obj_683_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_683_end_mask_0 = const()[name = string("obj_683_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_683_cast_fp16 = slice_by_index(begin = obj_683_begin_0, end = obj_683_end_0, end_mask = obj_683_end_mask_0, x = key_caches_cast_fp16)[name = string("obj_683_cast_fp16")]; + tensor obj_685_begin_0 = const()[name = string("obj_685_begin_0"), val = tensor([0, 2048, 0, 0])]; + tensor obj_685_end_0 = const()[name = string("obj_685_end_0"), val = tensor([1, 3072, 1, 16])]; + tensor obj_685_end_mask_0 = const()[name = string("obj_685_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_685_cast_fp16 = slice_by_index(begin = obj_685_begin_0, end = obj_685_end_0, end_mask = obj_685_end_mask_0, x = value_caches_cast_fp16)[name = string("obj_685_cast_fp16")]; + int32 var_23595 = const()[name = string("op_23595"), val = int32(3)]; + int32 var_23605 = const()[name = string("op_23605"), val = int32(-2)]; + tensor inputs_sq_645_cast_fp16 = mul(x = inputs_645_cast_fp16, y = inputs_645_cast_fp16)[name = string("inputs_sq_645_cast_fp16")]; + tensor variance_645_axes_0 = const()[name = string("variance_645_axes_0"), val = tensor([1])]; + bool variance_645_keep_dims_0 = const()[name = string("variance_645_keep_dims_0"), val = bool(true)]; + tensor variance_645_cast_fp16 = reduce_mean(axes = variance_645_axes_0, keep_dims = variance_645_keep_dims_0, x = inputs_sq_645_cast_fp16)[name = string("variance_645_cast_fp16")]; + fp16 var_23619_to_fp16 = const()[name = string("op_23619_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23620_cast_fp16 = add(x = variance_645_cast_fp16, y = var_23619_to_fp16)[name = string("op_23620_cast_fp16")]; + fp32 var_23621_epsilon_0 = const()[name = string("op_23621_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23621_cast_fp16 = rsqrt(epsilon = var_23621_epsilon_0, x = var_23620_cast_fp16)[name = string("op_23621_cast_fp16")]; + tensor hidden_states_799_cast_fp16 = mul(x = inputs_645_cast_fp16, y = var_23621_cast_fp16)[name = string("hidden_states_799_cast_fp16")]; + tensor obj_681_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_799_cast_fp16)[name = string("obj_681_cast_fp16")]; + string query_463_pad_type_0 = const()[name = string("query_463_pad_type_0"), val = string("valid")]; + tensor query_463_strides_0 = const()[name = string("query_463_strides_0"), val = tensor([1, 1])]; + tensor query_463_pad_0 = const()[name = string("query_463_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_463_dilations_0 = const()[name = string("query_463_dilations_0"), val = tensor([1, 1])]; + int32 query_463_groups_0 = const()[name = string("query_463_groups_0"), val = int32(1)]; + tensor query_463_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_463_dilations_0, groups = query_463_groups_0, pad = query_463_pad_0, pad_type = query_463_pad_type_0, strides = query_463_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_681_cast_fp16)[name = string("query_463_cast_fp16")]; + string current_key_309_pad_type_0 = const()[name = string("current_key_309_pad_type_0"), val = string("valid")]; + tensor current_key_309_strides_0 = const()[name = string("current_key_309_strides_0"), val = tensor([1, 1])]; + tensor current_key_309_pad_0 = const()[name = string("current_key_309_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_309_dilations_0 = const()[name = string("current_key_309_dilations_0"), val = tensor([1, 1])]; + int32 current_key_309_groups_0 = const()[name = string("current_key_309_groups_0"), val = int32(1)]; + tensor current_key_309_cast_fp16 = conv(dilations = current_key_309_dilations_0, groups = current_key_309_groups_0, pad = current_key_309_pad_0, pad_type = current_key_309_pad_type_0, strides = current_key_309_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_681_cast_fp16)[name = string("current_key_309_cast_fp16")]; + string current_value_155_pad_type_0 = const()[name = string("current_value_155_pad_type_0"), val = string("valid")]; + tensor current_value_155_strides_0 = const()[name = string("current_value_155_strides_0"), val = tensor([1, 1])]; + tensor current_value_155_pad_0 = const()[name = string("current_value_155_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_155_dilations_0 = const()[name = string("current_value_155_dilations_0"), val = tensor([1, 1])]; + int32 current_value_155_groups_0 = const()[name = string("current_value_155_groups_0"), val = int32(1)]; + tensor current_value_155_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_155_dilations_0, groups = current_value_155_groups_0, pad = current_value_155_pad_0, pad_type = current_value_155_pad_type_0, strides = current_value_155_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_681_cast_fp16)[name = string("current_value_155_cast_fp16")]; + tensor var_23658 = const()[name = string("op_23658"), val = tensor([16, 128, 1, 1])]; + tensor inputs_647_cast_fp16 = reshape(shape = var_23658, x = query_463_cast_fp16)[name = string("inputs_647_cast_fp16")]; + tensor inputs_sq_647_cast_fp16 = mul(x = inputs_647_cast_fp16, y = inputs_647_cast_fp16)[name = string("inputs_sq_647_cast_fp16")]; + tensor variance_647_axes_0 = const()[name = string("variance_647_axes_0"), val = tensor([1])]; + bool variance_647_keep_dims_0 = const()[name = string("variance_647_keep_dims_0"), val = bool(true)]; + tensor variance_647_cast_fp16 = reduce_mean(axes = variance_647_axes_0, keep_dims = variance_647_keep_dims_0, x = inputs_sq_647_cast_fp16)[name = string("variance_647_cast_fp16")]; + fp16 var_23664_to_fp16 = const()[name = string("op_23664_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23665_cast_fp16 = add(x = variance_647_cast_fp16, y = var_23664_to_fp16)[name = string("op_23665_cast_fp16")]; + fp32 var_23666_epsilon_0 = const()[name = string("op_23666_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23666_cast_fp16 = rsqrt(epsilon = var_23666_epsilon_0, x = var_23665_cast_fp16)[name = string("op_23666_cast_fp16")]; + tensor hidden_states_801_cast_fp16 = mul(x = inputs_647_cast_fp16, y = var_23666_cast_fp16)[name = string("hidden_states_801_cast_fp16")]; + tensor query_normed_155_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_801_cast_fp16)[name = string("query_normed_155_cast_fp16")]; + tensor var_23674 = const()[name = string("op_23674"), val = tensor([8, 128, 1, 1])]; + tensor inputs_649_cast_fp16 = reshape(shape = var_23674, x = current_key_309_cast_fp16)[name = string("inputs_649_cast_fp16")]; + tensor inputs_sq_649_cast_fp16 = mul(x = inputs_649_cast_fp16, y = inputs_649_cast_fp16)[name = string("inputs_sq_649_cast_fp16")]; + tensor variance_649_axes_0 = const()[name = string("variance_649_axes_0"), val = tensor([1])]; + bool variance_649_keep_dims_0 = const()[name = string("variance_649_keep_dims_0"), val = bool(true)]; + tensor variance_649_cast_fp16 = reduce_mean(axes = variance_649_axes_0, keep_dims = variance_649_keep_dims_0, x = inputs_sq_649_cast_fp16)[name = string("variance_649_cast_fp16")]; + fp16 var_23680_to_fp16 = const()[name = string("op_23680_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23681_cast_fp16 = add(x = variance_649_cast_fp16, y = var_23680_to_fp16)[name = string("op_23681_cast_fp16")]; + fp32 var_23682_epsilon_0 = const()[name = string("op_23682_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23682_cast_fp16 = rsqrt(epsilon = var_23682_epsilon_0, x = var_23681_cast_fp16)[name = string("op_23682_cast_fp16")]; + tensor hidden_states_803_cast_fp16 = mul(x = inputs_649_cast_fp16, y = var_23682_cast_fp16)[name = string("hidden_states_803_cast_fp16")]; + tensor current_key_normed_155_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_803_cast_fp16)[name = string("current_key_normed_155_cast_fp16")]; + tensor var_23700 = const()[name = string("op_23700"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_617_cast_fp16 = reshape(shape = var_23700, x = query_normed_155_cast_fp16)[name = string("mh_q_617_cast_fp16")]; + tensor var_23702 = const()[name = string("op_23702"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_617_cast_fp16 = reshape(shape = var_23702, x = current_key_normed_155_cast_fp16)[name = string("mh_k_617_cast_fp16")]; + tensor var_23706_cast_fp16 = mul(x = mh_q_617_cast_fp16, y = cos_151_to_fp16)[name = string("op_23706_cast_fp16")]; + tensor var_23711_begin_0 = const()[name = string("op_23711_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_23711_end_0 = const()[name = string("op_23711_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_23711_end_mask_0 = const()[name = string("op_23711_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_23711_cast_fp16 = slice_by_index(begin = var_23711_begin_0, end = var_23711_end_0, end_mask = var_23711_end_mask_0, x = mh_q_617_cast_fp16)[name = string("op_23711_cast_fp16")]; + tensor var_23717_begin_0 = const()[name = string("op_23717_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_23717_end_0 = const()[name = string("op_23717_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_23717_end_mask_0 = const()[name = string("op_23717_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_23717_cast_fp16 = slice_by_index(begin = var_23717_begin_0, end = var_23717_end_0, end_mask = var_23717_end_mask_0, x = mh_q_617_cast_fp16)[name = string("op_23717_cast_fp16")]; + fp16 const_1569_promoted_to_fp16 = const()[name = string("const_1569_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_23719_cast_fp16 = mul(x = var_23717_cast_fp16, y = const_1569_promoted_to_fp16)[name = string("op_23719_cast_fp16")]; + bool var_23721_interleave_0 = const()[name = string("op_23721_interleave_0"), val = bool(false)]; + tensor var_23721_cast_fp16 = concat(axis = var_23605, interleave = var_23721_interleave_0, values = (var_23719_cast_fp16, var_23711_cast_fp16))[name = string("op_23721_cast_fp16")]; + tensor var_23722_cast_fp16 = mul(x = var_23721_cast_fp16, y = sin_151_to_fp16)[name = string("op_23722_cast_fp16")]; + tensor mh_q_619_cast_fp16 = add(x = var_23706_cast_fp16, y = var_23722_cast_fp16)[name = string("mh_q_619_cast_fp16")]; + tensor var_23724_cast_fp16 = mul(x = mh_k_617_cast_fp16, y = cos_151_to_fp16)[name = string("op_23724_cast_fp16")]; + tensor var_23729_begin_0 = const()[name = string("op_23729_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_23729_end_0 = const()[name = string("op_23729_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_23729_end_mask_0 = const()[name = string("op_23729_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_23729_cast_fp16 = slice_by_index(begin = var_23729_begin_0, end = var_23729_end_0, end_mask = var_23729_end_mask_0, x = mh_k_617_cast_fp16)[name = string("op_23729_cast_fp16")]; + tensor var_23735_begin_0 = const()[name = string("op_23735_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_23735_end_0 = const()[name = string("op_23735_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_23735_end_mask_0 = const()[name = string("op_23735_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_23735_cast_fp16 = slice_by_index(begin = var_23735_begin_0, end = var_23735_end_0, end_mask = var_23735_end_mask_0, x = mh_k_617_cast_fp16)[name = string("op_23735_cast_fp16")]; + fp16 const_1572_promoted_to_fp16 = const()[name = string("const_1572_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_23737_cast_fp16 = mul(x = var_23735_cast_fp16, y = const_1572_promoted_to_fp16)[name = string("op_23737_cast_fp16")]; + bool var_23739_interleave_0 = const()[name = string("op_23739_interleave_0"), val = bool(false)]; + tensor var_23739_cast_fp16 = concat(axis = var_23605, interleave = var_23739_interleave_0, values = (var_23737_cast_fp16, var_23729_cast_fp16))[name = string("op_23739_cast_fp16")]; + tensor var_23740_cast_fp16 = mul(x = var_23739_cast_fp16, y = sin_151_to_fp16)[name = string("op_23740_cast_fp16")]; + tensor mh_k_619_cast_fp16 = add(x = var_23724_cast_fp16, y = var_23740_cast_fp16)[name = string("mh_k_619_cast_fp16")]; + tensor var_23744 = const()[name = string("op_23744"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_311_cast_fp16 = reshape(shape = var_23744, x = mh_k_619_cast_fp16)[name = string("current_key_311_cast_fp16")]; + tensor var_23750_to_fp16 = const()[name = string("op_23750_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204544)))]; + tensor var_23751_cast_fp16 = mul(x = obj_683_cast_fp16, y = var_23750_to_fp16)[name = string("op_23751_cast_fp16")]; + tensor var_23748_to_fp16 = const()[name = string("op_23748_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204672)))]; + tensor var_23752_cast_fp16 = mul(x = current_key_311_cast_fp16, y = var_23748_to_fp16)[name = string("op_23752_cast_fp16")]; + tensor key_311_cast_fp16 = add(x = var_23751_cast_fp16, y = var_23752_cast_fp16)[name = string("key_311_cast_fp16")]; + tensor var_23754_to_fp16 = const()[name = string("op_23754_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204544)))]; + tensor var_23755_cast_fp16 = mul(x = obj_685_cast_fp16, y = var_23754_to_fp16)[name = string("op_23755_cast_fp16")]; + tensor var_23756_cast_fp16 = mul(x = current_value_155_cast_fp16, y = var_23748_to_fp16)[name = string("op_23756_cast_fp16")]; + tensor value_155_cast_fp16 = add(x = var_23755_cast_fp16, y = var_23756_cast_fp16)[name = string("value_155_cast_fp16")]; + fp16 var_23763_to_fp16 = const()[name = string("op_23763_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_623_cast_fp16 = mul(x = mh_q_619_cast_fp16, y = var_23763_to_fp16)[name = string("mh_q_623_cast_fp16")]; + tensor var_23765 = const()[name = string("op_23765"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_621_cast_fp16 = reshape(shape = var_23765, x = key_311_cast_fp16)[name = string("mh_k_621_cast_fp16")]; + tensor var_23767 = const()[name = string("op_23767"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_309_cast_fp16 = reshape(shape = var_23767, x = value_155_cast_fp16)[name = string("mh_v_309_cast_fp16")]; + tensor transpose_308_perm_0 = const()[name = string("transpose_308_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_154_reps_0 = const()[name = string("tile_154_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_308_cast_fp16 = transpose(perm = transpose_308_perm_0, x = mh_k_621_cast_fp16)[name = string("transpose_17")]; + tensor tile_154_cast_fp16 = tile(reps = tile_154_reps_0, x = transpose_308_cast_fp16)[name = string("tile_154_cast_fp16")]; + tensor concat_386 = const()[name = string("concat_386"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_308_cast_fp16 = reshape(shape = concat_386, x = tile_154_cast_fp16)[name = string("reshape_308_cast_fp16")]; + tensor transpose_309_perm_0 = const()[name = string("transpose_309_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_387 = const()[name = string("concat_387"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_309_cast_fp16 = transpose(perm = transpose_309_perm_0, x = reshape_308_cast_fp16)[name = string("transpose_16")]; + tensor reshape_309_cast_fp16 = reshape(shape = concat_387, x = transpose_309_cast_fp16)[name = string("reshape_309_cast_fp16")]; + tensor transpose_310_perm_0 = const()[name = string("transpose_310_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_155_reps_0 = const()[name = string("tile_155_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_310_cast_fp16 = transpose(perm = transpose_310_perm_0, x = mh_v_309_cast_fp16)[name = string("transpose_15")]; + tensor tile_155_cast_fp16 = tile(reps = tile_155_reps_0, x = transpose_310_cast_fp16)[name = string("tile_155_cast_fp16")]; + tensor concat_388 = const()[name = string("concat_388"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_310_cast_fp16 = reshape(shape = concat_388, x = tile_155_cast_fp16)[name = string("reshape_310_cast_fp16")]; + tensor transpose_311_perm_0 = const()[name = string("transpose_311_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_389 = const()[name = string("concat_389"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_311_cast_fp16 = transpose(perm = transpose_311_perm_0, x = reshape_310_cast_fp16)[name = string("transpose_14")]; + tensor reshape_311_cast_fp16 = reshape(shape = concat_389, x = transpose_311_cast_fp16)[name = string("reshape_311_cast_fp16")]; + tensor transpose_625_perm_0 = const()[name = string("transpose_625_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_463_transpose_x_1 = const()[name = string("mh_w_463_transpose_x_1"), val = bool(true)]; + bool mh_w_463_transpose_y_1 = const()[name = string("mh_w_463_transpose_y_1"), val = bool(false)]; + tensor transpose_625_cast_fp16 = transpose(perm = transpose_625_perm_0, x = reshape_309_cast_fp16)[name = string("transpose_13")]; + tensor mh_w_463_cast_fp16 = matmul(transpose_x = mh_w_463_transpose_x_1, transpose_y = mh_w_463_transpose_y_1, x = mh_q_623_cast_fp16, y = transpose_625_cast_fp16)[name = string("mh_w_463_cast_fp16")]; + tensor mh_w_467_cast_fp16 = softmax(axis = var_23595, x = mh_w_463_cast_fp16)[name = string("mh_w_467_cast_fp16")]; + tensor transpose_626_perm_0 = const()[name = string("transpose_626_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_155_transpose_x_1 = const()[name = string("attn_155_transpose_x_1"), val = bool(false)]; + bool attn_155_transpose_y_1 = const()[name = string("attn_155_transpose_y_1"), val = bool(true)]; + tensor transpose_626_cast_fp16 = transpose(perm = transpose_626_perm_0, x = reshape_311_cast_fp16)[name = string("transpose_12")]; + tensor attn_155_cast_fp16 = matmul(transpose_x = attn_155_transpose_x_1, transpose_y = attn_155_transpose_y_1, x = transpose_626_cast_fp16, y = mh_w_467_cast_fp16)[name = string("attn_155_cast_fp16")]; + tensor var_23781 = const()[name = string("op_23781"), val = tensor([1, 2048, 1, 1])]; + tensor input_673_cast_fp16 = reshape(shape = var_23781, x = attn_155_cast_fp16)[name = string("input_673_cast_fp16")]; + string obj_687_pad_type_0 = const()[name = string("obj_687_pad_type_0"), val = string("valid")]; + tensor obj_687_strides_0 = const()[name = string("obj_687_strides_0"), val = tensor([1, 1])]; + tensor obj_687_pad_0 = const()[name = string("obj_687_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_687_dilations_0 = const()[name = string("obj_687_dilations_0"), val = tensor([1, 1])]; + int32 obj_687_groups_0 = const()[name = string("obj_687_groups_0"), val = int32(1)]; + tensor obj_687_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_687_dilations_0, groups = obj_687_groups_0, pad = obj_687_pad_0, pad_type = obj_687_pad_type_0, strides = obj_687_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_673_cast_fp16)[name = string("obj_687_cast_fp16")]; + tensor inputs_651_cast_fp16 = add(x = inputs_645_cast_fp16, y = obj_687_cast_fp16)[name = string("inputs_651_cast_fp16")]; + tensor inputs_sq_651_cast_fp16 = mul(x = inputs_651_cast_fp16, y = inputs_651_cast_fp16)[name = string("inputs_sq_651_cast_fp16")]; + tensor variance_651_axes_0 = const()[name = string("variance_651_axes_0"), val = tensor([1])]; + bool variance_651_keep_dims_0 = const()[name = string("variance_651_keep_dims_0"), val = bool(true)]; + tensor variance_651_cast_fp16 = reduce_mean(axes = variance_651_axes_0, keep_dims = variance_651_keep_dims_0, x = inputs_sq_651_cast_fp16)[name = string("variance_651_cast_fp16")]; + fp16 var_23799_to_fp16 = const()[name = string("op_23799_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23800_cast_fp16 = add(x = variance_651_cast_fp16, y = var_23799_to_fp16)[name = string("op_23800_cast_fp16")]; + fp32 var_23801_epsilon_0 = const()[name = string("op_23801_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23801_cast_fp16 = rsqrt(epsilon = var_23801_epsilon_0, x = var_23800_cast_fp16)[name = string("op_23801_cast_fp16")]; + tensor hidden_states_805_cast_fp16 = mul(x = inputs_651_cast_fp16, y = var_23801_cast_fp16)[name = string("hidden_states_805_cast_fp16")]; + tensor input_675_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_805_cast_fp16)[name = string("input_675_cast_fp16")]; + string input_677_pad_type_0 = const()[name = string("input_677_pad_type_0"), val = string("valid")]; + tensor input_677_strides_0 = const()[name = string("input_677_strides_0"), val = tensor([1, 1])]; + tensor input_677_pad_0 = const()[name = string("input_677_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_677_dilations_0 = const()[name = string("input_677_dilations_0"), val = tensor([1, 1])]; + int32 input_677_groups_0 = const()[name = string("input_677_groups_0"), val = int32(1)]; + tensor input_677_cast_fp16 = conv(dilations = input_677_dilations_0, groups = input_677_groups_0, pad = input_677_pad_0, pad_type = input_677_pad_type_0, strides = input_677_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_675_cast_fp16)[name = string("input_677_cast_fp16")]; + tensor var_23815_cast_fp16 = silu(x = input_677_cast_fp16)[name = string("op_23815_cast_fp16")]; + string var_23821_pad_type_0 = const()[name = string("op_23821_pad_type_0"), val = string("valid")]; + tensor var_23821_strides_0 = const()[name = string("op_23821_strides_0"), val = tensor([1, 1])]; + tensor var_23821_pad_0 = const()[name = string("op_23821_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_23821_dilations_0 = const()[name = string("op_23821_dilations_0"), val = tensor([1, 1])]; + int32 var_23821_groups_0 = const()[name = string("op_23821_groups_0"), val = int32(1)]; + tensor var_23821_cast_fp16 = conv(dilations = var_23821_dilations_0, groups = var_23821_groups_0, pad = var_23821_pad_0, pad_type = var_23821_pad_type_0, strides = var_23821_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_675_cast_fp16)[name = string("op_23821_cast_fp16")]; + tensor input_679_cast_fp16 = mul(x = var_23815_cast_fp16, y = var_23821_cast_fp16)[name = string("input_679_cast_fp16")]; + string hidden_states_807_pad_type_0 = const()[name = string("hidden_states_807_pad_type_0"), val = string("valid")]; + tensor hidden_states_807_strides_0 = const()[name = string("hidden_states_807_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_807_pad_0 = const()[name = string("hidden_states_807_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_807_dilations_0 = const()[name = string("hidden_states_807_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_807_groups_0 = const()[name = string("hidden_states_807_groups_0"), val = int32(1)]; + tensor hidden_states_807_cast_fp16 = conv(dilations = hidden_states_807_dilations_0, groups = hidden_states_807_groups_0, pad = hidden_states_807_pad_0, pad_type = hidden_states_807_pad_type_0, strides = hidden_states_807_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_679_cast_fp16)[name = string("hidden_states_807_cast_fp16")]; + tensor inputs_653_cast_fp16 = add(x = inputs_651_cast_fp16, y = hidden_states_807_cast_fp16)[name = string("inputs_653_cast_fp16")]; + tensor obj_691_begin_0 = const()[name = string("obj_691_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_691_end_0 = const()[name = string("obj_691_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_691_end_mask_0 = const()[name = string("obj_691_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_691_cast_fp16 = slice_by_index(begin = obj_691_begin_0, end = obj_691_end_0, end_mask = obj_691_end_mask_0, x = key_caches_cast_fp16)[name = string("obj_691_cast_fp16")]; + tensor obj_693_begin_0 = const()[name = string("obj_693_begin_0"), val = tensor([0, 3072, 0, 0])]; + tensor obj_693_end_0 = const()[name = string("obj_693_end_0"), val = tensor([1, 4096, 1, 16])]; + tensor obj_693_end_mask_0 = const()[name = string("obj_693_end_mask_0"), val = tensor([true, false, true, true])]; + tensor obj_693_cast_fp16 = slice_by_index(begin = obj_693_begin_0, end = obj_693_end_0, end_mask = obj_693_end_mask_0, x = value_caches_cast_fp16)[name = string("obj_693_cast_fp16")]; + int32 var_23855 = const()[name = string("op_23855"), val = int32(3)]; + int32 var_23865 = const()[name = string("op_23865"), val = int32(-2)]; + tensor inputs_sq_653_cast_fp16 = mul(x = inputs_653_cast_fp16, y = inputs_653_cast_fp16)[name = string("inputs_sq_653_cast_fp16")]; + tensor variance_653_axes_0 = const()[name = string("variance_653_axes_0"), val = tensor([1])]; + bool variance_653_keep_dims_0 = const()[name = string("variance_653_keep_dims_0"), val = bool(true)]; + tensor variance_653_cast_fp16 = reduce_mean(axes = variance_653_axes_0, keep_dims = variance_653_keep_dims_0, x = inputs_sq_653_cast_fp16)[name = string("variance_653_cast_fp16")]; + fp16 var_23879_to_fp16 = const()[name = string("op_23879_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23880_cast_fp16 = add(x = variance_653_cast_fp16, y = var_23879_to_fp16)[name = string("op_23880_cast_fp16")]; + fp32 var_23881_epsilon_0 = const()[name = string("op_23881_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23881_cast_fp16 = rsqrt(epsilon = var_23881_epsilon_0, x = var_23880_cast_fp16)[name = string("op_23881_cast_fp16")]; + tensor hidden_states_809_cast_fp16 = mul(x = inputs_653_cast_fp16, y = var_23881_cast_fp16)[name = string("hidden_states_809_cast_fp16")]; + tensor obj_689_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_809_cast_fp16)[name = string("obj_689_cast_fp16")]; + string query_469_pad_type_0 = const()[name = string("query_469_pad_type_0"), val = string("valid")]; + tensor query_469_strides_0 = const()[name = string("query_469_strides_0"), val = tensor([1, 1])]; + tensor query_469_pad_0 = const()[name = string("query_469_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_469_dilations_0 = const()[name = string("query_469_dilations_0"), val = tensor([1, 1])]; + int32 query_469_groups_0 = const()[name = string("query_469_groups_0"), val = int32(1)]; + tensor query_469_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_469_dilations_0, groups = query_469_groups_0, pad = query_469_pad_0, pad_type = query_469_pad_type_0, strides = query_469_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_689_cast_fp16)[name = string("query_469_cast_fp16")]; + string current_key_313_pad_type_0 = const()[name = string("current_key_313_pad_type_0"), val = string("valid")]; + tensor current_key_313_strides_0 = const()[name = string("current_key_313_strides_0"), val = tensor([1, 1])]; + tensor current_key_313_pad_0 = const()[name = string("current_key_313_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_313_dilations_0 = const()[name = string("current_key_313_dilations_0"), val = tensor([1, 1])]; + int32 current_key_313_groups_0 = const()[name = string("current_key_313_groups_0"), val = int32(1)]; + tensor current_key_313_cast_fp16 = conv(dilations = current_key_313_dilations_0, groups = current_key_313_groups_0, pad = current_key_313_pad_0, pad_type = current_key_313_pad_type_0, strides = current_key_313_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_689_cast_fp16)[name = string("current_key_313_cast_fp16")]; + string current_value_157_pad_type_0 = const()[name = string("current_value_157_pad_type_0"), val = string("valid")]; + tensor current_value_157_strides_0 = const()[name = string("current_value_157_strides_0"), val = tensor([1, 1])]; + tensor current_value_157_pad_0 = const()[name = string("current_value_157_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_157_dilations_0 = const()[name = string("current_value_157_dilations_0"), val = tensor([1, 1])]; + int32 current_value_157_groups_0 = const()[name = string("current_value_157_groups_0"), val = int32(1)]; + tensor current_value_157_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_157_dilations_0, groups = current_value_157_groups_0, pad = current_value_157_pad_0, pad_type = current_value_157_pad_type_0, strides = current_value_157_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_689_cast_fp16)[name = string("current_value_157_cast_fp16")]; + tensor var_23918 = const()[name = string("op_23918"), val = tensor([16, 128, 1, 1])]; + tensor inputs_655_cast_fp16 = reshape(shape = var_23918, x = query_469_cast_fp16)[name = string("inputs_655_cast_fp16")]; + tensor inputs_sq_655_cast_fp16 = mul(x = inputs_655_cast_fp16, y = inputs_655_cast_fp16)[name = string("inputs_sq_655_cast_fp16")]; + tensor variance_655_axes_0 = const()[name = string("variance_655_axes_0"), val = tensor([1])]; + bool variance_655_keep_dims_0 = const()[name = string("variance_655_keep_dims_0"), val = bool(true)]; + tensor variance_655_cast_fp16 = reduce_mean(axes = variance_655_axes_0, keep_dims = variance_655_keep_dims_0, x = inputs_sq_655_cast_fp16)[name = string("variance_655_cast_fp16")]; + fp16 var_23924_to_fp16 = const()[name = string("op_23924_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23925_cast_fp16 = add(x = variance_655_cast_fp16, y = var_23924_to_fp16)[name = string("op_23925_cast_fp16")]; + fp32 var_23926_epsilon_0 = const()[name = string("op_23926_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23926_cast_fp16 = rsqrt(epsilon = var_23926_epsilon_0, x = var_23925_cast_fp16)[name = string("op_23926_cast_fp16")]; + tensor hidden_states_811_cast_fp16 = mul(x = inputs_655_cast_fp16, y = var_23926_cast_fp16)[name = string("hidden_states_811_cast_fp16")]; + tensor query_normed_157_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_811_cast_fp16)[name = string("query_normed_157_cast_fp16")]; + tensor var_23934 = const()[name = string("op_23934"), val = tensor([8, 128, 1, 1])]; + tensor inputs_657_cast_fp16 = reshape(shape = var_23934, x = current_key_313_cast_fp16)[name = string("inputs_657_cast_fp16")]; + tensor inputs_sq_657_cast_fp16 = mul(x = inputs_657_cast_fp16, y = inputs_657_cast_fp16)[name = string("inputs_sq_657_cast_fp16")]; + tensor variance_657_axes_0 = const()[name = string("variance_657_axes_0"), val = tensor([1])]; + bool variance_657_keep_dims_0 = const()[name = string("variance_657_keep_dims_0"), val = bool(true)]; + tensor variance_657_cast_fp16 = reduce_mean(axes = variance_657_axes_0, keep_dims = variance_657_keep_dims_0, x = inputs_sq_657_cast_fp16)[name = string("variance_657_cast_fp16")]; + fp16 var_23940_to_fp16 = const()[name = string("op_23940_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_23941_cast_fp16 = add(x = variance_657_cast_fp16, y = var_23940_to_fp16)[name = string("op_23941_cast_fp16")]; + fp32 var_23942_epsilon_0 = const()[name = string("op_23942_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_23942_cast_fp16 = rsqrt(epsilon = var_23942_epsilon_0, x = var_23941_cast_fp16)[name = string("op_23942_cast_fp16")]; + tensor hidden_states_813_cast_fp16 = mul(x = inputs_657_cast_fp16, y = var_23942_cast_fp16)[name = string("hidden_states_813_cast_fp16")]; + tensor current_key_normed_157_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_813_cast_fp16)[name = string("current_key_normed_157_cast_fp16")]; + tensor var_23960 = const()[name = string("op_23960"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_625_cast_fp16 = reshape(shape = var_23960, x = query_normed_157_cast_fp16)[name = string("mh_q_625_cast_fp16")]; + tensor var_23962 = const()[name = string("op_23962"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_625_cast_fp16 = reshape(shape = var_23962, x = current_key_normed_157_cast_fp16)[name = string("mh_k_625_cast_fp16")]; + tensor var_23966_cast_fp16 = mul(x = mh_q_625_cast_fp16, y = cos_151_to_fp16)[name = string("op_23966_cast_fp16")]; + tensor var_23971_begin_0 = const()[name = string("op_23971_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_23971_end_0 = const()[name = string("op_23971_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_23971_end_mask_0 = const()[name = string("op_23971_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_23971_cast_fp16 = slice_by_index(begin = var_23971_begin_0, end = var_23971_end_0, end_mask = var_23971_end_mask_0, x = mh_q_625_cast_fp16)[name = string("op_23971_cast_fp16")]; + tensor var_23977_begin_0 = const()[name = string("op_23977_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_23977_end_0 = const()[name = string("op_23977_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_23977_end_mask_0 = const()[name = string("op_23977_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_23977_cast_fp16 = slice_by_index(begin = var_23977_begin_0, end = var_23977_end_0, end_mask = var_23977_end_mask_0, x = mh_q_625_cast_fp16)[name = string("op_23977_cast_fp16")]; + fp16 const_1589_promoted_to_fp16 = const()[name = string("const_1589_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_23979_cast_fp16 = mul(x = var_23977_cast_fp16, y = const_1589_promoted_to_fp16)[name = string("op_23979_cast_fp16")]; + bool var_23981_interleave_0 = const()[name = string("op_23981_interleave_0"), val = bool(false)]; + tensor var_23981_cast_fp16 = concat(axis = var_23865, interleave = var_23981_interleave_0, values = (var_23979_cast_fp16, var_23971_cast_fp16))[name = string("op_23981_cast_fp16")]; + tensor var_23982_cast_fp16 = mul(x = var_23981_cast_fp16, y = sin_151_to_fp16)[name = string("op_23982_cast_fp16")]; + tensor mh_q_627_cast_fp16 = add(x = var_23966_cast_fp16, y = var_23982_cast_fp16)[name = string("mh_q_627_cast_fp16")]; + tensor var_23984_cast_fp16 = mul(x = mh_k_625_cast_fp16, y = cos_151_to_fp16)[name = string("op_23984_cast_fp16")]; + tensor var_23989_begin_0 = const()[name = string("op_23989_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_23989_end_0 = const()[name = string("op_23989_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_23989_end_mask_0 = const()[name = string("op_23989_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_23989_cast_fp16 = slice_by_index(begin = var_23989_begin_0, end = var_23989_end_0, end_mask = var_23989_end_mask_0, x = mh_k_625_cast_fp16)[name = string("op_23989_cast_fp16")]; + tensor var_23995_begin_0 = const()[name = string("op_23995_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_23995_end_0 = const()[name = string("op_23995_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_23995_end_mask_0 = const()[name = string("op_23995_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_23995_cast_fp16 = slice_by_index(begin = var_23995_begin_0, end = var_23995_end_0, end_mask = var_23995_end_mask_0, x = mh_k_625_cast_fp16)[name = string("op_23995_cast_fp16")]; + fp16 const_1592_promoted_to_fp16 = const()[name = string("const_1592_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_23997_cast_fp16 = mul(x = var_23995_cast_fp16, y = const_1592_promoted_to_fp16)[name = string("op_23997_cast_fp16")]; + bool var_23999_interleave_0 = const()[name = string("op_23999_interleave_0"), val = bool(false)]; + tensor var_23999_cast_fp16 = concat(axis = var_23865, interleave = var_23999_interleave_0, values = (var_23997_cast_fp16, var_23989_cast_fp16))[name = string("op_23999_cast_fp16")]; + tensor var_24000_cast_fp16 = mul(x = var_23999_cast_fp16, y = sin_151_to_fp16)[name = string("op_24000_cast_fp16")]; + tensor mh_k_627_cast_fp16 = add(x = var_23984_cast_fp16, y = var_24000_cast_fp16)[name = string("mh_k_627_cast_fp16")]; + tensor var_24004 = const()[name = string("op_24004"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_315_cast_fp16 = reshape(shape = var_24004, x = mh_k_627_cast_fp16)[name = string("current_key_315_cast_fp16")]; + tensor var_24010_to_fp16 = const()[name = string("op_24010_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204544)))]; + tensor var_24011_cast_fp16 = mul(x = obj_691_cast_fp16, y = var_24010_to_fp16)[name = string("op_24011_cast_fp16")]; + tensor var_24008_to_fp16 = const()[name = string("op_24008_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204672)))]; + tensor var_24012_cast_fp16 = mul(x = current_key_315_cast_fp16, y = var_24008_to_fp16)[name = string("op_24012_cast_fp16")]; + tensor key_315_cast_fp16 = add(x = var_24011_cast_fp16, y = var_24012_cast_fp16)[name = string("key_315_cast_fp16")]; + tensor var_24014_to_fp16 = const()[name = string("op_24014_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204544)))]; + tensor var_24015_cast_fp16 = mul(x = obj_693_cast_fp16, y = var_24014_to_fp16)[name = string("op_24015_cast_fp16")]; + tensor var_24016_cast_fp16 = mul(x = current_value_157_cast_fp16, y = var_24008_to_fp16)[name = string("op_24016_cast_fp16")]; + tensor value_157_cast_fp16 = add(x = var_24015_cast_fp16, y = var_24016_cast_fp16)[name = string("value_157_cast_fp16")]; + fp16 var_24023_to_fp16 = const()[name = string("op_24023_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_631_cast_fp16 = mul(x = mh_q_627_cast_fp16, y = var_24023_to_fp16)[name = string("mh_q_631_cast_fp16")]; + tensor var_24025 = const()[name = string("op_24025"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_629_cast_fp16 = reshape(shape = var_24025, x = key_315_cast_fp16)[name = string("mh_k_629_cast_fp16")]; + tensor var_24027 = const()[name = string("op_24027"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_313_cast_fp16 = reshape(shape = var_24027, x = value_157_cast_fp16)[name = string("mh_v_313_cast_fp16")]; + tensor transpose_312_perm_0 = const()[name = string("transpose_312_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_156_reps_0 = const()[name = string("tile_156_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_312_cast_fp16 = transpose(perm = transpose_312_perm_0, x = mh_k_629_cast_fp16)[name = string("transpose_11")]; + tensor tile_156_cast_fp16 = tile(reps = tile_156_reps_0, x = transpose_312_cast_fp16)[name = string("tile_156_cast_fp16")]; + tensor concat_390 = const()[name = string("concat_390"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_312_cast_fp16 = reshape(shape = concat_390, x = tile_156_cast_fp16)[name = string("reshape_312_cast_fp16")]; + tensor transpose_313_perm_0 = const()[name = string("transpose_313_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_391 = const()[name = string("concat_391"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_313_cast_fp16 = transpose(perm = transpose_313_perm_0, x = reshape_312_cast_fp16)[name = string("transpose_10")]; + tensor reshape_313_cast_fp16 = reshape(shape = concat_391, x = transpose_313_cast_fp16)[name = string("reshape_313_cast_fp16")]; + tensor transpose_314_perm_0 = const()[name = string("transpose_314_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_157_reps_0 = const()[name = string("tile_157_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_314_cast_fp16 = transpose(perm = transpose_314_perm_0, x = mh_v_313_cast_fp16)[name = string("transpose_9")]; + tensor tile_157_cast_fp16 = tile(reps = tile_157_reps_0, x = transpose_314_cast_fp16)[name = string("tile_157_cast_fp16")]; + tensor concat_392 = const()[name = string("concat_392"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_314_cast_fp16 = reshape(shape = concat_392, x = tile_157_cast_fp16)[name = string("reshape_314_cast_fp16")]; + tensor transpose_315_perm_0 = const()[name = string("transpose_315_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_393 = const()[name = string("concat_393"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_315_cast_fp16 = transpose(perm = transpose_315_perm_0, x = reshape_314_cast_fp16)[name = string("transpose_8")]; + tensor reshape_315_cast_fp16 = reshape(shape = concat_393, x = transpose_315_cast_fp16)[name = string("reshape_315_cast_fp16")]; + tensor transpose_629_perm_0 = const()[name = string("transpose_629_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_469_transpose_x_1 = const()[name = string("mh_w_469_transpose_x_1"), val = bool(true)]; + bool mh_w_469_transpose_y_1 = const()[name = string("mh_w_469_transpose_y_1"), val = bool(false)]; + tensor transpose_629_cast_fp16 = transpose(perm = transpose_629_perm_0, x = reshape_313_cast_fp16)[name = string("transpose_7")]; + tensor mh_w_469_cast_fp16 = matmul(transpose_x = mh_w_469_transpose_x_1, transpose_y = mh_w_469_transpose_y_1, x = mh_q_631_cast_fp16, y = transpose_629_cast_fp16)[name = string("mh_w_469_cast_fp16")]; + tensor mh_w_473_cast_fp16 = softmax(axis = var_23855, x = mh_w_469_cast_fp16)[name = string("mh_w_473_cast_fp16")]; + tensor transpose_630_perm_0 = const()[name = string("transpose_630_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_157_transpose_x_1 = const()[name = string("attn_157_transpose_x_1"), val = bool(false)]; + bool attn_157_transpose_y_1 = const()[name = string("attn_157_transpose_y_1"), val = bool(true)]; + tensor transpose_630_cast_fp16 = transpose(perm = transpose_630_perm_0, x = reshape_315_cast_fp16)[name = string("transpose_6")]; + tensor attn_157_cast_fp16 = matmul(transpose_x = attn_157_transpose_x_1, transpose_y = attn_157_transpose_y_1, x = transpose_630_cast_fp16, y = mh_w_473_cast_fp16)[name = string("attn_157_cast_fp16")]; + tensor var_24041 = const()[name = string("op_24041"), val = tensor([1, 2048, 1, 1])]; + tensor input_681_cast_fp16 = reshape(shape = var_24041, x = attn_157_cast_fp16)[name = string("input_681_cast_fp16")]; + string obj_695_pad_type_0 = const()[name = string("obj_695_pad_type_0"), val = string("valid")]; + tensor obj_695_strides_0 = const()[name = string("obj_695_strides_0"), val = tensor([1, 1])]; + tensor obj_695_pad_0 = const()[name = string("obj_695_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_695_dilations_0 = const()[name = string("obj_695_dilations_0"), val = tensor([1, 1])]; + int32 obj_695_groups_0 = const()[name = string("obj_695_groups_0"), val = int32(1)]; + tensor obj_695_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_695_dilations_0, groups = obj_695_groups_0, pad = obj_695_pad_0, pad_type = obj_695_pad_type_0, strides = obj_695_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_681_cast_fp16)[name = string("obj_695_cast_fp16")]; + tensor inputs_659_cast_fp16 = add(x = inputs_653_cast_fp16, y = obj_695_cast_fp16)[name = string("inputs_659_cast_fp16")]; + tensor inputs_sq_659_cast_fp16 = mul(x = inputs_659_cast_fp16, y = inputs_659_cast_fp16)[name = string("inputs_sq_659_cast_fp16")]; + tensor variance_659_axes_0 = const()[name = string("variance_659_axes_0"), val = tensor([1])]; + bool variance_659_keep_dims_0 = const()[name = string("variance_659_keep_dims_0"), val = bool(true)]; + tensor variance_659_cast_fp16 = reduce_mean(axes = variance_659_axes_0, keep_dims = variance_659_keep_dims_0, x = inputs_sq_659_cast_fp16)[name = string("variance_659_cast_fp16")]; + fp16 var_24059_to_fp16 = const()[name = string("op_24059_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_24060_cast_fp16 = add(x = variance_659_cast_fp16, y = var_24059_to_fp16)[name = string("op_24060_cast_fp16")]; + fp32 var_24061_epsilon_0 = const()[name = string("op_24061_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_24061_cast_fp16 = rsqrt(epsilon = var_24061_epsilon_0, x = var_24060_cast_fp16)[name = string("op_24061_cast_fp16")]; + tensor hidden_states_815_cast_fp16 = mul(x = inputs_659_cast_fp16, y = var_24061_cast_fp16)[name = string("hidden_states_815_cast_fp16")]; + tensor input_683_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_815_cast_fp16)[name = string("input_683_cast_fp16")]; + string input_685_pad_type_0 = const()[name = string("input_685_pad_type_0"), val = string("valid")]; + tensor input_685_strides_0 = const()[name = string("input_685_strides_0"), val = tensor([1, 1])]; + tensor input_685_pad_0 = const()[name = string("input_685_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_685_dilations_0 = const()[name = string("input_685_dilations_0"), val = tensor([1, 1])]; + int32 input_685_groups_0 = const()[name = string("input_685_groups_0"), val = int32(1)]; + tensor input_685_cast_fp16 = conv(dilations = input_685_dilations_0, groups = input_685_groups_0, pad = input_685_pad_0, pad_type = input_685_pad_type_0, strides = input_685_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_683_cast_fp16)[name = string("input_685_cast_fp16")]; + tensor var_24075_cast_fp16 = silu(x = input_685_cast_fp16)[name = string("op_24075_cast_fp16")]; + string var_24081_pad_type_0 = const()[name = string("op_24081_pad_type_0"), val = string("valid")]; + tensor var_24081_strides_0 = const()[name = string("op_24081_strides_0"), val = tensor([1, 1])]; + tensor var_24081_pad_0 = const()[name = string("op_24081_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_24081_dilations_0 = const()[name = string("op_24081_dilations_0"), val = tensor([1, 1])]; + int32 var_24081_groups_0 = const()[name = string("op_24081_groups_0"), val = int32(1)]; + tensor var_24081_cast_fp16 = conv(dilations = var_24081_dilations_0, groups = var_24081_groups_0, pad = var_24081_pad_0, pad_type = var_24081_pad_type_0, strides = var_24081_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_683_cast_fp16)[name = string("op_24081_cast_fp16")]; + tensor input_687_cast_fp16 = mul(x = var_24075_cast_fp16, y = var_24081_cast_fp16)[name = string("input_687_cast_fp16")]; + string hidden_states_817_pad_type_0 = const()[name = string("hidden_states_817_pad_type_0"), val = string("valid")]; + tensor hidden_states_817_strides_0 = const()[name = string("hidden_states_817_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_817_pad_0 = const()[name = string("hidden_states_817_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_817_dilations_0 = const()[name = string("hidden_states_817_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_817_groups_0 = const()[name = string("hidden_states_817_groups_0"), val = int32(1)]; + tensor hidden_states_817_cast_fp16 = conv(dilations = hidden_states_817_dilations_0, groups = hidden_states_817_groups_0, pad = hidden_states_817_pad_0, pad_type = hidden_states_817_pad_type_0, strides = hidden_states_817_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_687_cast_fp16)[name = string("hidden_states_817_cast_fp16")]; + tensor inputs_661_cast_fp16 = add(x = inputs_659_cast_fp16, y = hidden_states_817_cast_fp16)[name = string("inputs_661_cast_fp16")]; + tensor obj_699_begin_0 = const()[name = string("obj_699_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_699_end_0 = const()[name = string("obj_699_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_699_end_mask_0 = const()[name = string("obj_699_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_699_cast_fp16 = slice_by_index(begin = obj_699_begin_0, end = obj_699_end_0, end_mask = obj_699_end_mask_0, x = key_caches_cast_fp16)[name = string("obj_699_cast_fp16")]; + tensor obj_701_begin_0 = const()[name = string("obj_701_begin_0"), val = tensor([0, 4096, 0, 0])]; + tensor obj_701_end_0 = const()[name = string("obj_701_end_0"), val = tensor([1, 1, 1, 16])]; + tensor obj_701_end_mask_0 = const()[name = string("obj_701_end_mask_0"), val = tensor([true, true, true, true])]; + tensor obj_701_cast_fp16 = slice_by_index(begin = obj_701_begin_0, end = obj_701_end_0, end_mask = obj_701_end_mask_0, x = value_caches_cast_fp16)[name = string("obj_701_cast_fp16")]; + int32 var_24115 = const()[name = string("op_24115"), val = int32(3)]; + int32 var_24125 = const()[name = string("op_24125"), val = int32(-2)]; + tensor inputs_sq_661_cast_fp16 = mul(x = inputs_661_cast_fp16, y = inputs_661_cast_fp16)[name = string("inputs_sq_661_cast_fp16")]; + tensor variance_661_axes_0 = const()[name = string("variance_661_axes_0"), val = tensor([1])]; + bool variance_661_keep_dims_0 = const()[name = string("variance_661_keep_dims_0"), val = bool(true)]; + tensor variance_661_cast_fp16 = reduce_mean(axes = variance_661_axes_0, keep_dims = variance_661_keep_dims_0, x = inputs_sq_661_cast_fp16)[name = string("variance_661_cast_fp16")]; + fp16 var_24139_to_fp16 = const()[name = string("op_24139_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_24140_cast_fp16 = add(x = variance_661_cast_fp16, y = var_24139_to_fp16)[name = string("op_24140_cast_fp16")]; + fp32 var_24141_epsilon_0 = const()[name = string("op_24141_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_24141_cast_fp16 = rsqrt(epsilon = var_24141_epsilon_0, x = var_24140_cast_fp16)[name = string("op_24141_cast_fp16")]; + tensor hidden_states_819_cast_fp16 = mul(x = inputs_661_cast_fp16, y = var_24141_cast_fp16)[name = string("hidden_states_819_cast_fp16")]; + tensor obj_697_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_819_cast_fp16)[name = string("obj_697_cast_fp16")]; + string query_475_pad_type_0 = const()[name = string("query_475_pad_type_0"), val = string("valid")]; + tensor query_475_strides_0 = const()[name = string("query_475_strides_0"), val = tensor([1, 1])]; + tensor query_475_pad_0 = const()[name = string("query_475_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_475_dilations_0 = const()[name = string("query_475_dilations_0"), val = tensor([1, 1])]; + int32 query_475_groups_0 = const()[name = string("query_475_groups_0"), val = int32(1)]; + tensor query_475_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_475_dilations_0, groups = query_475_groups_0, pad = query_475_pad_0, pad_type = query_475_pad_type_0, strides = query_475_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_697_cast_fp16)[name = string("query_475_cast_fp16")]; + string current_key_317_pad_type_0 = const()[name = string("current_key_317_pad_type_0"), val = string("valid")]; + tensor current_key_317_strides_0 = const()[name = string("current_key_317_strides_0"), val = tensor([1, 1])]; + tensor current_key_317_pad_0 = const()[name = string("current_key_317_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_317_dilations_0 = const()[name = string("current_key_317_dilations_0"), val = tensor([1, 1])]; + int32 current_key_317_groups_0 = const()[name = string("current_key_317_groups_0"), val = int32(1)]; + tensor current_key_317_cast_fp16 = conv(dilations = current_key_317_dilations_0, groups = current_key_317_groups_0, pad = current_key_317_pad_0, pad_type = current_key_317_pad_type_0, strides = current_key_317_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_697_cast_fp16)[name = string("current_key_317_cast_fp16")]; + string current_value_pad_type_0 = const()[name = string("current_value_pad_type_0"), val = string("valid")]; + tensor current_value_strides_0 = const()[name = string("current_value_strides_0"), val = tensor([1, 1])]; + tensor current_value_pad_0 = const()[name = string("current_value_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_dilations_0 = const()[name = string("current_value_dilations_0"), val = tensor([1, 1])]; + int32 current_value_groups_0 = const()[name = string("current_value_groups_0"), val = int32(1)]; + tensor current_value_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_dilations_0, groups = current_value_groups_0, pad = current_value_pad_0, pad_type = current_value_pad_type_0, strides = current_value_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_697_cast_fp16)[name = string("current_value_cast_fp16")]; + tensor var_24178 = const()[name = string("op_24178"), val = tensor([16, 128, 1, 1])]; + tensor inputs_663_cast_fp16 = reshape(shape = var_24178, x = query_475_cast_fp16)[name = string("inputs_663_cast_fp16")]; + tensor inputs_sq_663_cast_fp16 = mul(x = inputs_663_cast_fp16, y = inputs_663_cast_fp16)[name = string("inputs_sq_663_cast_fp16")]; + tensor variance_663_axes_0 = const()[name = string("variance_663_axes_0"), val = tensor([1])]; + bool variance_663_keep_dims_0 = const()[name = string("variance_663_keep_dims_0"), val = bool(true)]; + tensor variance_663_cast_fp16 = reduce_mean(axes = variance_663_axes_0, keep_dims = variance_663_keep_dims_0, x = inputs_sq_663_cast_fp16)[name = string("variance_663_cast_fp16")]; + fp16 var_24184_to_fp16 = const()[name = string("op_24184_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_24185_cast_fp16 = add(x = variance_663_cast_fp16, y = var_24184_to_fp16)[name = string("op_24185_cast_fp16")]; + fp32 var_24186_epsilon_0 = const()[name = string("op_24186_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_24186_cast_fp16 = rsqrt(epsilon = var_24186_epsilon_0, x = var_24185_cast_fp16)[name = string("op_24186_cast_fp16")]; + tensor hidden_states_821_cast_fp16 = mul(x = inputs_663_cast_fp16, y = var_24186_cast_fp16)[name = string("hidden_states_821_cast_fp16")]; + tensor query_normed_cast_fp16 = mul(x = w_75_to_fp16, y = hidden_states_821_cast_fp16)[name = string("query_normed_cast_fp16")]; + tensor var_24194 = const()[name = string("op_24194"), val = tensor([8, 128, 1, 1])]; + tensor inputs_665_cast_fp16 = reshape(shape = var_24194, x = current_key_317_cast_fp16)[name = string("inputs_665_cast_fp16")]; + tensor inputs_sq_665_cast_fp16 = mul(x = inputs_665_cast_fp16, y = inputs_665_cast_fp16)[name = string("inputs_sq_665_cast_fp16")]; + tensor variance_665_axes_0 = const()[name = string("variance_665_axes_0"), val = tensor([1])]; + bool variance_665_keep_dims_0 = const()[name = string("variance_665_keep_dims_0"), val = bool(true)]; + tensor variance_665_cast_fp16 = reduce_mean(axes = variance_665_axes_0, keep_dims = variance_665_keep_dims_0, x = inputs_sq_665_cast_fp16)[name = string("variance_665_cast_fp16")]; + fp16 var_24200_to_fp16 = const()[name = string("op_24200_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_24201_cast_fp16 = add(x = variance_665_cast_fp16, y = var_24200_to_fp16)[name = string("op_24201_cast_fp16")]; + fp32 var_24202_epsilon_0 = const()[name = string("op_24202_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_24202_cast_fp16 = rsqrt(epsilon = var_24202_epsilon_0, x = var_24201_cast_fp16)[name = string("op_24202_cast_fp16")]; + tensor hidden_states_823_cast_fp16 = mul(x = inputs_665_cast_fp16, y = var_24202_cast_fp16)[name = string("hidden_states_823_cast_fp16")]; + tensor current_key_normed_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_823_cast_fp16)[name = string("current_key_normed_cast_fp16")]; + tensor var_24220 = const()[name = string("op_24220"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_633_cast_fp16 = reshape(shape = var_24220, x = query_normed_cast_fp16)[name = string("mh_q_633_cast_fp16")]; + tensor var_24222 = const()[name = string("op_24222"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_633_cast_fp16 = reshape(shape = var_24222, x = current_key_normed_cast_fp16)[name = string("mh_k_633_cast_fp16")]; + tensor var_24226_cast_fp16 = mul(x = mh_q_633_cast_fp16, y = cos_151_to_fp16)[name = string("op_24226_cast_fp16")]; + tensor var_24231_begin_0 = const()[name = string("op_24231_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_24231_end_0 = const()[name = string("op_24231_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_24231_end_mask_0 = const()[name = string("op_24231_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_24231_cast_fp16 = slice_by_index(begin = var_24231_begin_0, end = var_24231_end_0, end_mask = var_24231_end_mask_0, x = mh_q_633_cast_fp16)[name = string("op_24231_cast_fp16")]; + tensor var_24237_begin_0 = const()[name = string("op_24237_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_24237_end_0 = const()[name = string("op_24237_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_24237_end_mask_0 = const()[name = string("op_24237_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_24237_cast_fp16 = slice_by_index(begin = var_24237_begin_0, end = var_24237_end_0, end_mask = var_24237_end_mask_0, x = mh_q_633_cast_fp16)[name = string("op_24237_cast_fp16")]; + fp16 const_1609_promoted_to_fp16 = const()[name = string("const_1609_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_24239_cast_fp16 = mul(x = var_24237_cast_fp16, y = const_1609_promoted_to_fp16)[name = string("op_24239_cast_fp16")]; + bool var_24241_interleave_0 = const()[name = string("op_24241_interleave_0"), val = bool(false)]; + tensor var_24241_cast_fp16 = concat(axis = var_24125, interleave = var_24241_interleave_0, values = (var_24239_cast_fp16, var_24231_cast_fp16))[name = string("op_24241_cast_fp16")]; + tensor var_24242_cast_fp16 = mul(x = var_24241_cast_fp16, y = sin_151_to_fp16)[name = string("op_24242_cast_fp16")]; + tensor mh_q_635_cast_fp16 = add(x = var_24226_cast_fp16, y = var_24242_cast_fp16)[name = string("mh_q_635_cast_fp16")]; + tensor var_24244_cast_fp16 = mul(x = mh_k_633_cast_fp16, y = cos_151_to_fp16)[name = string("op_24244_cast_fp16")]; + tensor var_24249_begin_0 = const()[name = string("op_24249_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_24249_end_0 = const()[name = string("op_24249_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_24249_end_mask_0 = const()[name = string("op_24249_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_24249_cast_fp16 = slice_by_index(begin = var_24249_begin_0, end = var_24249_end_0, end_mask = var_24249_end_mask_0, x = mh_k_633_cast_fp16)[name = string("op_24249_cast_fp16")]; + tensor var_24255_begin_0 = const()[name = string("op_24255_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_24255_end_0 = const()[name = string("op_24255_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_24255_end_mask_0 = const()[name = string("op_24255_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_24255_cast_fp16 = slice_by_index(begin = var_24255_begin_0, end = var_24255_end_0, end_mask = var_24255_end_mask_0, x = mh_k_633_cast_fp16)[name = string("op_24255_cast_fp16")]; + fp16 const_1612_promoted_to_fp16 = const()[name = string("const_1612_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_24257_cast_fp16 = mul(x = var_24255_cast_fp16, y = const_1612_promoted_to_fp16)[name = string("op_24257_cast_fp16")]; + bool var_24259_interleave_0 = const()[name = string("op_24259_interleave_0"), val = bool(false)]; + tensor var_24259_cast_fp16 = concat(axis = var_24125, interleave = var_24259_interleave_0, values = (var_24257_cast_fp16, var_24249_cast_fp16))[name = string("op_24259_cast_fp16")]; + tensor var_24260_cast_fp16 = mul(x = var_24259_cast_fp16, y = sin_151_to_fp16)[name = string("op_24260_cast_fp16")]; + tensor mh_k_635_cast_fp16 = add(x = var_24244_cast_fp16, y = var_24260_cast_fp16)[name = string("mh_k_635_cast_fp16")]; + tensor var_24264 = const()[name = string("op_24264"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_cast_fp16 = reshape(shape = var_24264, x = mh_k_635_cast_fp16)[name = string("current_key_cast_fp16")]; + tensor var_24270_to_fp16 = const()[name = string("op_24270_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204544)))]; + tensor var_24271_cast_fp16 = mul(x = obj_699_cast_fp16, y = var_24270_to_fp16)[name = string("op_24271_cast_fp16")]; + tensor var_24268_to_fp16 = const()[name = string("op_24268_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204672)))]; + tensor var_24272_cast_fp16 = mul(x = current_key_cast_fp16, y = var_24268_to_fp16)[name = string("op_24272_cast_fp16")]; + tensor key_cast_fp16 = add(x = var_24271_cast_fp16, y = var_24272_cast_fp16)[name = string("key_cast_fp16")]; + tensor var_24274_to_fp16 = const()[name = string("op_24274_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175204544)))]; + tensor var_24275_cast_fp16 = mul(x = obj_701_cast_fp16, y = var_24274_to_fp16)[name = string("op_24275_cast_fp16")]; + tensor var_24276_cast_fp16 = mul(x = current_value_cast_fp16, y = var_24268_to_fp16)[name = string("op_24276_cast_fp16")]; + tensor value_cast_fp16 = add(x = var_24275_cast_fp16, y = var_24276_cast_fp16)[name = string("value_cast_fp16")]; + fp16 var_24283_to_fp16 = const()[name = string("op_24283_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor mh_q_cast_fp16 = mul(x = mh_q_635_cast_fp16, y = var_24283_to_fp16)[name = string("mh_q_cast_fp16")]; + tensor var_24285 = const()[name = string("op_24285"), val = tensor([1, 8, 128, 16])]; + tensor mh_k_637_cast_fp16 = reshape(shape = var_24285, x = key_cast_fp16)[name = string("mh_k_637_cast_fp16")]; + tensor var_24287 = const()[name = string("op_24287"), val = tensor([1, 8, 128, 16])]; + tensor mh_v_317_cast_fp16 = reshape(shape = var_24287, x = value_cast_fp16)[name = string("mh_v_317_cast_fp16")]; + tensor transpose_316_perm_0 = const()[name = string("transpose_316_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_158_reps_0 = const()[name = string("tile_158_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_316_cast_fp16 = transpose(perm = transpose_316_perm_0, x = mh_k_637_cast_fp16)[name = string("transpose_5")]; + tensor tile_158_cast_fp16 = tile(reps = tile_158_reps_0, x = transpose_316_cast_fp16)[name = string("tile_158_cast_fp16")]; + tensor concat_394 = const()[name = string("concat_394"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_316_cast_fp16 = reshape(shape = concat_394, x = tile_158_cast_fp16)[name = string("reshape_316_cast_fp16")]; + tensor transpose_317_perm_0 = const()[name = string("transpose_317_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_395 = const()[name = string("concat_395"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_317_cast_fp16 = transpose(perm = transpose_317_perm_0, x = reshape_316_cast_fp16)[name = string("transpose_4")]; + tensor reshape_317_cast_fp16 = reshape(shape = concat_395, x = transpose_317_cast_fp16)[name = string("reshape_317_cast_fp16")]; + tensor transpose_318_perm_0 = const()[name = string("transpose_318_perm_0"), val = tensor([1, 0, 2, 3])]; + tensor tile_159_reps_0 = const()[name = string("tile_159_reps_0"), val = tensor([2, 1, 1, 1])]; + tensor transpose_318_cast_fp16 = transpose(perm = transpose_318_perm_0, x = mh_v_317_cast_fp16)[name = string("transpose_3")]; + tensor tile_159_cast_fp16 = tile(reps = tile_159_reps_0, x = transpose_318_cast_fp16)[name = string("tile_159_cast_fp16")]; + tensor concat_396 = const()[name = string("concat_396"), val = tensor([2, 8, 1, 128, 16])]; + tensor reshape_318_cast_fp16 = reshape(shape = concat_396, x = tile_159_cast_fp16)[name = string("reshape_318_cast_fp16")]; + tensor transpose_319_perm_0 = const()[name = string("transpose_319_perm_0"), val = tensor([1, 0, 2, 3, 4])]; + tensor concat_397 = const()[name = string("concat_397"), val = tensor([-1, 1, 128, 16])]; + tensor transpose_319_cast_fp16 = transpose(perm = transpose_319_perm_0, x = reshape_318_cast_fp16)[name = string("transpose_2")]; + tensor reshape_319_cast_fp16 = reshape(shape = concat_397, x = transpose_319_cast_fp16)[name = string("reshape_319_cast_fp16")]; + tensor transpose_633_perm_0 = const()[name = string("transpose_633_perm_0"), val = tensor([1, 0, -2, -1])]; + bool mh_w_475_transpose_x_1 = const()[name = string("mh_w_475_transpose_x_1"), val = bool(true)]; + bool mh_w_475_transpose_y_1 = const()[name = string("mh_w_475_transpose_y_1"), val = bool(false)]; + tensor transpose_633_cast_fp16 = transpose(perm = transpose_633_perm_0, x = reshape_317_cast_fp16)[name = string("transpose_1")]; + tensor mh_w_475_cast_fp16 = matmul(transpose_x = mh_w_475_transpose_x_1, transpose_y = mh_w_475_transpose_y_1, x = mh_q_cast_fp16, y = transpose_633_cast_fp16)[name = string("mh_w_475_cast_fp16")]; + tensor mh_w_cast_fp16 = softmax(axis = var_24115, x = mh_w_475_cast_fp16)[name = string("mh_w_cast_fp16")]; + tensor transpose_634_perm_0 = const()[name = string("transpose_634_perm_0"), val = tensor([1, 0, -2, -1])]; + bool attn_transpose_x_1 = const()[name = string("attn_transpose_x_1"), val = bool(false)]; + bool attn_transpose_y_1 = const()[name = string("attn_transpose_y_1"), val = bool(true)]; + tensor transpose_634_cast_fp16 = transpose(perm = transpose_634_perm_0, x = reshape_319_cast_fp16)[name = string("transpose_0")]; + tensor attn_cast_fp16 = matmul(transpose_x = attn_transpose_x_1, transpose_y = attn_transpose_y_1, x = transpose_634_cast_fp16, y = mh_w_cast_fp16)[name = string("attn_cast_fp16")]; + tensor var_24301 = const()[name = string("op_24301"), val = tensor([1, 2048, 1, 1])]; + tensor input_689_cast_fp16 = reshape(shape = var_24301, x = attn_cast_fp16)[name = string("input_689_cast_fp16")]; + string obj_pad_type_0 = const()[name = string("obj_pad_type_0"), val = string("valid")]; + tensor obj_strides_0 = const()[name = string("obj_strides_0"), val = tensor([1, 1])]; + tensor obj_pad_0 = const()[name = string("obj_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_dilations_0 = const()[name = string("obj_dilations_0"), val = tensor([1, 1])]; + int32 obj_groups_0 = const()[name = string("obj_groups_0"), val = int32(1)]; + tensor obj_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_dilations_0, groups = obj_groups_0, pad = obj_pad_0, pad_type = obj_pad_type_0, strides = obj_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_689_cast_fp16)[name = string("obj_cast_fp16")]; + tensor inputs_667_cast_fp16 = add(x = inputs_661_cast_fp16, y = obj_cast_fp16)[name = string("inputs_667_cast_fp16")]; + tensor inputs_sq_667_cast_fp16 = mul(x = inputs_667_cast_fp16, y = inputs_667_cast_fp16)[name = string("inputs_sq_667_cast_fp16")]; + tensor variance_667_axes_0 = const()[name = string("variance_667_axes_0"), val = tensor([1])]; + bool variance_667_keep_dims_0 = const()[name = string("variance_667_keep_dims_0"), val = bool(true)]; + tensor variance_667_cast_fp16 = reduce_mean(axes = variance_667_axes_0, keep_dims = variance_667_keep_dims_0, x = inputs_sq_667_cast_fp16)[name = string("variance_667_cast_fp16")]; + fp16 var_24319_to_fp16 = const()[name = string("op_24319_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_24320_cast_fp16 = add(x = variance_667_cast_fp16, y = var_24319_to_fp16)[name = string("op_24320_cast_fp16")]; + fp32 var_24321_epsilon_0 = const()[name = string("op_24321_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_24321_cast_fp16 = rsqrt(epsilon = var_24321_epsilon_0, x = var_24320_cast_fp16)[name = string("op_24321_cast_fp16")]; + tensor hidden_states_825_cast_fp16 = mul(x = inputs_667_cast_fp16, y = var_24321_cast_fp16)[name = string("hidden_states_825_cast_fp16")]; + tensor input_691_cast_fp16 = mul(x = w_79_to_fp16, y = hidden_states_825_cast_fp16)[name = string("input_691_cast_fp16")]; + string input_693_pad_type_0 = const()[name = string("input_693_pad_type_0"), val = string("valid")]; + tensor input_693_strides_0 = const()[name = string("input_693_strides_0"), val = tensor([1, 1])]; + tensor input_693_pad_0 = const()[name = string("input_693_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_693_dilations_0 = const()[name = string("input_693_dilations_0"), val = tensor([1, 1])]; + int32 input_693_groups_0 = const()[name = string("input_693_groups_0"), val = int32(1)]; + tensor input_693_cast_fp16 = conv(dilations = input_693_dilations_0, groups = input_693_groups_0, pad = input_693_pad_0, pad_type = input_693_pad_type_0, strides = input_693_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_691_cast_fp16)[name = string("input_693_cast_fp16")]; + tensor var_24335_cast_fp16 = silu(x = input_693_cast_fp16)[name = string("op_24335_cast_fp16")]; + string var_24341_pad_type_0 = const()[name = string("op_24341_pad_type_0"), val = string("valid")]; + tensor var_24341_strides_0 = const()[name = string("op_24341_strides_0"), val = tensor([1, 1])]; + tensor var_24341_pad_0 = const()[name = string("op_24341_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_24341_dilations_0 = const()[name = string("op_24341_dilations_0"), val = tensor([1, 1])]; + int32 var_24341_groups_0 = const()[name = string("op_24341_groups_0"), val = int32(1)]; + tensor var_24341_cast_fp16 = conv(dilations = var_24341_dilations_0, groups = var_24341_groups_0, pad = var_24341_pad_0, pad_type = var_24341_pad_type_0, strides = var_24341_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_691_cast_fp16)[name = string("op_24341_cast_fp16")]; + tensor input_695_cast_fp16 = mul(x = var_24335_cast_fp16, y = var_24341_cast_fp16)[name = string("input_695_cast_fp16")]; + string hidden_states_827_pad_type_0 = const()[name = string("hidden_states_827_pad_type_0"), val = string("valid")]; + tensor hidden_states_827_strides_0 = const()[name = string("hidden_states_827_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_827_pad_0 = const()[name = string("hidden_states_827_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_827_dilations_0 = const()[name = string("hidden_states_827_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_827_groups_0 = const()[name = string("hidden_states_827_groups_0"), val = int32(1)]; + tensor hidden_states_827_cast_fp16 = conv(dilations = hidden_states_827_dilations_0, groups = hidden_states_827_groups_0, pad = hidden_states_827_pad_0, pad_type = hidden_states_827_pad_type_0, strides = hidden_states_827_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_695_cast_fp16)[name = string("hidden_states_827_cast_fp16")]; + tensor inputs_cast_fp16 = add(x = inputs_667_cast_fp16, y = hidden_states_827_cast_fp16)[name = string("inputs_cast_fp16")]; + tensor inputs_sq_cast_fp16 = mul(x = inputs_cast_fp16, y = inputs_cast_fp16)[name = string("inputs_sq_cast_fp16")]; + tensor variance_axes_0 = const()[name = string("variance_axes_0"), val = tensor([1])]; + bool variance_keep_dims_0 = const()[name = string("variance_keep_dims_0"), val = bool(true)]; + tensor variance_cast_fp16 = reduce_mean(axes = variance_axes_0, keep_dims = variance_keep_dims_0, x = inputs_sq_cast_fp16)[name = string("variance_cast_fp16")]; + fp16 var_24362_to_fp16 = const()[name = string("op_24362_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_24363_cast_fp16 = add(x = variance_cast_fp16, y = var_24362_to_fp16)[name = string("op_24363_cast_fp16")]; + fp32 var_24364_epsilon_0 = const()[name = string("op_24364_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_24364_cast_fp16 = rsqrt(epsilon = var_24364_epsilon_0, x = var_24363_cast_fp16)[name = string("op_24364_cast_fp16")]; + tensor hidden_states_cast_fp16 = mul(x = inputs_cast_fp16, y = var_24364_cast_fp16)[name = string("hidden_states_cast_fp16")]; + tensor input_697_cast_fp16 = mul(x = w_81_to_fp16, y = hidden_states_cast_fp16)[name = string("input_697_cast_fp16")]; + string logits_57_pad_type_0 = const()[name = string("logits_57_pad_type_0"), val = string("valid")]; + tensor logits_57_strides_0 = const()[name = string("logits_57_strides_0"), val = tensor([1, 1])]; + tensor logits_57_pad_0 = const()[name = string("logits_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_57_dilations_0 = const()[name = string("logits_57_dilations_0"), val = tensor([1, 1])]; + int32 logits_57_groups_0 = const()[name = string("logits_57_groups_0"), val = int32(1)]; + tensor lm_heads_14_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110175680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112272896))))[name = string("lm_heads_14_weight_to_fp16_palettized")]; + tensor logits_57_cast_fp16 = conv(dilations = logits_57_dilations_0, groups = logits_57_groups_0, pad = logits_57_pad_0, pad_type = logits_57_pad_type_0, strides = logits_57_strides_0, weight = lm_heads_14_weight_to_fp16_palettized, x = input_697_cast_fp16)[name = string("logits_57_cast_fp16")]; + tensor var_24382 = const()[name = string("op_24382"), val = tensor([1, 2048])]; + tensor logits_cast_fp16 = reshape(shape = var_24382, x = logits_57_cast_fp16)[name = string("logits_cast_fp16")]; + tensor scaled_logits_cast_fp16 = real_div(x = logits_cast_fp16, y = temperature)[name = string("scaled_logits_cast_fp16")]; + int32 var_24392 = const()[name = string("op_24392"), val = int32(100)]; + int32 top_values_axis_0 = const()[name = string("top_values_axis_0"), val = int32(1)]; + bool top_values_ascending_0 = const()[name = string("top_values_ascending_0"), val = bool(false)]; + bool top_values_sort_0 = const()[name = string("top_values_sort_0"), val = bool(true)]; + bool top_values_return_indices_0 = const()[name = string("top_values_return_indices_0"), val = bool(true)]; + string top_values_cast_fp16_cast_uint16_output_indices_dtype_0 = const()[name = string("top_values_cast_fp16_cast_uint16_output_indices_dtype_0"), val = string("uint16")]; + tensor top_values_cast_fp16_cast_uint16_0, tensor top_values_cast_fp16_cast_uint16_1 = topk(ascending = top_values_ascending_0, axis = top_values_axis_0, k = var_24392, output_indices_dtype = top_values_cast_fp16_cast_uint16_output_indices_dtype_0, return_indices = top_values_return_indices_0, sort = top_values_sort_0, x = scaled_logits_cast_fp16)[name = string("top_values_cast_fp16_cast_uint16")]; + tensor var_24398_cast_fp16 = mul(x = top_values_cast_fp16_cast_uint16_0, y = var_2996_cast_fp16_to_fp16)[name = string("op_24398_cast_fp16")]; + tensor var_24402_cast_fp16 = add(x = var_24398_cast_fp16, y = var_3001_cast_fp16)[name = string("op_24402_cast_fp16")]; + tensor reduce_min_14_axes_0 = const()[name = string("reduce_min_14_axes_0"), val = tensor([1])]; + bool reduce_min_14_keep_dims_0 = const()[name = string("reduce_min_14_keep_dims_0"), val = bool(true)]; + tensor reduce_min_14_cast_fp16 = reduce_min(axes = reduce_min_14_axes_0, keep_dims = reduce_min_14_keep_dims_0, x = var_24402_cast_fp16)[name = string("reduce_min_14_cast_fp16")]; + tensor var_24405_cast_fp16 = greater_equal(x = scaled_logits_cast_fp16, y = reduce_min_14_cast_fp16)[name = string("op_24405_cast_fp16")]; + fp16 var_24406_value_0_to_fp16 = const()[name = string("op_24406_value_0_to_fp16"), val = fp16(-0x1.d4cp+14)]; + tensor var_24406_cast_fp16 = fill_like(ref_tensor = scaled_logits_cast_fp16, value = var_24406_value_0_to_fp16)[name = string("op_24406_cast_fp16")]; + tensor masked_logits_cast_fp16 = select(a = scaled_logits_cast_fp16, b = var_24406_cast_fp16, cond = var_24405_cast_fp16)[name = string("masked_logits_cast_fp16")]; + tensor var_24410_begin_0 = const()[name = string("op_24410_begin_0"), val = tensor([14, 0])]; + tensor var_24410_end_0 = const()[name = string("op_24410_end_0"), val = tensor([15, 2048])]; + tensor var_24410_end_mask_0 = const()[name = string("op_24410_end_mask_0"), val = tensor([false, true])]; + tensor var_24410_squeeze_mask_0 = const()[name = string("op_24410_squeeze_mask_0"), val = tensor([true, false])]; + tensor var_24410_cast_fp16 = slice_by_index(begin = var_24410_begin_0, end = var_24410_end_0, end_mask = var_24410_end_mask_0, squeeze_mask = var_24410_squeeze_mask_0, x = gumbel)[name = string("op_24410_cast_fp16")]; + tensor var_24413 = const()[name = string("op_24413"), val = tensor([1, 2048])]; + tensor var_24414_cast_fp16 = reshape(shape = var_24413, x = var_24410_cast_fp16)[name = string("op_24414_cast_fp16")]; + tensor noisy_logits_cast_fp16 = add(x = masked_logits_cast_fp16, y = var_24414_cast_fp16)[name = string("noisy_logits_cast_fp16")]; + int32 code_29_axis_0 = const()[name = string("code_29_axis_0"), val = int32(1)]; + bool code_29_keep_dims_0 = const()[name = string("code_29_keep_dims_0"), val = bool(false)]; + string code_29_output_dtype_0 = const()[name = string("code_29_output_dtype_0"), val = string("int32")]; + tensor code_29_cast_fp16 = reduce_argmax(axis = code_29_axis_0, keep_dims = code_29_keep_dims_0, output_dtype = code_29_output_dtype_0, x = noisy_logits_cast_fp16)[name = string("code_29_cast_fp16")]; + int32 var_24425 = const()[name = string("op_24425"), val = int32(28672)]; + tensor input = add(x = code_29_cast_fp16, y = var_24425)[name = string("input")]; + int32 code_embed_57_axis_0 = const()[name = string("code_embed_57_axis_0"), val = int32(0)]; + int32 code_embed_57_batch_dims_0 = const()[name = string("code_embed_57_batch_dims_0"), val = int32(0)]; + bool code_embed_57_validate_indices_0 = const()[name = string("code_embed_57_validate_indices_0"), val = bool(false)]; + string input_to_uint16_dtype_0 = const()[name = string("input_to_uint16_dtype_0"), val = string("uint16")]; + tensor input_to_uint16 = cast(dtype = input_to_uint16_dtype_0, x = input)[name = string("cast_0")]; + tensor code_embed_57_cast_fp16_cast_uint16 = gather(axis = code_embed_57_axis_0, batch_dims = code_embed_57_batch_dims_0, indices = input_to_uint16, validate_indices = code_embed_57_validate_indices_0, x = codec_embedding_embedding_weight_to_fp16_palettized)[name = string("code_embed_57_cast_fp16_cast_uint16")]; + tensor var_24429 = const()[name = string("op_24429"), val = tensor([1, 2048, 1, 1])]; + tensor code_embed_cast_fp16 = reshape(shape = var_24429, x = code_embed_57_cast_fp16_cast_uint16)[name = string("code_embed_cast_fp16")]; + tensor embed_sum = add(x = embed_sum_cast_fp16, y = code_embed_cast_fp16)[name = string("op_24435_cast_fp16")]; + int32 var_24438_axis_0 = const()[name = string("op_24438_axis_0"), val = int32(1)]; + tensor codes = stack(axis = var_24438_axis_0, values = (code_1_cast_fp16, code_3_cast_fp16, code_5_cast_fp16, code_7_cast_fp16, code_9_cast_fp16, code_11_cast_fp16, code_13_cast_fp16, code_15_cast_fp16, code_17_cast_fp16, code_19_cast_fp16, code_21_cast_fp16, code_23_cast_fp16, code_25_cast_fp16, code_27_cast_fp16, code_29_cast_fp16))[name = string("op_24438")]; + } -> (codes, embed_sum); + func stepped(tensor cache_length, tensor input_embeds, tensor key_cache, tensor key_padding_mask, tensor kv_cache_update_mask, tensor value_cache) { + string inputs_1_pad_type_0 = const()[name = string("inputs_1_pad_type_0"), val = string("valid")]; + tensor inputs_1_strides_0 = const()[name = string("inputs_1_strides_0"), val = tensor([1, 1])]; + tensor inputs_1_pad_0 = const()[name = string("inputs_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor inputs_1_dilations_0 = const()[name = string("inputs_1_dilations_0"), val = tensor([1, 1])]; + int32 inputs_1_groups_0 = const()[name = string("inputs_1_groups_0"), val = int32(1)]; + tensor input_projection_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2097280))))[name = string("input_projection_weight_to_fp16_palettized")]; + tensor input_projection_bias_to_fp16 = const()[name = string("input_projection_bias_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2097856)))]; + tensor inputs_1_cast_fp16 = conv(bias = input_projection_bias_to_fp16, dilations = inputs_1_dilations_0, groups = inputs_1_groups_0, pad = inputs_1_pad_0, pad_type = inputs_1_pad_type_0, strides = inputs_1_strides_0, weight = input_projection_weight_to_fp16_palettized, x = input_embeds)[name = string("inputs_1_cast_fp16")]; + int32 pos_cos_batch_dims_0 = const()[name = string("pos_cos_batch_dims_0"), val = int32(0)]; + bool pos_cos_validate_indices_0 = const()[name = string("pos_cos_validate_indices_0"), val = bool(false)]; + tensor position_embeddings_cos_weight_to_fp16 = const()[name = string("position_embeddings_cos_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2099968)))]; + string cache_length_to_int16_dtype_0 = const()[name = string("cache_length_to_int16_dtype_0"), val = string("int16")]; + string cast_111_dtype_0 = const()[name = string("cast_111_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; + tensor cache_length_to_int16 = cast(dtype = cache_length_to_int16_dtype_0, x = cache_length)[name = string("cast_5")]; + tensor cast_111 = cast(dtype = cast_111_dtype_0, x = cache_length_to_int16)[name = string("cast_4")]; + tensor greater_equal_0 = greater_equal(x = cast_111, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; + int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(16)]; + tensor add_0 = add(x = cast_111, y = slice_by_index_0)[name = string("add_0")]; + tensor select_0 = select(a = cast_111, b = add_0, cond = greater_equal_0)[name = string("select_0")]; + string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; + string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; + tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; + tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; + tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; + int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(16)]; + tensor add_0_1 = add(x = cast_0, y = slice_by_index_0_1)[name = string("add_0_1")]; + tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; + int32 pos_cos_cast_fp16_cast_uint16_cast_uint16_axis_0 = const()[name = string("pos_cos_cast_fp16_cast_uint16_cast_uint16_axis_0"), val = int32(0)]; + tensor pos_cos_cast_fp16_cast_uint16_cast_uint16 = gather(axis = pos_cos_cast_fp16_cast_uint16_cast_uint16_axis_0, batch_dims = pos_cos_batch_dims_0, indices = select_0_1, validate_indices = pos_cos_validate_indices_0, x = position_embeddings_cos_weight_to_fp16)[name = string("pos_cos_cast_fp16_cast_uint16_cast_uint16")]; + tensor obj_7_axes_0 = const()[name = string("obj_7_axes_0"), val = tensor([2])]; + tensor obj_7_cast_fp16 = expand_dims(axes = obj_7_axes_0, x = pos_cos_cast_fp16_cast_uint16_cast_uint16)[name = string("obj_7_cast_fp16")]; + int32 pos_sin_axis_0 = const()[name = string("pos_sin_axis_0"), val = int32(0)]; + int32 pos_sin_batch_dims_0 = const()[name = string("pos_sin_batch_dims_0"), val = int32(0)]; + bool pos_sin_validate_indices_0 = const()[name = string("pos_sin_validate_indices_0"), val = bool(false)]; + tensor position_embeddings_sin_weight_to_fp16 = const()[name = string("position_embeddings_sin_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2104128)))]; + string cache_length_to_uint16_dtype_0 = const()[name = string("cache_length_to_uint16_dtype_0"), val = string("uint16")]; + tensor cache_length_to_uint16 = cast(dtype = cache_length_to_uint16_dtype_0, x = cache_length)[name = string("cast_1")]; + tensor pos_sin_cast_fp16_cast_uint16 = gather(axis = pos_sin_axis_0, batch_dims = pos_sin_batch_dims_0, indices = cache_length_to_uint16, validate_indices = pos_sin_validate_indices_0, x = position_embeddings_sin_weight_to_fp16)[name = string("pos_sin_cast_fp16_cast_uint16")]; + tensor obj_9_axes_0 = const()[name = string("obj_9_axes_0"), val = tensor([2])]; + tensor obj_9_cast_fp16 = expand_dims(axes = obj_9_axes_0, x = pos_sin_cast_fp16_cast_uint16)[name = string("obj_9_cast_fp16")]; + tensor tile_0 = const()[name = string("tile_0"), val = tensor([1024, 1024, 1024, 1024, 1024])]; + int32 var_96_axis_0 = const()[name = string("op_96_axis_0"), val = int32(1)]; + tensor var_96_cast_fp16_0, tensor var_96_cast_fp16_1, tensor var_96_cast_fp16_2, tensor var_96_cast_fp16_3, tensor var_96_cast_fp16_4 = split(axis = var_96_axis_0, split_sizes = tile_0, x = key_cache)[name = string("op_96_cast_fp16")]; + tensor tile_1 = const()[name = string("tile_1"), val = tensor([1024, 1024, 1024, 1024, 1024])]; + int32 var_104_axis_0 = const()[name = string("op_104_axis_0"), val = int32(1)]; + tensor var_104_cast_fp16_0, tensor var_104_cast_fp16_1, tensor var_104_cast_fp16_2, tensor var_104_cast_fp16_3, tensor var_104_cast_fp16_4 = split(axis = var_104_axis_0, split_sizes = tile_1, x = value_cache)[name = string("op_104_cast_fp16")]; + int32 var_111 = const()[name = string("op_111"), val = int32(3)]; + int32 var_121 = const()[name = string("op_121"), val = int32(-2)]; + int32 var_129 = const()[name = string("op_129"), val = int32(1)]; + tensor inputs_sq_1_cast_fp16 = mul(x = inputs_1_cast_fp16, y = inputs_1_cast_fp16)[name = string("inputs_sq_1_cast_fp16")]; + tensor variance_1_axes_0 = const()[name = string("variance_1_axes_0"), val = tensor([1])]; + bool variance_1_keep_dims_0 = const()[name = string("variance_1_keep_dims_0"), val = bool(true)]; + tensor variance_1_cast_fp16 = reduce_mean(axes = variance_1_axes_0, keep_dims = variance_1_keep_dims_0, x = inputs_sq_1_cast_fp16)[name = string("variance_1_cast_fp16")]; + fp16 var_141_to_fp16 = const()[name = string("op_141_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_142_cast_fp16 = add(x = variance_1_cast_fp16, y = var_141_to_fp16)[name = string("op_142_cast_fp16")]; + fp32 var_143_epsilon_0 = const()[name = string("op_143_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_143_cast_fp16 = rsqrt(epsilon = var_143_epsilon_0, x = var_142_cast_fp16)[name = string("op_143_cast_fp16")]; + tensor hidden_states_1_cast_fp16 = mul(x = inputs_1_cast_fp16, y = var_143_cast_fp16)[name = string("hidden_states_1_cast_fp16")]; + tensor w_1_to_fp16 = const()[name = string("w_1_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2108288)))]; + tensor obj_1_cast_fp16 = mul(x = w_1_to_fp16, y = hidden_states_1_cast_fp16)[name = string("obj_1_cast_fp16")]; + string query_1_pad_type_0 = const()[name = string("query_1_pad_type_0"), val = string("valid")]; + tensor query_1_strides_0 = const()[name = string("query_1_strides_0"), val = tensor([1, 1])]; + tensor query_1_pad_0 = const()[name = string("query_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_1_dilations_0 = const()[name = string("query_1_dilations_0"), val = tensor([1, 1])]; + int32 query_1_groups_0 = const()[name = string("query_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2110400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4207616))))[name = string("layers_0_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor layers_0_self_attn_q_proj_bias_to_fp16 = const()[name = string("layers_0_self_attn_q_proj_bias_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4208192)))]; + tensor query_1_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_1_dilations_0, groups = query_1_groups_0, pad = query_1_pad_0, pad_type = query_1_pad_type_0, strides = query_1_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16_palettized, x = obj_1_cast_fp16)[name = string("query_1_cast_fp16")]; + string current_key_1_pad_type_0 = const()[name = string("current_key_1_pad_type_0"), val = string("valid")]; + tensor current_key_1_strides_0 = const()[name = string("current_key_1_strides_0"), val = tensor([1, 1])]; + tensor current_key_1_pad_0 = const()[name = string("current_key_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_1_dilations_0 = const()[name = string("current_key_1_dilations_0"), val = tensor([1, 1])]; + int32 current_key_1_groups_0 = const()[name = string("current_key_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4212352))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5260992))))[name = string("layers_0_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_1_cast_fp16 = conv(dilations = current_key_1_dilations_0, groups = current_key_1_groups_0, pad = current_key_1_pad_0, pad_type = current_key_1_pad_type_0, strides = current_key_1_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16_palettized, x = obj_1_cast_fp16)[name = string("current_key_1_cast_fp16")]; + string current_value_1_pad_type_0 = const()[name = string("current_value_1_pad_type_0"), val = string("valid")]; + tensor current_value_1_strides_0 = const()[name = string("current_value_1_strides_0"), val = tensor([1, 1])]; + tensor current_value_1_pad_0 = const()[name = string("current_value_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_1_dilations_0 = const()[name = string("current_value_1_dilations_0"), val = tensor([1, 1])]; + int32 current_value_1_groups_0 = const()[name = string("current_value_1_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(5261568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6310208))))[name = string("layers_0_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor layers_0_self_attn_v_proj_bias_to_fp16 = const()[name = string("layers_0_self_attn_v_proj_bias_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6310784)))]; + tensor current_value_1_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_1_dilations_0, groups = current_value_1_groups_0, pad = current_value_1_pad_0, pad_type = current_value_1_pad_type_0, strides = current_value_1_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16_palettized, x = obj_1_cast_fp16)[name = string("current_value_1_cast_fp16")]; + tensor var_180 = const()[name = string("op_180"), val = tensor([16, 128, 1, 1])]; + tensor inputs_3_cast_fp16 = reshape(shape = var_180, x = query_1_cast_fp16)[name = string("inputs_3_cast_fp16")]; + tensor inputs_sq_3_cast_fp16 = mul(x = inputs_3_cast_fp16, y = inputs_3_cast_fp16)[name = string("inputs_sq_3_cast_fp16")]; + tensor variance_3_axes_0 = const()[name = string("variance_3_axes_0"), val = tensor([1])]; + bool variance_3_keep_dims_0 = const()[name = string("variance_3_keep_dims_0"), val = bool(true)]; + tensor variance_3_cast_fp16 = reduce_mean(axes = variance_3_axes_0, keep_dims = variance_3_keep_dims_0, x = inputs_sq_3_cast_fp16)[name = string("variance_3_cast_fp16")]; + fp16 var_186_to_fp16 = const()[name = string("op_186_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_187_cast_fp16 = add(x = variance_3_cast_fp16, y = var_186_to_fp16)[name = string("op_187_cast_fp16")]; + fp32 var_188_epsilon_0 = const()[name = string("op_188_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_188_cast_fp16 = rsqrt(epsilon = var_188_epsilon_0, x = var_187_cast_fp16)[name = string("op_188_cast_fp16")]; + tensor hidden_states_3_cast_fp16 = mul(x = inputs_3_cast_fp16, y = var_188_cast_fp16)[name = string("hidden_states_3_cast_fp16")]; + tensor w_3_to_fp16 = const()[name = string("w_3_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6312896)))]; + tensor query_normed_1_cast_fp16 = mul(x = w_3_to_fp16, y = hidden_states_3_cast_fp16)[name = string("query_normed_1_cast_fp16")]; + tensor var_196 = const()[name = string("op_196"), val = tensor([8, 128, 1, 1])]; + tensor inputs_5_cast_fp16 = reshape(shape = var_196, x = current_key_1_cast_fp16)[name = string("inputs_5_cast_fp16")]; + tensor inputs_sq_5_cast_fp16 = mul(x = inputs_5_cast_fp16, y = inputs_5_cast_fp16)[name = string("inputs_sq_5_cast_fp16")]; + tensor variance_5_axes_0 = const()[name = string("variance_5_axes_0"), val = tensor([1])]; + bool variance_5_keep_dims_0 = const()[name = string("variance_5_keep_dims_0"), val = bool(true)]; + tensor variance_5_cast_fp16 = reduce_mean(axes = variance_5_axes_0, keep_dims = variance_5_keep_dims_0, x = inputs_sq_5_cast_fp16)[name = string("variance_5_cast_fp16")]; + fp16 var_202_to_fp16 = const()[name = string("op_202_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_203_cast_fp16 = add(x = variance_5_cast_fp16, y = var_202_to_fp16)[name = string("op_203_cast_fp16")]; + fp32 var_204_epsilon_0 = const()[name = string("op_204_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_204_cast_fp16 = rsqrt(epsilon = var_204_epsilon_0, x = var_203_cast_fp16)[name = string("op_204_cast_fp16")]; + tensor hidden_states_5_cast_fp16 = mul(x = inputs_5_cast_fp16, y = var_204_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; + tensor w_5_to_fp16 = const()[name = string("w_5_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6313216)))]; + tensor current_key_normed_1_cast_fp16 = mul(x = w_5_to_fp16, y = hidden_states_5_cast_fp16)[name = string("current_key_normed_1_cast_fp16")]; + tensor var_222 = const()[name = string("op_222"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_1_cast_fp16 = reshape(shape = var_222, x = query_normed_1_cast_fp16)[name = string("mh_q_1_cast_fp16")]; + tensor var_224 = const()[name = string("op_224"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_1_cast_fp16 = reshape(shape = var_224, x = current_key_normed_1_cast_fp16)[name = string("mh_k_1_cast_fp16")]; + tensor cos_1_axes_0 = const()[name = string("cos_1_axes_0"), val = tensor([1])]; + tensor cos_1_cast_fp16 = expand_dims(axes = cos_1_axes_0, x = obj_7_cast_fp16)[name = string("cos_1_cast_fp16")]; + tensor sin_1_axes_0 = const()[name = string("sin_1_axes_0"), val = tensor([1])]; + tensor sin_1_cast_fp16 = expand_dims(axes = sin_1_axes_0, x = obj_9_cast_fp16)[name = string("sin_1_cast_fp16")]; + tensor var_228_cast_fp16 = mul(x = mh_q_1_cast_fp16, y = cos_1_cast_fp16)[name = string("op_228_cast_fp16")]; + tensor var_233_begin_0 = const()[name = string("op_233_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_233_end_0 = const()[name = string("op_233_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_233_end_mask_0 = const()[name = string("op_233_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_233_cast_fp16 = slice_by_index(begin = var_233_begin_0, end = var_233_end_0, end_mask = var_233_end_mask_0, x = mh_q_1_cast_fp16)[name = string("op_233_cast_fp16")]; + tensor var_239_begin_0 = const()[name = string("op_239_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_239_end_0 = const()[name = string("op_239_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_239_end_mask_0 = const()[name = string("op_239_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_239_cast_fp16 = slice_by_index(begin = var_239_begin_0, end = var_239_end_0, end_mask = var_239_end_mask_0, x = mh_q_1_cast_fp16)[name = string("op_239_cast_fp16")]; + fp16 const_17_promoted_to_fp16 = const()[name = string("const_17_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_241_cast_fp16 = mul(x = var_239_cast_fp16, y = const_17_promoted_to_fp16)[name = string("op_241_cast_fp16")]; + bool var_243_interleave_0 = const()[name = string("op_243_interleave_0"), val = bool(false)]; + tensor var_243_cast_fp16 = concat(axis = var_121, interleave = var_243_interleave_0, values = (var_241_cast_fp16, var_233_cast_fp16))[name = string("op_243_cast_fp16")]; + tensor var_244_cast_fp16 = mul(x = var_243_cast_fp16, y = sin_1_cast_fp16)[name = string("op_244_cast_fp16")]; + tensor mh_q_3_cast_fp16 = add(x = var_228_cast_fp16, y = var_244_cast_fp16)[name = string("mh_q_3_cast_fp16")]; + tensor var_246_cast_fp16 = mul(x = mh_k_1_cast_fp16, y = cos_1_cast_fp16)[name = string("op_246_cast_fp16")]; + tensor var_251_begin_0 = const()[name = string("op_251_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_251_end_0 = const()[name = string("op_251_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_251_end_mask_0 = const()[name = string("op_251_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_251_cast_fp16 = slice_by_index(begin = var_251_begin_0, end = var_251_end_0, end_mask = var_251_end_mask_0, x = mh_k_1_cast_fp16)[name = string("op_251_cast_fp16")]; + tensor var_257_begin_0 = const()[name = string("op_257_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_257_end_0 = const()[name = string("op_257_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_257_end_mask_0 = const()[name = string("op_257_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_257_cast_fp16 = slice_by_index(begin = var_257_begin_0, end = var_257_end_0, end_mask = var_257_end_mask_0, x = mh_k_1_cast_fp16)[name = string("op_257_cast_fp16")]; + fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_259_cast_fp16 = mul(x = var_257_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_259_cast_fp16")]; + bool var_261_interleave_0 = const()[name = string("op_261_interleave_0"), val = bool(false)]; + tensor var_261_cast_fp16 = concat(axis = var_121, interleave = var_261_interleave_0, values = (var_259_cast_fp16, var_251_cast_fp16))[name = string("op_261_cast_fp16")]; + tensor var_262_cast_fp16 = mul(x = var_261_cast_fp16, y = sin_1_cast_fp16)[name = string("op_262_cast_fp16")]; + tensor mh_k_3_cast_fp16 = add(x = var_246_cast_fp16, y = var_262_cast_fp16)[name = string("mh_k_3_cast_fp16")]; + tensor var_266 = const()[name = string("op_266"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_3_cast_fp16 = reshape(shape = var_266, x = mh_k_3_cast_fp16)[name = string("current_key_3_cast_fp16")]; + tensor var_269_axes_0 = const()[name = string("op_269_axes_0"), val = tensor([1])]; + tensor var_269_cast_fp16 = expand_dims(axes = var_269_axes_0, x = kv_cache_update_mask)[name = string("op_269_cast_fp16")]; + tensor var_270_axes_0 = const()[name = string("op_270_axes_0"), val = tensor([2])]; + tensor var_270_cast_fp16 = expand_dims(axes = var_270_axes_0, x = var_269_cast_fp16)[name = string("op_270_cast_fp16")]; + fp16 var_122_to_fp16 = const()[name = string("op_122_to_fp16"), val = fp16(0x1p+0)]; + tensor var_272_cast_fp16 = sub(x = var_122_to_fp16, y = var_270_cast_fp16)[name = string("op_272_cast_fp16")]; + tensor var_273_cast_fp16 = mul(x = var_96_cast_fp16_0, y = var_272_cast_fp16)[name = string("op_273_cast_fp16")]; + tensor var_274_cast_fp16 = mul(x = current_key_3_cast_fp16, y = var_270_cast_fp16)[name = string("op_274_cast_fp16")]; + tensor key_3_cast_fp16 = add(x = var_273_cast_fp16, y = var_274_cast_fp16)[name = string("key_3_cast_fp16")]; + tensor var_277_cast_fp16 = mul(x = var_104_cast_fp16_0, y = var_272_cast_fp16)[name = string("op_277_cast_fp16")]; + tensor var_278_cast_fp16 = mul(x = current_value_1_cast_fp16, y = var_270_cast_fp16)[name = string("op_278_cast_fp16")]; + tensor value_1_cast_fp16 = add(x = var_277_cast_fp16, y = var_278_cast_fp16)[name = string("value_1_cast_fp16")]; + tensor var_282 = const()[name = string("op_282"), val = tensor([1, 8, 128, 16])]; + tensor key_heads_1_cast_fp16 = reshape(shape = var_282, x = key_3_cast_fp16)[name = string("key_heads_1_cast_fp16")]; + tensor var_284 = const()[name = string("op_284"), val = tensor([1, 8, 128, 16])]; + tensor value_heads_1_cast_fp16 = reshape(shape = var_284, x = value_1_cast_fp16)[name = string("value_heads_1_cast_fp16")]; + tensor var_287_begin_0 = const()[name = string("op_287_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_287_end_0 = const()[name = string("op_287_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_287_end_mask_0 = const()[name = string("op_287_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_287_cast_fp16 = slice_by_index(begin = var_287_begin_0, end = var_287_end_0, end_mask = var_287_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_287_cast_fp16")]; + tensor var_291_begin_0 = const()[name = string("op_291_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_291_end_0 = const()[name = string("op_291_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_291_end_mask_0 = const()[name = string("op_291_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_291_cast_fp16 = slice_by_index(begin = var_291_begin_0, end = var_291_end_0, end_mask = var_291_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_291_cast_fp16")]; + tensor var_303_begin_0 = const()[name = string("op_303_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_303_end_0 = const()[name = string("op_303_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_303_end_mask_0 = const()[name = string("op_303_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_303_cast_fp16 = slice_by_index(begin = var_303_begin_0, end = var_303_end_0, end_mask = var_303_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_303_cast_fp16")]; + tensor var_307_begin_0 = const()[name = string("op_307_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_307_end_0 = const()[name = string("op_307_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_307_end_mask_0 = const()[name = string("op_307_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_307_cast_fp16 = slice_by_index(begin = var_307_begin_0, end = var_307_end_0, end_mask = var_307_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_307_cast_fp16")]; + tensor var_319_begin_0 = const()[name = string("op_319_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_319_end_0 = const()[name = string("op_319_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_319_end_mask_0 = const()[name = string("op_319_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_319_cast_fp16 = slice_by_index(begin = var_319_begin_0, end = var_319_end_0, end_mask = var_319_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_319_cast_fp16")]; + tensor var_323_begin_0 = const()[name = string("op_323_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_323_end_0 = const()[name = string("op_323_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_323_end_mask_0 = const()[name = string("op_323_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_323_cast_fp16 = slice_by_index(begin = var_323_begin_0, end = var_323_end_0, end_mask = var_323_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_323_cast_fp16")]; + tensor var_335_begin_0 = const()[name = string("op_335_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_335_end_0 = const()[name = string("op_335_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_335_end_mask_0 = const()[name = string("op_335_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_335_cast_fp16 = slice_by_index(begin = var_335_begin_0, end = var_335_end_0, end_mask = var_335_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_335_cast_fp16")]; + tensor var_339_begin_0 = const()[name = string("op_339_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_339_end_0 = const()[name = string("op_339_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_339_end_mask_0 = const()[name = string("op_339_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_339_cast_fp16 = slice_by_index(begin = var_339_begin_0, end = var_339_end_0, end_mask = var_339_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_339_cast_fp16")]; + tensor var_351_begin_0 = const()[name = string("op_351_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_351_end_0 = const()[name = string("op_351_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_351_end_mask_0 = const()[name = string("op_351_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_351_cast_fp16 = slice_by_index(begin = var_351_begin_0, end = var_351_end_0, end_mask = var_351_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_351_cast_fp16")]; + tensor var_355_begin_0 = const()[name = string("op_355_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_355_end_0 = const()[name = string("op_355_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_355_end_mask_0 = const()[name = string("op_355_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_355_cast_fp16 = slice_by_index(begin = var_355_begin_0, end = var_355_end_0, end_mask = var_355_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_355_cast_fp16")]; + tensor var_367_begin_0 = const()[name = string("op_367_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_367_end_0 = const()[name = string("op_367_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_367_end_mask_0 = const()[name = string("op_367_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_367_cast_fp16 = slice_by_index(begin = var_367_begin_0, end = var_367_end_0, end_mask = var_367_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_367_cast_fp16")]; + tensor var_371_begin_0 = const()[name = string("op_371_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_371_end_0 = const()[name = string("op_371_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_371_end_mask_0 = const()[name = string("op_371_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_371_cast_fp16 = slice_by_index(begin = var_371_begin_0, end = var_371_end_0, end_mask = var_371_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_371_cast_fp16")]; + tensor var_383_begin_0 = const()[name = string("op_383_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_383_end_0 = const()[name = string("op_383_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_383_end_mask_0 = const()[name = string("op_383_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_383_cast_fp16 = slice_by_index(begin = var_383_begin_0, end = var_383_end_0, end_mask = var_383_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_383_cast_fp16")]; + tensor var_387_begin_0 = const()[name = string("op_387_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_387_end_0 = const()[name = string("op_387_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_387_end_mask_0 = const()[name = string("op_387_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_387_cast_fp16 = slice_by_index(begin = var_387_begin_0, end = var_387_end_0, end_mask = var_387_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_387_cast_fp16")]; + tensor var_399_begin_0 = const()[name = string("op_399_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_399_end_0 = const()[name = string("op_399_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_399_end_mask_0 = const()[name = string("op_399_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_399_cast_fp16 = slice_by_index(begin = var_399_begin_0, end = var_399_end_0, end_mask = var_399_end_mask_0, x = key_heads_1_cast_fp16)[name = string("op_399_cast_fp16")]; + tensor var_403_begin_0 = const()[name = string("op_403_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_403_end_0 = const()[name = string("op_403_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_403_end_mask_0 = const()[name = string("op_403_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_403_cast_fp16 = slice_by_index(begin = var_403_begin_0, end = var_403_end_0, end_mask = var_403_end_mask_0, x = value_heads_1_cast_fp16)[name = string("op_403_cast_fp16")]; + bool key_heads_3_interleave_0 = const()[name = string("key_heads_3_interleave_0"), val = bool(false)]; + tensor key_heads_3_cast_fp16 = concat(axis = var_129, interleave = key_heads_3_interleave_0, values = (var_287_cast_fp16, var_287_cast_fp16, var_303_cast_fp16, var_303_cast_fp16, var_319_cast_fp16, var_319_cast_fp16, var_335_cast_fp16, var_335_cast_fp16, var_351_cast_fp16, var_351_cast_fp16, var_367_cast_fp16, var_367_cast_fp16, var_383_cast_fp16, var_383_cast_fp16, var_399_cast_fp16, var_399_cast_fp16))[name = string("key_heads_3_cast_fp16")]; + bool value_heads_3_interleave_0 = const()[name = string("value_heads_3_interleave_0"), val = bool(false)]; + tensor value_heads_3_cast_fp16 = concat(axis = var_129, interleave = value_heads_3_interleave_0, values = (var_291_cast_fp16, var_291_cast_fp16, var_307_cast_fp16, var_307_cast_fp16, var_323_cast_fp16, var_323_cast_fp16, var_339_cast_fp16, var_339_cast_fp16, var_355_cast_fp16, var_355_cast_fp16, var_371_cast_fp16, var_371_cast_fp16, var_387_cast_fp16, var_387_cast_fp16, var_403_cast_fp16, var_403_cast_fp16))[name = string("value_heads_3_cast_fp16")]; + fp16 var_426_to_fp16 = const()[name = string("op_426_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_427_cast_fp16 = mul(x = mh_q_3_cast_fp16, y = var_426_to_fp16)[name = string("op_427_cast_fp16")]; + bool mh_w_1_transpose_x_0 = const()[name = string("mh_w_1_transpose_x_0"), val = bool(true)]; + bool mh_w_1_transpose_y_0 = const()[name = string("mh_w_1_transpose_y_0"), val = bool(false)]; + tensor mh_w_1_cast_fp16 = matmul(transpose_x = mh_w_1_transpose_x_0, transpose_y = mh_w_1_transpose_y_0, x = var_427_cast_fp16, y = key_heads_3_cast_fp16)[name = string("mh_w_1_cast_fp16")]; + tensor var_435_axes_0 = const()[name = string("op_435_axes_0"), val = tensor([1])]; + tensor var_435_cast_fp16 = expand_dims(axes = var_435_axes_0, x = key_padding_mask)[name = string("op_435_cast_fp16")]; + tensor var_436_axes_0 = const()[name = string("op_436_axes_0"), val = tensor([2])]; + tensor var_436_cast_fp16 = expand_dims(axes = var_436_axes_0, x = var_435_cast_fp16)[name = string("op_436_cast_fp16")]; + tensor mh_w_3_cast_fp16 = add(x = mh_w_1_cast_fp16, y = var_436_cast_fp16)[name = string("mh_w_3_cast_fp16")]; + tensor var_439_cast_fp16 = softmax(axis = var_111, x = mh_w_3_cast_fp16)[name = string("op_439_cast_fp16")]; + bool attn_1_transpose_x_0 = const()[name = string("attn_1_transpose_x_0"), val = bool(false)]; + bool attn_1_transpose_y_0 = const()[name = string("attn_1_transpose_y_0"), val = bool(true)]; + tensor attn_1_cast_fp16 = matmul(transpose_x = attn_1_transpose_x_0, transpose_y = attn_1_transpose_y_0, x = value_heads_3_cast_fp16, y = var_439_cast_fp16)[name = string("attn_1_cast_fp16")]; + tensor var_444 = const()[name = string("op_444"), val = tensor([1, -1, 1, 1])]; + tensor input_1_cast_fp16 = reshape(shape = var_444, x = attn_1_cast_fp16)[name = string("input_1_cast_fp16")]; + string obj_11_pad_type_0 = const()[name = string("obj_11_pad_type_0"), val = string("valid")]; + tensor obj_11_strides_0 = const()[name = string("obj_11_strides_0"), val = tensor([1, 1])]; + tensor obj_11_pad_0 = const()[name = string("obj_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_11_dilations_0 = const()[name = string("obj_11_dilations_0"), val = tensor([1, 1])]; + int32 obj_11_groups_0 = const()[name = string("obj_11_groups_0"), val = int32(1)]; + tensor layers_0_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6313536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8410752))))[name = string("layers_0_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_11_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_11_dilations_0, groups = obj_11_groups_0, pad = obj_11_pad_0, pad_type = obj_11_pad_type_0, strides = obj_11_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16_palettized, x = input_1_cast_fp16)[name = string("obj_11_cast_fp16")]; + tensor inputs_7_cast_fp16 = add(x = inputs_1_cast_fp16, y = obj_11_cast_fp16)[name = string("inputs_7_cast_fp16")]; + tensor inputs_sq_7_cast_fp16 = mul(x = inputs_7_cast_fp16, y = inputs_7_cast_fp16)[name = string("inputs_sq_7_cast_fp16")]; + tensor variance_7_axes_0 = const()[name = string("variance_7_axes_0"), val = tensor([1])]; + bool variance_7_keep_dims_0 = const()[name = string("variance_7_keep_dims_0"), val = bool(true)]; + tensor variance_7_cast_fp16 = reduce_mean(axes = variance_7_axes_0, keep_dims = variance_7_keep_dims_0, x = inputs_sq_7_cast_fp16)[name = string("variance_7_cast_fp16")]; + fp16 var_462_to_fp16 = const()[name = string("op_462_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_463_cast_fp16 = add(x = variance_7_cast_fp16, y = var_462_to_fp16)[name = string("op_463_cast_fp16")]; + fp32 var_464_epsilon_0 = const()[name = string("op_464_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_464_cast_fp16 = rsqrt(epsilon = var_464_epsilon_0, x = var_463_cast_fp16)[name = string("op_464_cast_fp16")]; + tensor hidden_states_7_cast_fp16 = mul(x = inputs_7_cast_fp16, y = var_464_cast_fp16)[name = string("hidden_states_7_cast_fp16")]; + tensor w_7_to_fp16 = const()[name = string("w_7_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8411328)))]; + tensor input_3_cast_fp16 = mul(x = w_7_to_fp16, y = hidden_states_7_cast_fp16)[name = string("input_3_cast_fp16")]; + string input_5_pad_type_0 = const()[name = string("input_5_pad_type_0"), val = string("valid")]; + tensor input_5_strides_0 = const()[name = string("input_5_strides_0"), val = tensor([1, 1])]; + tensor input_5_pad_0 = const()[name = string("input_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_5_dilations_0 = const()[name = string("input_5_dilations_0"), val = tensor([1, 1])]; + int32 input_5_groups_0 = const()[name = string("input_5_groups_0"), val = int32(1)]; + tensor layers_0_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8413440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11559232))))[name = string("layers_0_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_5_cast_fp16 = conv(dilations = input_5_dilations_0, groups = input_5_groups_0, pad = input_5_pad_0, pad_type = input_5_pad_type_0, strides = input_5_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16_palettized, x = input_3_cast_fp16)[name = string("input_5_cast_fp16")]; + tensor var_478_cast_fp16 = silu(x = input_5_cast_fp16)[name = string("op_478_cast_fp16")]; + string var_484_pad_type_0 = const()[name = string("op_484_pad_type_0"), val = string("valid")]; + tensor var_484_strides_0 = const()[name = string("op_484_strides_0"), val = tensor([1, 1])]; + tensor var_484_pad_0 = const()[name = string("op_484_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_484_dilations_0 = const()[name = string("op_484_dilations_0"), val = tensor([1, 1])]; + int32 var_484_groups_0 = const()[name = string("op_484_groups_0"), val = int32(1)]; + tensor layers_0_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11559808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14705600))))[name = string("layers_0_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_484_cast_fp16 = conv(dilations = var_484_dilations_0, groups = var_484_groups_0, pad = var_484_pad_0, pad_type = var_484_pad_type_0, strides = var_484_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16_palettized, x = input_3_cast_fp16)[name = string("op_484_cast_fp16")]; + tensor input_7_cast_fp16 = mul(x = var_478_cast_fp16, y = var_484_cast_fp16)[name = string("input_7_cast_fp16")]; + string hidden_states_9_pad_type_0 = const()[name = string("hidden_states_9_pad_type_0"), val = string("valid")]; + tensor hidden_states_9_strides_0 = const()[name = string("hidden_states_9_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_9_pad_0 = const()[name = string("hidden_states_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_9_dilations_0 = const()[name = string("hidden_states_9_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_9_groups_0 = const()[name = string("hidden_states_9_groups_0"), val = int32(1)]; + tensor layers_0_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(14706176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17851968))))[name = string("layers_0_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_9_cast_fp16 = conv(dilations = hidden_states_9_dilations_0, groups = hidden_states_9_groups_0, pad = hidden_states_9_pad_0, pad_type = hidden_states_9_pad_type_0, strides = hidden_states_9_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16_palettized, x = input_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; + tensor inputs_9_cast_fp16 = add(x = inputs_7_cast_fp16, y = hidden_states_9_cast_fp16)[name = string("inputs_9_cast_fp16")]; + int32 var_498 = const()[name = string("op_498"), val = int32(3)]; + int32 var_508 = const()[name = string("op_508"), val = int32(-2)]; + int32 var_516 = const()[name = string("op_516"), val = int32(1)]; + tensor inputs_sq_9_cast_fp16 = mul(x = inputs_9_cast_fp16, y = inputs_9_cast_fp16)[name = string("inputs_sq_9_cast_fp16")]; + tensor variance_9_axes_0 = const()[name = string("variance_9_axes_0"), val = tensor([1])]; + bool variance_9_keep_dims_0 = const()[name = string("variance_9_keep_dims_0"), val = bool(true)]; + tensor variance_9_cast_fp16 = reduce_mean(axes = variance_9_axes_0, keep_dims = variance_9_keep_dims_0, x = inputs_sq_9_cast_fp16)[name = string("variance_9_cast_fp16")]; + fp16 var_528_to_fp16 = const()[name = string("op_528_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_529_cast_fp16 = add(x = variance_9_cast_fp16, y = var_528_to_fp16)[name = string("op_529_cast_fp16")]; + fp32 var_530_epsilon_0 = const()[name = string("op_530_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_530_cast_fp16 = rsqrt(epsilon = var_530_epsilon_0, x = var_529_cast_fp16)[name = string("op_530_cast_fp16")]; + tensor hidden_states_11_cast_fp16 = mul(x = inputs_9_cast_fp16, y = var_530_cast_fp16)[name = string("hidden_states_11_cast_fp16")]; + tensor w_9_to_fp16 = const()[name = string("w_9_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17852544)))]; + tensor obj_13_cast_fp16 = mul(x = w_9_to_fp16, y = hidden_states_11_cast_fp16)[name = string("obj_13_cast_fp16")]; + string query_7_pad_type_0 = const()[name = string("query_7_pad_type_0"), val = string("valid")]; + tensor query_7_strides_0 = const()[name = string("query_7_strides_0"), val = tensor([1, 1])]; + tensor query_7_pad_0 = const()[name = string("query_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_7_dilations_0 = const()[name = string("query_7_dilations_0"), val = tensor([1, 1])]; + int32 query_7_groups_0 = const()[name = string("query_7_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(17854656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19951872))))[name = string("layers_1_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor query_7_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_7_dilations_0, groups = query_7_groups_0, pad = query_7_pad_0, pad_type = query_7_pad_type_0, strides = query_7_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16_palettized, x = obj_13_cast_fp16)[name = string("query_7_cast_fp16")]; + string current_key_5_pad_type_0 = const()[name = string("current_key_5_pad_type_0"), val = string("valid")]; + tensor current_key_5_strides_0 = const()[name = string("current_key_5_strides_0"), val = tensor([1, 1])]; + tensor current_key_5_pad_0 = const()[name = string("current_key_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_5_dilations_0 = const()[name = string("current_key_5_dilations_0"), val = tensor([1, 1])]; + int32 current_key_5_groups_0 = const()[name = string("current_key_5_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19952448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21001088))))[name = string("layers_1_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_5_cast_fp16 = conv(dilations = current_key_5_dilations_0, groups = current_key_5_groups_0, pad = current_key_5_pad_0, pad_type = current_key_5_pad_type_0, strides = current_key_5_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16_palettized, x = obj_13_cast_fp16)[name = string("current_key_5_cast_fp16")]; + string current_value_3_pad_type_0 = const()[name = string("current_value_3_pad_type_0"), val = string("valid")]; + tensor current_value_3_strides_0 = const()[name = string("current_value_3_strides_0"), val = tensor([1, 1])]; + tensor current_value_3_pad_0 = const()[name = string("current_value_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_3_dilations_0 = const()[name = string("current_value_3_dilations_0"), val = tensor([1, 1])]; + int32 current_value_3_groups_0 = const()[name = string("current_value_3_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21001664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22050304))))[name = string("layers_1_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor current_value_3_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_3_dilations_0, groups = current_value_3_groups_0, pad = current_value_3_pad_0, pad_type = current_value_3_pad_type_0, strides = current_value_3_strides_0, weight = layers_1_self_attn_v_proj_weight_to_fp16_palettized, x = obj_13_cast_fp16)[name = string("current_value_3_cast_fp16")]; + tensor var_567 = const()[name = string("op_567"), val = tensor([16, 128, 1, 1])]; + tensor inputs_11_cast_fp16 = reshape(shape = var_567, x = query_7_cast_fp16)[name = string("inputs_11_cast_fp16")]; + tensor inputs_sq_11_cast_fp16 = mul(x = inputs_11_cast_fp16, y = inputs_11_cast_fp16)[name = string("inputs_sq_11_cast_fp16")]; + tensor variance_11_axes_0 = const()[name = string("variance_11_axes_0"), val = tensor([1])]; + bool variance_11_keep_dims_0 = const()[name = string("variance_11_keep_dims_0"), val = bool(true)]; + tensor variance_11_cast_fp16 = reduce_mean(axes = variance_11_axes_0, keep_dims = variance_11_keep_dims_0, x = inputs_sq_11_cast_fp16)[name = string("variance_11_cast_fp16")]; + fp16 var_573_to_fp16 = const()[name = string("op_573_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_574_cast_fp16 = add(x = variance_11_cast_fp16, y = var_573_to_fp16)[name = string("op_574_cast_fp16")]; + fp32 var_575_epsilon_0 = const()[name = string("op_575_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_575_cast_fp16 = rsqrt(epsilon = var_575_epsilon_0, x = var_574_cast_fp16)[name = string("op_575_cast_fp16")]; + tensor hidden_states_13_cast_fp16 = mul(x = inputs_11_cast_fp16, y = var_575_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; + tensor w_11_to_fp16 = const()[name = string("w_11_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22050880)))]; + tensor query_normed_3_cast_fp16 = mul(x = w_11_to_fp16, y = hidden_states_13_cast_fp16)[name = string("query_normed_3_cast_fp16")]; + tensor var_583 = const()[name = string("op_583"), val = tensor([8, 128, 1, 1])]; + tensor inputs_13_cast_fp16 = reshape(shape = var_583, x = current_key_5_cast_fp16)[name = string("inputs_13_cast_fp16")]; + tensor inputs_sq_13_cast_fp16 = mul(x = inputs_13_cast_fp16, y = inputs_13_cast_fp16)[name = string("inputs_sq_13_cast_fp16")]; + tensor variance_13_axes_0 = const()[name = string("variance_13_axes_0"), val = tensor([1])]; + bool variance_13_keep_dims_0 = const()[name = string("variance_13_keep_dims_0"), val = bool(true)]; + tensor variance_13_cast_fp16 = reduce_mean(axes = variance_13_axes_0, keep_dims = variance_13_keep_dims_0, x = inputs_sq_13_cast_fp16)[name = string("variance_13_cast_fp16")]; + fp16 var_589_to_fp16 = const()[name = string("op_589_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_590_cast_fp16 = add(x = variance_13_cast_fp16, y = var_589_to_fp16)[name = string("op_590_cast_fp16")]; + fp32 var_591_epsilon_0 = const()[name = string("op_591_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_591_cast_fp16 = rsqrt(epsilon = var_591_epsilon_0, x = var_590_cast_fp16)[name = string("op_591_cast_fp16")]; + tensor hidden_states_15_cast_fp16 = mul(x = inputs_13_cast_fp16, y = var_591_cast_fp16)[name = string("hidden_states_15_cast_fp16")]; + tensor w_13_to_fp16 = const()[name = string("w_13_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22051200)))]; + tensor current_key_normed_3_cast_fp16 = mul(x = w_13_to_fp16, y = hidden_states_15_cast_fp16)[name = string("current_key_normed_3_cast_fp16")]; + tensor var_609 = const()[name = string("op_609"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_7_cast_fp16 = reshape(shape = var_609, x = query_normed_3_cast_fp16)[name = string("mh_q_7_cast_fp16")]; + tensor var_611 = const()[name = string("op_611"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_5_cast_fp16 = reshape(shape = var_611, x = current_key_normed_3_cast_fp16)[name = string("mh_k_5_cast_fp16")]; + tensor var_615_cast_fp16 = mul(x = mh_q_7_cast_fp16, y = cos_1_cast_fp16)[name = string("op_615_cast_fp16")]; + tensor var_620_begin_0 = const()[name = string("op_620_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_620_end_0 = const()[name = string("op_620_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_620_end_mask_0 = const()[name = string("op_620_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_620_cast_fp16 = slice_by_index(begin = var_620_begin_0, end = var_620_end_0, end_mask = var_620_end_mask_0, x = mh_q_7_cast_fp16)[name = string("op_620_cast_fp16")]; + tensor var_626_begin_0 = const()[name = string("op_626_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_626_end_0 = const()[name = string("op_626_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_626_end_mask_0 = const()[name = string("op_626_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_626_cast_fp16 = slice_by_index(begin = var_626_begin_0, end = var_626_end_0, end_mask = var_626_end_mask_0, x = mh_q_7_cast_fp16)[name = string("op_626_cast_fp16")]; + fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_628_cast_fp16 = mul(x = var_626_cast_fp16, y = const_40_promoted_to_fp16)[name = string("op_628_cast_fp16")]; + bool var_630_interleave_0 = const()[name = string("op_630_interleave_0"), val = bool(false)]; + tensor var_630_cast_fp16 = concat(axis = var_508, interleave = var_630_interleave_0, values = (var_628_cast_fp16, var_620_cast_fp16))[name = string("op_630_cast_fp16")]; + tensor var_631_cast_fp16 = mul(x = var_630_cast_fp16, y = sin_1_cast_fp16)[name = string("op_631_cast_fp16")]; + tensor mh_q_9_cast_fp16 = add(x = var_615_cast_fp16, y = var_631_cast_fp16)[name = string("mh_q_9_cast_fp16")]; + tensor var_633_cast_fp16 = mul(x = mh_k_5_cast_fp16, y = cos_1_cast_fp16)[name = string("op_633_cast_fp16")]; + tensor var_638_begin_0 = const()[name = string("op_638_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_638_end_0 = const()[name = string("op_638_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_638_end_mask_0 = const()[name = string("op_638_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_638_cast_fp16 = slice_by_index(begin = var_638_begin_0, end = var_638_end_0, end_mask = var_638_end_mask_0, x = mh_k_5_cast_fp16)[name = string("op_638_cast_fp16")]; + tensor var_644_begin_0 = const()[name = string("op_644_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_644_end_0 = const()[name = string("op_644_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_644_end_mask_0 = const()[name = string("op_644_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_644_cast_fp16 = slice_by_index(begin = var_644_begin_0, end = var_644_end_0, end_mask = var_644_end_mask_0, x = mh_k_5_cast_fp16)[name = string("op_644_cast_fp16")]; + fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_646_cast_fp16 = mul(x = var_644_cast_fp16, y = const_43_promoted_to_fp16)[name = string("op_646_cast_fp16")]; + bool var_648_interleave_0 = const()[name = string("op_648_interleave_0"), val = bool(false)]; + tensor var_648_cast_fp16 = concat(axis = var_508, interleave = var_648_interleave_0, values = (var_646_cast_fp16, var_638_cast_fp16))[name = string("op_648_cast_fp16")]; + tensor var_649_cast_fp16 = mul(x = var_648_cast_fp16, y = sin_1_cast_fp16)[name = string("op_649_cast_fp16")]; + tensor mh_k_7_cast_fp16 = add(x = var_633_cast_fp16, y = var_649_cast_fp16)[name = string("mh_k_7_cast_fp16")]; + tensor var_653 = const()[name = string("op_653"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_7_cast_fp16 = reshape(shape = var_653, x = mh_k_7_cast_fp16)[name = string("current_key_7_cast_fp16")]; + tensor var_660_cast_fp16 = mul(x = var_96_cast_fp16_1, y = var_272_cast_fp16)[name = string("op_660_cast_fp16")]; + tensor var_661_cast_fp16 = mul(x = current_key_7_cast_fp16, y = var_270_cast_fp16)[name = string("op_661_cast_fp16")]; + tensor key_9_cast_fp16 = add(x = var_660_cast_fp16, y = var_661_cast_fp16)[name = string("key_9_cast_fp16")]; + tensor var_664_cast_fp16 = mul(x = var_104_cast_fp16_1, y = var_272_cast_fp16)[name = string("op_664_cast_fp16")]; + tensor var_665_cast_fp16 = mul(x = current_value_3_cast_fp16, y = var_270_cast_fp16)[name = string("op_665_cast_fp16")]; + tensor value_5_cast_fp16 = add(x = var_664_cast_fp16, y = var_665_cast_fp16)[name = string("value_5_cast_fp16")]; + tensor var_669 = const()[name = string("op_669"), val = tensor([1, 8, 128, 16])]; + tensor key_heads_5_cast_fp16 = reshape(shape = var_669, x = key_9_cast_fp16)[name = string("key_heads_5_cast_fp16")]; + tensor var_671 = const()[name = string("op_671"), val = tensor([1, 8, 128, 16])]; + tensor value_heads_5_cast_fp16 = reshape(shape = var_671, x = value_5_cast_fp16)[name = string("value_heads_5_cast_fp16")]; + tensor var_674_begin_0 = const()[name = string("op_674_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_674_end_0 = const()[name = string("op_674_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_674_end_mask_0 = const()[name = string("op_674_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_674_cast_fp16 = slice_by_index(begin = var_674_begin_0, end = var_674_end_0, end_mask = var_674_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_674_cast_fp16")]; + tensor var_678_begin_0 = const()[name = string("op_678_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_678_end_0 = const()[name = string("op_678_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_678_end_mask_0 = const()[name = string("op_678_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_678_cast_fp16 = slice_by_index(begin = var_678_begin_0, end = var_678_end_0, end_mask = var_678_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_678_cast_fp16")]; + tensor var_690_begin_0 = const()[name = string("op_690_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_690_end_0 = const()[name = string("op_690_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_690_end_mask_0 = const()[name = string("op_690_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_690_cast_fp16 = slice_by_index(begin = var_690_begin_0, end = var_690_end_0, end_mask = var_690_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_690_cast_fp16")]; + tensor var_694_begin_0 = const()[name = string("op_694_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_694_end_0 = const()[name = string("op_694_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_694_end_mask_0 = const()[name = string("op_694_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_694_cast_fp16 = slice_by_index(begin = var_694_begin_0, end = var_694_end_0, end_mask = var_694_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_694_cast_fp16")]; + tensor var_706_begin_0 = const()[name = string("op_706_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_706_end_0 = const()[name = string("op_706_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_706_end_mask_0 = const()[name = string("op_706_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_706_cast_fp16 = slice_by_index(begin = var_706_begin_0, end = var_706_end_0, end_mask = var_706_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_706_cast_fp16")]; + tensor var_710_begin_0 = const()[name = string("op_710_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_710_end_0 = const()[name = string("op_710_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_710_end_mask_0 = const()[name = string("op_710_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_710_cast_fp16 = slice_by_index(begin = var_710_begin_0, end = var_710_end_0, end_mask = var_710_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_710_cast_fp16")]; + tensor var_722_begin_0 = const()[name = string("op_722_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_722_end_0 = const()[name = string("op_722_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_722_end_mask_0 = const()[name = string("op_722_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_722_cast_fp16 = slice_by_index(begin = var_722_begin_0, end = var_722_end_0, end_mask = var_722_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_722_cast_fp16")]; + tensor var_726_begin_0 = const()[name = string("op_726_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_726_end_0 = const()[name = string("op_726_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_726_end_mask_0 = const()[name = string("op_726_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_726_cast_fp16 = slice_by_index(begin = var_726_begin_0, end = var_726_end_0, end_mask = var_726_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_726_cast_fp16")]; + tensor var_738_begin_0 = const()[name = string("op_738_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_738_end_0 = const()[name = string("op_738_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_738_end_mask_0 = const()[name = string("op_738_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_738_cast_fp16 = slice_by_index(begin = var_738_begin_0, end = var_738_end_0, end_mask = var_738_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_738_cast_fp16")]; + tensor var_742_begin_0 = const()[name = string("op_742_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_742_end_0 = const()[name = string("op_742_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_742_end_mask_0 = const()[name = string("op_742_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_742_cast_fp16 = slice_by_index(begin = var_742_begin_0, end = var_742_end_0, end_mask = var_742_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_742_cast_fp16")]; + tensor var_754_begin_0 = const()[name = string("op_754_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_754_end_0 = const()[name = string("op_754_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_754_end_mask_0 = const()[name = string("op_754_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_754_cast_fp16 = slice_by_index(begin = var_754_begin_0, end = var_754_end_0, end_mask = var_754_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_754_cast_fp16")]; + tensor var_758_begin_0 = const()[name = string("op_758_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_758_end_0 = const()[name = string("op_758_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_758_end_mask_0 = const()[name = string("op_758_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_758_cast_fp16 = slice_by_index(begin = var_758_begin_0, end = var_758_end_0, end_mask = var_758_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_758_cast_fp16")]; + tensor var_770_begin_0 = const()[name = string("op_770_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_770_end_0 = const()[name = string("op_770_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_770_end_mask_0 = const()[name = string("op_770_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_770_cast_fp16 = slice_by_index(begin = var_770_begin_0, end = var_770_end_0, end_mask = var_770_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_770_cast_fp16")]; + tensor var_774_begin_0 = const()[name = string("op_774_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_774_end_0 = const()[name = string("op_774_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_774_end_mask_0 = const()[name = string("op_774_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_774_cast_fp16 = slice_by_index(begin = var_774_begin_0, end = var_774_end_0, end_mask = var_774_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_774_cast_fp16")]; + tensor var_786_begin_0 = const()[name = string("op_786_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_786_end_0 = const()[name = string("op_786_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_786_end_mask_0 = const()[name = string("op_786_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_786_cast_fp16 = slice_by_index(begin = var_786_begin_0, end = var_786_end_0, end_mask = var_786_end_mask_0, x = key_heads_5_cast_fp16)[name = string("op_786_cast_fp16")]; + tensor var_790_begin_0 = const()[name = string("op_790_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_790_end_0 = const()[name = string("op_790_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_790_end_mask_0 = const()[name = string("op_790_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_790_cast_fp16 = slice_by_index(begin = var_790_begin_0, end = var_790_end_0, end_mask = var_790_end_mask_0, x = value_heads_5_cast_fp16)[name = string("op_790_cast_fp16")]; + bool key_heads_7_interleave_0 = const()[name = string("key_heads_7_interleave_0"), val = bool(false)]; + tensor key_heads_7_cast_fp16 = concat(axis = var_516, interleave = key_heads_7_interleave_0, values = (var_674_cast_fp16, var_674_cast_fp16, var_690_cast_fp16, var_690_cast_fp16, var_706_cast_fp16, var_706_cast_fp16, var_722_cast_fp16, var_722_cast_fp16, var_738_cast_fp16, var_738_cast_fp16, var_754_cast_fp16, var_754_cast_fp16, var_770_cast_fp16, var_770_cast_fp16, var_786_cast_fp16, var_786_cast_fp16))[name = string("key_heads_7_cast_fp16")]; + bool value_heads_7_interleave_0 = const()[name = string("value_heads_7_interleave_0"), val = bool(false)]; + tensor value_heads_7_cast_fp16 = concat(axis = var_516, interleave = value_heads_7_interleave_0, values = (var_678_cast_fp16, var_678_cast_fp16, var_694_cast_fp16, var_694_cast_fp16, var_710_cast_fp16, var_710_cast_fp16, var_726_cast_fp16, var_726_cast_fp16, var_742_cast_fp16, var_742_cast_fp16, var_758_cast_fp16, var_758_cast_fp16, var_774_cast_fp16, var_774_cast_fp16, var_790_cast_fp16, var_790_cast_fp16))[name = string("value_heads_7_cast_fp16")]; + fp16 var_813_to_fp16 = const()[name = string("op_813_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_814_cast_fp16 = mul(x = mh_q_9_cast_fp16, y = var_813_to_fp16)[name = string("op_814_cast_fp16")]; + bool mh_w_5_transpose_x_0 = const()[name = string("mh_w_5_transpose_x_0"), val = bool(true)]; + bool mh_w_5_transpose_y_0 = const()[name = string("mh_w_5_transpose_y_0"), val = bool(false)]; + tensor mh_w_5_cast_fp16 = matmul(transpose_x = mh_w_5_transpose_x_0, transpose_y = mh_w_5_transpose_y_0, x = var_814_cast_fp16, y = key_heads_7_cast_fp16)[name = string("mh_w_5_cast_fp16")]; + tensor mh_w_7_cast_fp16 = add(x = mh_w_5_cast_fp16, y = var_436_cast_fp16)[name = string("mh_w_7_cast_fp16")]; + tensor var_826_cast_fp16 = softmax(axis = var_498, x = mh_w_7_cast_fp16)[name = string("op_826_cast_fp16")]; + bool attn_3_transpose_x_0 = const()[name = string("attn_3_transpose_x_0"), val = bool(false)]; + bool attn_3_transpose_y_0 = const()[name = string("attn_3_transpose_y_0"), val = bool(true)]; + tensor attn_3_cast_fp16 = matmul(transpose_x = attn_3_transpose_x_0, transpose_y = attn_3_transpose_y_0, x = value_heads_7_cast_fp16, y = var_826_cast_fp16)[name = string("attn_3_cast_fp16")]; + tensor var_831 = const()[name = string("op_831"), val = tensor([1, -1, 1, 1])]; + tensor input_9_cast_fp16 = reshape(shape = var_831, x = attn_3_cast_fp16)[name = string("input_9_cast_fp16")]; + string obj_19_pad_type_0 = const()[name = string("obj_19_pad_type_0"), val = string("valid")]; + tensor obj_19_strides_0 = const()[name = string("obj_19_strides_0"), val = tensor([1, 1])]; + tensor obj_19_pad_0 = const()[name = string("obj_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_19_dilations_0 = const()[name = string("obj_19_dilations_0"), val = tensor([1, 1])]; + int32 obj_19_groups_0 = const()[name = string("obj_19_groups_0"), val = int32(1)]; + tensor layers_1_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22051520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24148736))))[name = string("layers_1_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_19_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_19_dilations_0, groups = obj_19_groups_0, pad = obj_19_pad_0, pad_type = obj_19_pad_type_0, strides = obj_19_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16_palettized, x = input_9_cast_fp16)[name = string("obj_19_cast_fp16")]; + tensor inputs_15_cast_fp16 = add(x = inputs_9_cast_fp16, y = obj_19_cast_fp16)[name = string("inputs_15_cast_fp16")]; + tensor inputs_sq_15_cast_fp16 = mul(x = inputs_15_cast_fp16, y = inputs_15_cast_fp16)[name = string("inputs_sq_15_cast_fp16")]; + tensor variance_15_axes_0 = const()[name = string("variance_15_axes_0"), val = tensor([1])]; + bool variance_15_keep_dims_0 = const()[name = string("variance_15_keep_dims_0"), val = bool(true)]; + tensor variance_15_cast_fp16 = reduce_mean(axes = variance_15_axes_0, keep_dims = variance_15_keep_dims_0, x = inputs_sq_15_cast_fp16)[name = string("variance_15_cast_fp16")]; + fp16 var_849_to_fp16 = const()[name = string("op_849_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_850_cast_fp16 = add(x = variance_15_cast_fp16, y = var_849_to_fp16)[name = string("op_850_cast_fp16")]; + fp32 var_851_epsilon_0 = const()[name = string("op_851_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_851_cast_fp16 = rsqrt(epsilon = var_851_epsilon_0, x = var_850_cast_fp16)[name = string("op_851_cast_fp16")]; + tensor hidden_states_17_cast_fp16 = mul(x = inputs_15_cast_fp16, y = var_851_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; + tensor w_15_to_fp16 = const()[name = string("w_15_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24149312)))]; + tensor input_11_cast_fp16 = mul(x = w_15_to_fp16, y = hidden_states_17_cast_fp16)[name = string("input_11_cast_fp16")]; + string input_13_pad_type_0 = const()[name = string("input_13_pad_type_0"), val = string("valid")]; + tensor input_13_strides_0 = const()[name = string("input_13_strides_0"), val = tensor([1, 1])]; + tensor input_13_pad_0 = const()[name = string("input_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_13_dilations_0 = const()[name = string("input_13_dilations_0"), val = tensor([1, 1])]; + int32 input_13_groups_0 = const()[name = string("input_13_groups_0"), val = int32(1)]; + tensor layers_1_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24151424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(27297216))))[name = string("layers_1_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_13_cast_fp16 = conv(dilations = input_13_dilations_0, groups = input_13_groups_0, pad = input_13_pad_0, pad_type = input_13_pad_type_0, strides = input_13_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16_palettized, x = input_11_cast_fp16)[name = string("input_13_cast_fp16")]; + tensor var_865_cast_fp16 = silu(x = input_13_cast_fp16)[name = string("op_865_cast_fp16")]; + string var_871_pad_type_0 = const()[name = string("op_871_pad_type_0"), val = string("valid")]; + tensor var_871_strides_0 = const()[name = string("op_871_strides_0"), val = tensor([1, 1])]; + tensor var_871_pad_0 = const()[name = string("op_871_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_871_dilations_0 = const()[name = string("op_871_dilations_0"), val = tensor([1, 1])]; + int32 var_871_groups_0 = const()[name = string("op_871_groups_0"), val = int32(1)]; + tensor layers_1_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(27297792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30443584))))[name = string("layers_1_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_871_cast_fp16 = conv(dilations = var_871_dilations_0, groups = var_871_groups_0, pad = var_871_pad_0, pad_type = var_871_pad_type_0, strides = var_871_strides_0, weight = layers_1_mlp_up_proj_weight_to_fp16_palettized, x = input_11_cast_fp16)[name = string("op_871_cast_fp16")]; + tensor input_15_cast_fp16 = mul(x = var_865_cast_fp16, y = var_871_cast_fp16)[name = string("input_15_cast_fp16")]; + string hidden_states_19_pad_type_0 = const()[name = string("hidden_states_19_pad_type_0"), val = string("valid")]; + tensor hidden_states_19_strides_0 = const()[name = string("hidden_states_19_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_19_pad_0 = const()[name = string("hidden_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_19_dilations_0 = const()[name = string("hidden_states_19_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_19_groups_0 = const()[name = string("hidden_states_19_groups_0"), val = int32(1)]; + tensor layers_1_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30444160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33589952))))[name = string("layers_1_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_19_cast_fp16 = conv(dilations = hidden_states_19_dilations_0, groups = hidden_states_19_groups_0, pad = hidden_states_19_pad_0, pad_type = hidden_states_19_pad_type_0, strides = hidden_states_19_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16_palettized, x = input_15_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; + tensor inputs_17_cast_fp16 = add(x = inputs_15_cast_fp16, y = hidden_states_19_cast_fp16)[name = string("inputs_17_cast_fp16")]; + int32 var_885 = const()[name = string("op_885"), val = int32(3)]; + int32 var_895 = const()[name = string("op_895"), val = int32(-2)]; + int32 var_903 = const()[name = string("op_903"), val = int32(1)]; + tensor inputs_sq_17_cast_fp16 = mul(x = inputs_17_cast_fp16, y = inputs_17_cast_fp16)[name = string("inputs_sq_17_cast_fp16")]; + tensor variance_17_axes_0 = const()[name = string("variance_17_axes_0"), val = tensor([1])]; + bool variance_17_keep_dims_0 = const()[name = string("variance_17_keep_dims_0"), val = bool(true)]; + tensor variance_17_cast_fp16 = reduce_mean(axes = variance_17_axes_0, keep_dims = variance_17_keep_dims_0, x = inputs_sq_17_cast_fp16)[name = string("variance_17_cast_fp16")]; + fp16 var_915_to_fp16 = const()[name = string("op_915_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_916_cast_fp16 = add(x = variance_17_cast_fp16, y = var_915_to_fp16)[name = string("op_916_cast_fp16")]; + fp32 var_917_epsilon_0 = const()[name = string("op_917_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_917_cast_fp16 = rsqrt(epsilon = var_917_epsilon_0, x = var_916_cast_fp16)[name = string("op_917_cast_fp16")]; + tensor hidden_states_21_cast_fp16 = mul(x = inputs_17_cast_fp16, y = var_917_cast_fp16)[name = string("hidden_states_21_cast_fp16")]; + tensor w_17_to_fp16 = const()[name = string("w_17_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33590528)))]; + tensor obj_21_cast_fp16 = mul(x = w_17_to_fp16, y = hidden_states_21_cast_fp16)[name = string("obj_21_cast_fp16")]; + string query_13_pad_type_0 = const()[name = string("query_13_pad_type_0"), val = string("valid")]; + tensor query_13_strides_0 = const()[name = string("query_13_strides_0"), val = tensor([1, 1])]; + tensor query_13_pad_0 = const()[name = string("query_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_13_dilations_0 = const()[name = string("query_13_dilations_0"), val = tensor([1, 1])]; + int32 query_13_groups_0 = const()[name = string("query_13_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33592640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35689856))))[name = string("layers_2_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor query_13_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_13_dilations_0, groups = query_13_groups_0, pad = query_13_pad_0, pad_type = query_13_pad_type_0, strides = query_13_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16_palettized, x = obj_21_cast_fp16)[name = string("query_13_cast_fp16")]; + string current_key_9_pad_type_0 = const()[name = string("current_key_9_pad_type_0"), val = string("valid")]; + tensor current_key_9_strides_0 = const()[name = string("current_key_9_strides_0"), val = tensor([1, 1])]; + tensor current_key_9_pad_0 = const()[name = string("current_key_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_9_dilations_0 = const()[name = string("current_key_9_dilations_0"), val = tensor([1, 1])]; + int32 current_key_9_groups_0 = const()[name = string("current_key_9_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35690432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36739072))))[name = string("layers_2_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_9_cast_fp16 = conv(dilations = current_key_9_dilations_0, groups = current_key_9_groups_0, pad = current_key_9_pad_0, pad_type = current_key_9_pad_type_0, strides = current_key_9_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16_palettized, x = obj_21_cast_fp16)[name = string("current_key_9_cast_fp16")]; + string current_value_5_pad_type_0 = const()[name = string("current_value_5_pad_type_0"), val = string("valid")]; + tensor current_value_5_strides_0 = const()[name = string("current_value_5_strides_0"), val = tensor([1, 1])]; + tensor current_value_5_pad_0 = const()[name = string("current_value_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_5_dilations_0 = const()[name = string("current_value_5_dilations_0"), val = tensor([1, 1])]; + int32 current_value_5_groups_0 = const()[name = string("current_value_5_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36739648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37788288))))[name = string("layers_2_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor current_value_5_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_5_dilations_0, groups = current_value_5_groups_0, pad = current_value_5_pad_0, pad_type = current_value_5_pad_type_0, strides = current_value_5_strides_0, weight = layers_2_self_attn_v_proj_weight_to_fp16_palettized, x = obj_21_cast_fp16)[name = string("current_value_5_cast_fp16")]; + tensor var_954 = const()[name = string("op_954"), val = tensor([16, 128, 1, 1])]; + tensor inputs_19_cast_fp16 = reshape(shape = var_954, x = query_13_cast_fp16)[name = string("inputs_19_cast_fp16")]; + tensor inputs_sq_19_cast_fp16 = mul(x = inputs_19_cast_fp16, y = inputs_19_cast_fp16)[name = string("inputs_sq_19_cast_fp16")]; + tensor variance_19_axes_0 = const()[name = string("variance_19_axes_0"), val = tensor([1])]; + bool variance_19_keep_dims_0 = const()[name = string("variance_19_keep_dims_0"), val = bool(true)]; + tensor variance_19_cast_fp16 = reduce_mean(axes = variance_19_axes_0, keep_dims = variance_19_keep_dims_0, x = inputs_sq_19_cast_fp16)[name = string("variance_19_cast_fp16")]; + fp16 var_960_to_fp16 = const()[name = string("op_960_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_961_cast_fp16 = add(x = variance_19_cast_fp16, y = var_960_to_fp16)[name = string("op_961_cast_fp16")]; + fp32 var_962_epsilon_0 = const()[name = string("op_962_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_962_cast_fp16 = rsqrt(epsilon = var_962_epsilon_0, x = var_961_cast_fp16)[name = string("op_962_cast_fp16")]; + tensor hidden_states_23_cast_fp16 = mul(x = inputs_19_cast_fp16, y = var_962_cast_fp16)[name = string("hidden_states_23_cast_fp16")]; + tensor w_19_to_fp16 = const()[name = string("w_19_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37788864)))]; + tensor query_normed_5_cast_fp16 = mul(x = w_19_to_fp16, y = hidden_states_23_cast_fp16)[name = string("query_normed_5_cast_fp16")]; + tensor var_970 = const()[name = string("op_970"), val = tensor([8, 128, 1, 1])]; + tensor inputs_21_cast_fp16 = reshape(shape = var_970, x = current_key_9_cast_fp16)[name = string("inputs_21_cast_fp16")]; + tensor inputs_sq_21_cast_fp16 = mul(x = inputs_21_cast_fp16, y = inputs_21_cast_fp16)[name = string("inputs_sq_21_cast_fp16")]; + tensor variance_21_axes_0 = const()[name = string("variance_21_axes_0"), val = tensor([1])]; + bool variance_21_keep_dims_0 = const()[name = string("variance_21_keep_dims_0"), val = bool(true)]; + tensor variance_21_cast_fp16 = reduce_mean(axes = variance_21_axes_0, keep_dims = variance_21_keep_dims_0, x = inputs_sq_21_cast_fp16)[name = string("variance_21_cast_fp16")]; + fp16 var_976_to_fp16 = const()[name = string("op_976_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_977_cast_fp16 = add(x = variance_21_cast_fp16, y = var_976_to_fp16)[name = string("op_977_cast_fp16")]; + fp32 var_978_epsilon_0 = const()[name = string("op_978_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_978_cast_fp16 = rsqrt(epsilon = var_978_epsilon_0, x = var_977_cast_fp16)[name = string("op_978_cast_fp16")]; + tensor hidden_states_25_cast_fp16 = mul(x = inputs_21_cast_fp16, y = var_978_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; + tensor w_21_to_fp16 = const()[name = string("w_21_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37789184)))]; + tensor current_key_normed_5_cast_fp16 = mul(x = w_21_to_fp16, y = hidden_states_25_cast_fp16)[name = string("current_key_normed_5_cast_fp16")]; + tensor var_996 = const()[name = string("op_996"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_13_cast_fp16 = reshape(shape = var_996, x = query_normed_5_cast_fp16)[name = string("mh_q_13_cast_fp16")]; + tensor var_998 = const()[name = string("op_998"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_9_cast_fp16 = reshape(shape = var_998, x = current_key_normed_5_cast_fp16)[name = string("mh_k_9_cast_fp16")]; + tensor var_1002_cast_fp16 = mul(x = mh_q_13_cast_fp16, y = cos_1_cast_fp16)[name = string("op_1002_cast_fp16")]; + tensor var_1007_begin_0 = const()[name = string("op_1007_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1007_end_0 = const()[name = string("op_1007_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_1007_end_mask_0 = const()[name = string("op_1007_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1007_cast_fp16 = slice_by_index(begin = var_1007_begin_0, end = var_1007_end_0, end_mask = var_1007_end_mask_0, x = mh_q_13_cast_fp16)[name = string("op_1007_cast_fp16")]; + tensor var_1013_begin_0 = const()[name = string("op_1013_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1013_end_0 = const()[name = string("op_1013_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_1013_end_mask_0 = const()[name = string("op_1013_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1013_cast_fp16 = slice_by_index(begin = var_1013_begin_0, end = var_1013_end_0, end_mask = var_1013_end_mask_0, x = mh_q_13_cast_fp16)[name = string("op_1013_cast_fp16")]; + fp16 const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1015_cast_fp16 = mul(x = var_1013_cast_fp16, y = const_63_promoted_to_fp16)[name = string("op_1015_cast_fp16")]; + bool var_1017_interleave_0 = const()[name = string("op_1017_interleave_0"), val = bool(false)]; + tensor var_1017_cast_fp16 = concat(axis = var_895, interleave = var_1017_interleave_0, values = (var_1015_cast_fp16, var_1007_cast_fp16))[name = string("op_1017_cast_fp16")]; + tensor var_1018_cast_fp16 = mul(x = var_1017_cast_fp16, y = sin_1_cast_fp16)[name = string("op_1018_cast_fp16")]; + tensor mh_q_15_cast_fp16 = add(x = var_1002_cast_fp16, y = var_1018_cast_fp16)[name = string("mh_q_15_cast_fp16")]; + tensor var_1020_cast_fp16 = mul(x = mh_k_9_cast_fp16, y = cos_1_cast_fp16)[name = string("op_1020_cast_fp16")]; + tensor var_1025_begin_0 = const()[name = string("op_1025_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1025_end_0 = const()[name = string("op_1025_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_1025_end_mask_0 = const()[name = string("op_1025_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1025_cast_fp16 = slice_by_index(begin = var_1025_begin_0, end = var_1025_end_0, end_mask = var_1025_end_mask_0, x = mh_k_9_cast_fp16)[name = string("op_1025_cast_fp16")]; + tensor var_1031_begin_0 = const()[name = string("op_1031_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1031_end_0 = const()[name = string("op_1031_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_1031_end_mask_0 = const()[name = string("op_1031_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1031_cast_fp16 = slice_by_index(begin = var_1031_begin_0, end = var_1031_end_0, end_mask = var_1031_end_mask_0, x = mh_k_9_cast_fp16)[name = string("op_1031_cast_fp16")]; + fp16 const_66_promoted_to_fp16 = const()[name = string("const_66_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1033_cast_fp16 = mul(x = var_1031_cast_fp16, y = const_66_promoted_to_fp16)[name = string("op_1033_cast_fp16")]; + bool var_1035_interleave_0 = const()[name = string("op_1035_interleave_0"), val = bool(false)]; + tensor var_1035_cast_fp16 = concat(axis = var_895, interleave = var_1035_interleave_0, values = (var_1033_cast_fp16, var_1025_cast_fp16))[name = string("op_1035_cast_fp16")]; + tensor var_1036_cast_fp16 = mul(x = var_1035_cast_fp16, y = sin_1_cast_fp16)[name = string("op_1036_cast_fp16")]; + tensor mh_k_11_cast_fp16 = add(x = var_1020_cast_fp16, y = var_1036_cast_fp16)[name = string("mh_k_11_cast_fp16")]; + tensor var_1040 = const()[name = string("op_1040"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_11_cast_fp16 = reshape(shape = var_1040, x = mh_k_11_cast_fp16)[name = string("current_key_11_cast_fp16")]; + tensor var_1047_cast_fp16 = mul(x = var_96_cast_fp16_2, y = var_272_cast_fp16)[name = string("op_1047_cast_fp16")]; + tensor var_1048_cast_fp16 = mul(x = current_key_11_cast_fp16, y = var_270_cast_fp16)[name = string("op_1048_cast_fp16")]; + tensor key_15_cast_fp16 = add(x = var_1047_cast_fp16, y = var_1048_cast_fp16)[name = string("key_15_cast_fp16")]; + tensor var_1051_cast_fp16 = mul(x = var_104_cast_fp16_2, y = var_272_cast_fp16)[name = string("op_1051_cast_fp16")]; + tensor var_1052_cast_fp16 = mul(x = current_value_5_cast_fp16, y = var_270_cast_fp16)[name = string("op_1052_cast_fp16")]; + tensor value_9_cast_fp16 = add(x = var_1051_cast_fp16, y = var_1052_cast_fp16)[name = string("value_9_cast_fp16")]; + tensor var_1056 = const()[name = string("op_1056"), val = tensor([1, 8, 128, 16])]; + tensor key_heads_9_cast_fp16 = reshape(shape = var_1056, x = key_15_cast_fp16)[name = string("key_heads_9_cast_fp16")]; + tensor var_1058 = const()[name = string("op_1058"), val = tensor([1, 8, 128, 16])]; + tensor value_heads_9_cast_fp16 = reshape(shape = var_1058, x = value_9_cast_fp16)[name = string("value_heads_9_cast_fp16")]; + tensor var_1061_begin_0 = const()[name = string("op_1061_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1061_end_0 = const()[name = string("op_1061_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1061_end_mask_0 = const()[name = string("op_1061_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1061_cast_fp16 = slice_by_index(begin = var_1061_begin_0, end = var_1061_end_0, end_mask = var_1061_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1061_cast_fp16")]; + tensor var_1065_begin_0 = const()[name = string("op_1065_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1065_end_0 = const()[name = string("op_1065_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1065_end_mask_0 = const()[name = string("op_1065_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1065_cast_fp16 = slice_by_index(begin = var_1065_begin_0, end = var_1065_end_0, end_mask = var_1065_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1065_cast_fp16")]; + tensor var_1077_begin_0 = const()[name = string("op_1077_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_1077_end_0 = const()[name = string("op_1077_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_1077_end_mask_0 = const()[name = string("op_1077_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1077_cast_fp16 = slice_by_index(begin = var_1077_begin_0, end = var_1077_end_0, end_mask = var_1077_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1077_cast_fp16")]; + tensor var_1081_begin_0 = const()[name = string("op_1081_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_1081_end_0 = const()[name = string("op_1081_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_1081_end_mask_0 = const()[name = string("op_1081_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1081_cast_fp16 = slice_by_index(begin = var_1081_begin_0, end = var_1081_end_0, end_mask = var_1081_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1081_cast_fp16")]; + tensor var_1093_begin_0 = const()[name = string("op_1093_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_1093_end_0 = const()[name = string("op_1093_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_1093_end_mask_0 = const()[name = string("op_1093_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1093_cast_fp16 = slice_by_index(begin = var_1093_begin_0, end = var_1093_end_0, end_mask = var_1093_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1093_cast_fp16")]; + tensor var_1097_begin_0 = const()[name = string("op_1097_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_1097_end_0 = const()[name = string("op_1097_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_1097_end_mask_0 = const()[name = string("op_1097_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1097_cast_fp16 = slice_by_index(begin = var_1097_begin_0, end = var_1097_end_0, end_mask = var_1097_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1097_cast_fp16")]; + tensor var_1109_begin_0 = const()[name = string("op_1109_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_1109_end_0 = const()[name = string("op_1109_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_1109_end_mask_0 = const()[name = string("op_1109_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1109_cast_fp16 = slice_by_index(begin = var_1109_begin_0, end = var_1109_end_0, end_mask = var_1109_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1109_cast_fp16")]; + tensor var_1113_begin_0 = const()[name = string("op_1113_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_1113_end_0 = const()[name = string("op_1113_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_1113_end_mask_0 = const()[name = string("op_1113_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1113_cast_fp16 = slice_by_index(begin = var_1113_begin_0, end = var_1113_end_0, end_mask = var_1113_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1113_cast_fp16")]; + tensor var_1125_begin_0 = const()[name = string("op_1125_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_1125_end_0 = const()[name = string("op_1125_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_1125_end_mask_0 = const()[name = string("op_1125_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1125_cast_fp16 = slice_by_index(begin = var_1125_begin_0, end = var_1125_end_0, end_mask = var_1125_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1125_cast_fp16")]; + tensor var_1129_begin_0 = const()[name = string("op_1129_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_1129_end_0 = const()[name = string("op_1129_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_1129_end_mask_0 = const()[name = string("op_1129_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1129_cast_fp16 = slice_by_index(begin = var_1129_begin_0, end = var_1129_end_0, end_mask = var_1129_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1129_cast_fp16")]; + tensor var_1141_begin_0 = const()[name = string("op_1141_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_1141_end_0 = const()[name = string("op_1141_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_1141_end_mask_0 = const()[name = string("op_1141_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1141_cast_fp16 = slice_by_index(begin = var_1141_begin_0, end = var_1141_end_0, end_mask = var_1141_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1141_cast_fp16")]; + tensor var_1145_begin_0 = const()[name = string("op_1145_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_1145_end_0 = const()[name = string("op_1145_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_1145_end_mask_0 = const()[name = string("op_1145_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1145_cast_fp16 = slice_by_index(begin = var_1145_begin_0, end = var_1145_end_0, end_mask = var_1145_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1145_cast_fp16")]; + tensor var_1157_begin_0 = const()[name = string("op_1157_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_1157_end_0 = const()[name = string("op_1157_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_1157_end_mask_0 = const()[name = string("op_1157_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1157_cast_fp16 = slice_by_index(begin = var_1157_begin_0, end = var_1157_end_0, end_mask = var_1157_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1157_cast_fp16")]; + tensor var_1161_begin_0 = const()[name = string("op_1161_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_1161_end_0 = const()[name = string("op_1161_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_1161_end_mask_0 = const()[name = string("op_1161_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1161_cast_fp16 = slice_by_index(begin = var_1161_begin_0, end = var_1161_end_0, end_mask = var_1161_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1161_cast_fp16")]; + tensor var_1173_begin_0 = const()[name = string("op_1173_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_1173_end_0 = const()[name = string("op_1173_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1173_end_mask_0 = const()[name = string("op_1173_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1173_cast_fp16 = slice_by_index(begin = var_1173_begin_0, end = var_1173_end_0, end_mask = var_1173_end_mask_0, x = key_heads_9_cast_fp16)[name = string("op_1173_cast_fp16")]; + tensor var_1177_begin_0 = const()[name = string("op_1177_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_1177_end_0 = const()[name = string("op_1177_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1177_end_mask_0 = const()[name = string("op_1177_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1177_cast_fp16 = slice_by_index(begin = var_1177_begin_0, end = var_1177_end_0, end_mask = var_1177_end_mask_0, x = value_heads_9_cast_fp16)[name = string("op_1177_cast_fp16")]; + bool key_heads_11_interleave_0 = const()[name = string("key_heads_11_interleave_0"), val = bool(false)]; + tensor key_heads_11_cast_fp16 = concat(axis = var_903, interleave = key_heads_11_interleave_0, values = (var_1061_cast_fp16, var_1061_cast_fp16, var_1077_cast_fp16, var_1077_cast_fp16, var_1093_cast_fp16, var_1093_cast_fp16, var_1109_cast_fp16, var_1109_cast_fp16, var_1125_cast_fp16, var_1125_cast_fp16, var_1141_cast_fp16, var_1141_cast_fp16, var_1157_cast_fp16, var_1157_cast_fp16, var_1173_cast_fp16, var_1173_cast_fp16))[name = string("key_heads_11_cast_fp16")]; + bool value_heads_11_interleave_0 = const()[name = string("value_heads_11_interleave_0"), val = bool(false)]; + tensor value_heads_11_cast_fp16 = concat(axis = var_903, interleave = value_heads_11_interleave_0, values = (var_1065_cast_fp16, var_1065_cast_fp16, var_1081_cast_fp16, var_1081_cast_fp16, var_1097_cast_fp16, var_1097_cast_fp16, var_1113_cast_fp16, var_1113_cast_fp16, var_1129_cast_fp16, var_1129_cast_fp16, var_1145_cast_fp16, var_1145_cast_fp16, var_1161_cast_fp16, var_1161_cast_fp16, var_1177_cast_fp16, var_1177_cast_fp16))[name = string("value_heads_11_cast_fp16")]; + fp16 var_1200_to_fp16 = const()[name = string("op_1200_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_1201_cast_fp16 = mul(x = mh_q_15_cast_fp16, y = var_1200_to_fp16)[name = string("op_1201_cast_fp16")]; + bool mh_w_9_transpose_x_0 = const()[name = string("mh_w_9_transpose_x_0"), val = bool(true)]; + bool mh_w_9_transpose_y_0 = const()[name = string("mh_w_9_transpose_y_0"), val = bool(false)]; + tensor mh_w_9_cast_fp16 = matmul(transpose_x = mh_w_9_transpose_x_0, transpose_y = mh_w_9_transpose_y_0, x = var_1201_cast_fp16, y = key_heads_11_cast_fp16)[name = string("mh_w_9_cast_fp16")]; + tensor mh_w_11_cast_fp16 = add(x = mh_w_9_cast_fp16, y = var_436_cast_fp16)[name = string("mh_w_11_cast_fp16")]; + tensor var_1213_cast_fp16 = softmax(axis = var_885, x = mh_w_11_cast_fp16)[name = string("op_1213_cast_fp16")]; + bool attn_5_transpose_x_0 = const()[name = string("attn_5_transpose_x_0"), val = bool(false)]; + bool attn_5_transpose_y_0 = const()[name = string("attn_5_transpose_y_0"), val = bool(true)]; + tensor attn_5_cast_fp16 = matmul(transpose_x = attn_5_transpose_x_0, transpose_y = attn_5_transpose_y_0, x = value_heads_11_cast_fp16, y = var_1213_cast_fp16)[name = string("attn_5_cast_fp16")]; + tensor var_1218 = const()[name = string("op_1218"), val = tensor([1, -1, 1, 1])]; + tensor input_17_cast_fp16 = reshape(shape = var_1218, x = attn_5_cast_fp16)[name = string("input_17_cast_fp16")]; + string obj_27_pad_type_0 = const()[name = string("obj_27_pad_type_0"), val = string("valid")]; + tensor obj_27_strides_0 = const()[name = string("obj_27_strides_0"), val = tensor([1, 1])]; + tensor obj_27_pad_0 = const()[name = string("obj_27_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_27_dilations_0 = const()[name = string("obj_27_dilations_0"), val = tensor([1, 1])]; + int32 obj_27_groups_0 = const()[name = string("obj_27_groups_0"), val = int32(1)]; + tensor layers_2_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(37789504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39886720))))[name = string("layers_2_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_27_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_27_dilations_0, groups = obj_27_groups_0, pad = obj_27_pad_0, pad_type = obj_27_pad_type_0, strides = obj_27_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16_palettized, x = input_17_cast_fp16)[name = string("obj_27_cast_fp16")]; + tensor inputs_23_cast_fp16 = add(x = inputs_17_cast_fp16, y = obj_27_cast_fp16)[name = string("inputs_23_cast_fp16")]; + tensor inputs_sq_23_cast_fp16 = mul(x = inputs_23_cast_fp16, y = inputs_23_cast_fp16)[name = string("inputs_sq_23_cast_fp16")]; + tensor variance_23_axes_0 = const()[name = string("variance_23_axes_0"), val = tensor([1])]; + bool variance_23_keep_dims_0 = const()[name = string("variance_23_keep_dims_0"), val = bool(true)]; + tensor variance_23_cast_fp16 = reduce_mean(axes = variance_23_axes_0, keep_dims = variance_23_keep_dims_0, x = inputs_sq_23_cast_fp16)[name = string("variance_23_cast_fp16")]; + fp16 var_1236_to_fp16 = const()[name = string("op_1236_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1237_cast_fp16 = add(x = variance_23_cast_fp16, y = var_1236_to_fp16)[name = string("op_1237_cast_fp16")]; + fp32 var_1238_epsilon_0 = const()[name = string("op_1238_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1238_cast_fp16 = rsqrt(epsilon = var_1238_epsilon_0, x = var_1237_cast_fp16)[name = string("op_1238_cast_fp16")]; + tensor hidden_states_27_cast_fp16 = mul(x = inputs_23_cast_fp16, y = var_1238_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; + tensor w_23_to_fp16 = const()[name = string("w_23_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39887296)))]; + tensor input_19_cast_fp16 = mul(x = w_23_to_fp16, y = hidden_states_27_cast_fp16)[name = string("input_19_cast_fp16")]; + string input_21_pad_type_0 = const()[name = string("input_21_pad_type_0"), val = string("valid")]; + tensor input_21_strides_0 = const()[name = string("input_21_strides_0"), val = tensor([1, 1])]; + tensor input_21_pad_0 = const()[name = string("input_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_21_dilations_0 = const()[name = string("input_21_dilations_0"), val = tensor([1, 1])]; + int32 input_21_groups_0 = const()[name = string("input_21_groups_0"), val = int32(1)]; + tensor layers_2_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39889408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43035200))))[name = string("layers_2_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_21_cast_fp16 = conv(dilations = input_21_dilations_0, groups = input_21_groups_0, pad = input_21_pad_0, pad_type = input_21_pad_type_0, strides = input_21_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16_palettized, x = input_19_cast_fp16)[name = string("input_21_cast_fp16")]; + tensor var_1252_cast_fp16 = silu(x = input_21_cast_fp16)[name = string("op_1252_cast_fp16")]; + string var_1258_pad_type_0 = const()[name = string("op_1258_pad_type_0"), val = string("valid")]; + tensor var_1258_strides_0 = const()[name = string("op_1258_strides_0"), val = tensor([1, 1])]; + tensor var_1258_pad_0 = const()[name = string("op_1258_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1258_dilations_0 = const()[name = string("op_1258_dilations_0"), val = tensor([1, 1])]; + int32 var_1258_groups_0 = const()[name = string("op_1258_groups_0"), val = int32(1)]; + tensor layers_2_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43035776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46181568))))[name = string("layers_2_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_1258_cast_fp16 = conv(dilations = var_1258_dilations_0, groups = var_1258_groups_0, pad = var_1258_pad_0, pad_type = var_1258_pad_type_0, strides = var_1258_strides_0, weight = layers_2_mlp_up_proj_weight_to_fp16_palettized, x = input_19_cast_fp16)[name = string("op_1258_cast_fp16")]; + tensor input_23_cast_fp16 = mul(x = var_1252_cast_fp16, y = var_1258_cast_fp16)[name = string("input_23_cast_fp16")]; + string hidden_states_29_pad_type_0 = const()[name = string("hidden_states_29_pad_type_0"), val = string("valid")]; + tensor hidden_states_29_strides_0 = const()[name = string("hidden_states_29_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_29_pad_0 = const()[name = string("hidden_states_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_29_dilations_0 = const()[name = string("hidden_states_29_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_29_groups_0 = const()[name = string("hidden_states_29_groups_0"), val = int32(1)]; + tensor layers_2_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46182144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49327936))))[name = string("layers_2_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_29_cast_fp16 = conv(dilations = hidden_states_29_dilations_0, groups = hidden_states_29_groups_0, pad = hidden_states_29_pad_0, pad_type = hidden_states_29_pad_type_0, strides = hidden_states_29_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16_palettized, x = input_23_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; + tensor inputs_25_cast_fp16 = add(x = inputs_23_cast_fp16, y = hidden_states_29_cast_fp16)[name = string("inputs_25_cast_fp16")]; + int32 var_1272 = const()[name = string("op_1272"), val = int32(3)]; + int32 var_1282 = const()[name = string("op_1282"), val = int32(-2)]; + int32 var_1290 = const()[name = string("op_1290"), val = int32(1)]; + tensor inputs_sq_25_cast_fp16 = mul(x = inputs_25_cast_fp16, y = inputs_25_cast_fp16)[name = string("inputs_sq_25_cast_fp16")]; + tensor variance_25_axes_0 = const()[name = string("variance_25_axes_0"), val = tensor([1])]; + bool variance_25_keep_dims_0 = const()[name = string("variance_25_keep_dims_0"), val = bool(true)]; + tensor variance_25_cast_fp16 = reduce_mean(axes = variance_25_axes_0, keep_dims = variance_25_keep_dims_0, x = inputs_sq_25_cast_fp16)[name = string("variance_25_cast_fp16")]; + fp16 var_1302_to_fp16 = const()[name = string("op_1302_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1303_cast_fp16 = add(x = variance_25_cast_fp16, y = var_1302_to_fp16)[name = string("op_1303_cast_fp16")]; + fp32 var_1304_epsilon_0 = const()[name = string("op_1304_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1304_cast_fp16 = rsqrt(epsilon = var_1304_epsilon_0, x = var_1303_cast_fp16)[name = string("op_1304_cast_fp16")]; + tensor hidden_states_31_cast_fp16 = mul(x = inputs_25_cast_fp16, y = var_1304_cast_fp16)[name = string("hidden_states_31_cast_fp16")]; + tensor w_25_to_fp16 = const()[name = string("w_25_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49328512)))]; + tensor obj_29_cast_fp16 = mul(x = w_25_to_fp16, y = hidden_states_31_cast_fp16)[name = string("obj_29_cast_fp16")]; + string query_19_pad_type_0 = const()[name = string("query_19_pad_type_0"), val = string("valid")]; + tensor query_19_strides_0 = const()[name = string("query_19_strides_0"), val = tensor([1, 1])]; + tensor query_19_pad_0 = const()[name = string("query_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_19_dilations_0 = const()[name = string("query_19_dilations_0"), val = tensor([1, 1])]; + int32 query_19_groups_0 = const()[name = string("query_19_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(49330624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51427840))))[name = string("layers_3_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor query_19_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_19_dilations_0, groups = query_19_groups_0, pad = query_19_pad_0, pad_type = query_19_pad_type_0, strides = query_19_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16_palettized, x = obj_29_cast_fp16)[name = string("query_19_cast_fp16")]; + string current_key_13_pad_type_0 = const()[name = string("current_key_13_pad_type_0"), val = string("valid")]; + tensor current_key_13_strides_0 = const()[name = string("current_key_13_strides_0"), val = tensor([1, 1])]; + tensor current_key_13_pad_0 = const()[name = string("current_key_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_13_dilations_0 = const()[name = string("current_key_13_dilations_0"), val = tensor([1, 1])]; + int32 current_key_13_groups_0 = const()[name = string("current_key_13_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51428416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52477056))))[name = string("layers_3_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_13_cast_fp16 = conv(dilations = current_key_13_dilations_0, groups = current_key_13_groups_0, pad = current_key_13_pad_0, pad_type = current_key_13_pad_type_0, strides = current_key_13_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16_palettized, x = obj_29_cast_fp16)[name = string("current_key_13_cast_fp16")]; + string current_value_7_pad_type_0 = const()[name = string("current_value_7_pad_type_0"), val = string("valid")]; + tensor current_value_7_strides_0 = const()[name = string("current_value_7_strides_0"), val = tensor([1, 1])]; + tensor current_value_7_pad_0 = const()[name = string("current_value_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_7_dilations_0 = const()[name = string("current_value_7_dilations_0"), val = tensor([1, 1])]; + int32 current_value_7_groups_0 = const()[name = string("current_value_7_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52477632))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53526272))))[name = string("layers_3_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor current_value_7_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_7_dilations_0, groups = current_value_7_groups_0, pad = current_value_7_pad_0, pad_type = current_value_7_pad_type_0, strides = current_value_7_strides_0, weight = layers_3_self_attn_v_proj_weight_to_fp16_palettized, x = obj_29_cast_fp16)[name = string("current_value_7_cast_fp16")]; + tensor var_1341 = const()[name = string("op_1341"), val = tensor([16, 128, 1, 1])]; + tensor inputs_27_cast_fp16 = reshape(shape = var_1341, x = query_19_cast_fp16)[name = string("inputs_27_cast_fp16")]; + tensor inputs_sq_27_cast_fp16 = mul(x = inputs_27_cast_fp16, y = inputs_27_cast_fp16)[name = string("inputs_sq_27_cast_fp16")]; + tensor variance_27_axes_0 = const()[name = string("variance_27_axes_0"), val = tensor([1])]; + bool variance_27_keep_dims_0 = const()[name = string("variance_27_keep_dims_0"), val = bool(true)]; + tensor variance_27_cast_fp16 = reduce_mean(axes = variance_27_axes_0, keep_dims = variance_27_keep_dims_0, x = inputs_sq_27_cast_fp16)[name = string("variance_27_cast_fp16")]; + fp16 var_1347_to_fp16 = const()[name = string("op_1347_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1348_cast_fp16 = add(x = variance_27_cast_fp16, y = var_1347_to_fp16)[name = string("op_1348_cast_fp16")]; + fp32 var_1349_epsilon_0 = const()[name = string("op_1349_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1349_cast_fp16 = rsqrt(epsilon = var_1349_epsilon_0, x = var_1348_cast_fp16)[name = string("op_1349_cast_fp16")]; + tensor hidden_states_33_cast_fp16 = mul(x = inputs_27_cast_fp16, y = var_1349_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; + tensor w_27_to_fp16 = const()[name = string("w_27_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53526848)))]; + tensor query_normed_7_cast_fp16 = mul(x = w_27_to_fp16, y = hidden_states_33_cast_fp16)[name = string("query_normed_7_cast_fp16")]; + tensor var_1357 = const()[name = string("op_1357"), val = tensor([8, 128, 1, 1])]; + tensor inputs_29_cast_fp16 = reshape(shape = var_1357, x = current_key_13_cast_fp16)[name = string("inputs_29_cast_fp16")]; + tensor inputs_sq_29_cast_fp16 = mul(x = inputs_29_cast_fp16, y = inputs_29_cast_fp16)[name = string("inputs_sq_29_cast_fp16")]; + tensor variance_29_axes_0 = const()[name = string("variance_29_axes_0"), val = tensor([1])]; + bool variance_29_keep_dims_0 = const()[name = string("variance_29_keep_dims_0"), val = bool(true)]; + tensor variance_29_cast_fp16 = reduce_mean(axes = variance_29_axes_0, keep_dims = variance_29_keep_dims_0, x = inputs_sq_29_cast_fp16)[name = string("variance_29_cast_fp16")]; + fp16 var_1363_to_fp16 = const()[name = string("op_1363_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1364_cast_fp16 = add(x = variance_29_cast_fp16, y = var_1363_to_fp16)[name = string("op_1364_cast_fp16")]; + fp32 var_1365_epsilon_0 = const()[name = string("op_1365_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1365_cast_fp16 = rsqrt(epsilon = var_1365_epsilon_0, x = var_1364_cast_fp16)[name = string("op_1365_cast_fp16")]; + tensor hidden_states_35_cast_fp16 = mul(x = inputs_29_cast_fp16, y = var_1365_cast_fp16)[name = string("hidden_states_35_cast_fp16")]; + tensor w_29_to_fp16 = const()[name = string("w_29_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53527168)))]; + tensor current_key_normed_7_cast_fp16 = mul(x = w_29_to_fp16, y = hidden_states_35_cast_fp16)[name = string("current_key_normed_7_cast_fp16")]; + tensor var_1383 = const()[name = string("op_1383"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_19_cast_fp16 = reshape(shape = var_1383, x = query_normed_7_cast_fp16)[name = string("mh_q_19_cast_fp16")]; + tensor var_1385 = const()[name = string("op_1385"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_13_cast_fp16 = reshape(shape = var_1385, x = current_key_normed_7_cast_fp16)[name = string("mh_k_13_cast_fp16")]; + tensor var_1389_cast_fp16 = mul(x = mh_q_19_cast_fp16, y = cos_1_cast_fp16)[name = string("op_1389_cast_fp16")]; + tensor var_1394_begin_0 = const()[name = string("op_1394_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1394_end_0 = const()[name = string("op_1394_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_1394_end_mask_0 = const()[name = string("op_1394_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1394_cast_fp16 = slice_by_index(begin = var_1394_begin_0, end = var_1394_end_0, end_mask = var_1394_end_mask_0, x = mh_q_19_cast_fp16)[name = string("op_1394_cast_fp16")]; + tensor var_1400_begin_0 = const()[name = string("op_1400_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1400_end_0 = const()[name = string("op_1400_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_1400_end_mask_0 = const()[name = string("op_1400_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1400_cast_fp16 = slice_by_index(begin = var_1400_begin_0, end = var_1400_end_0, end_mask = var_1400_end_mask_0, x = mh_q_19_cast_fp16)[name = string("op_1400_cast_fp16")]; + fp16 const_86_promoted_to_fp16 = const()[name = string("const_86_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1402_cast_fp16 = mul(x = var_1400_cast_fp16, y = const_86_promoted_to_fp16)[name = string("op_1402_cast_fp16")]; + bool var_1404_interleave_0 = const()[name = string("op_1404_interleave_0"), val = bool(false)]; + tensor var_1404_cast_fp16 = concat(axis = var_1282, interleave = var_1404_interleave_0, values = (var_1402_cast_fp16, var_1394_cast_fp16))[name = string("op_1404_cast_fp16")]; + tensor var_1405_cast_fp16 = mul(x = var_1404_cast_fp16, y = sin_1_cast_fp16)[name = string("op_1405_cast_fp16")]; + tensor mh_q_21_cast_fp16 = add(x = var_1389_cast_fp16, y = var_1405_cast_fp16)[name = string("mh_q_21_cast_fp16")]; + tensor var_1407_cast_fp16 = mul(x = mh_k_13_cast_fp16, y = cos_1_cast_fp16)[name = string("op_1407_cast_fp16")]; + tensor var_1412_begin_0 = const()[name = string("op_1412_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1412_end_0 = const()[name = string("op_1412_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_1412_end_mask_0 = const()[name = string("op_1412_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1412_cast_fp16 = slice_by_index(begin = var_1412_begin_0, end = var_1412_end_0, end_mask = var_1412_end_mask_0, x = mh_k_13_cast_fp16)[name = string("op_1412_cast_fp16")]; + tensor var_1418_begin_0 = const()[name = string("op_1418_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1418_end_0 = const()[name = string("op_1418_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_1418_end_mask_0 = const()[name = string("op_1418_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1418_cast_fp16 = slice_by_index(begin = var_1418_begin_0, end = var_1418_end_0, end_mask = var_1418_end_mask_0, x = mh_k_13_cast_fp16)[name = string("op_1418_cast_fp16")]; + fp16 const_89_promoted_to_fp16 = const()[name = string("const_89_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1420_cast_fp16 = mul(x = var_1418_cast_fp16, y = const_89_promoted_to_fp16)[name = string("op_1420_cast_fp16")]; + bool var_1422_interleave_0 = const()[name = string("op_1422_interleave_0"), val = bool(false)]; + tensor var_1422_cast_fp16 = concat(axis = var_1282, interleave = var_1422_interleave_0, values = (var_1420_cast_fp16, var_1412_cast_fp16))[name = string("op_1422_cast_fp16")]; + tensor var_1423_cast_fp16 = mul(x = var_1422_cast_fp16, y = sin_1_cast_fp16)[name = string("op_1423_cast_fp16")]; + tensor mh_k_15_cast_fp16 = add(x = var_1407_cast_fp16, y = var_1423_cast_fp16)[name = string("mh_k_15_cast_fp16")]; + tensor var_1427 = const()[name = string("op_1427"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_15_cast_fp16 = reshape(shape = var_1427, x = mh_k_15_cast_fp16)[name = string("current_key_15_cast_fp16")]; + tensor var_1434_cast_fp16 = mul(x = var_96_cast_fp16_3, y = var_272_cast_fp16)[name = string("op_1434_cast_fp16")]; + tensor var_1435_cast_fp16 = mul(x = current_key_15_cast_fp16, y = var_270_cast_fp16)[name = string("op_1435_cast_fp16")]; + tensor key_21_cast_fp16 = add(x = var_1434_cast_fp16, y = var_1435_cast_fp16)[name = string("key_21_cast_fp16")]; + tensor var_1438_cast_fp16 = mul(x = var_104_cast_fp16_3, y = var_272_cast_fp16)[name = string("op_1438_cast_fp16")]; + tensor var_1439_cast_fp16 = mul(x = current_value_7_cast_fp16, y = var_270_cast_fp16)[name = string("op_1439_cast_fp16")]; + tensor value_13_cast_fp16 = add(x = var_1438_cast_fp16, y = var_1439_cast_fp16)[name = string("value_13_cast_fp16")]; + tensor var_1443 = const()[name = string("op_1443"), val = tensor([1, 8, 128, 16])]; + tensor key_heads_13_cast_fp16 = reshape(shape = var_1443, x = key_21_cast_fp16)[name = string("key_heads_13_cast_fp16")]; + tensor var_1445 = const()[name = string("op_1445"), val = tensor([1, 8, 128, 16])]; + tensor value_heads_13_cast_fp16 = reshape(shape = var_1445, x = value_13_cast_fp16)[name = string("value_heads_13_cast_fp16")]; + tensor var_1448_begin_0 = const()[name = string("op_1448_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1448_end_0 = const()[name = string("op_1448_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1448_end_mask_0 = const()[name = string("op_1448_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1448_cast_fp16 = slice_by_index(begin = var_1448_begin_0, end = var_1448_end_0, end_mask = var_1448_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1448_cast_fp16")]; + tensor var_1452_begin_0 = const()[name = string("op_1452_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1452_end_0 = const()[name = string("op_1452_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1452_end_mask_0 = const()[name = string("op_1452_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1452_cast_fp16 = slice_by_index(begin = var_1452_begin_0, end = var_1452_end_0, end_mask = var_1452_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1452_cast_fp16")]; + tensor var_1464_begin_0 = const()[name = string("op_1464_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_1464_end_0 = const()[name = string("op_1464_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_1464_end_mask_0 = const()[name = string("op_1464_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1464_cast_fp16 = slice_by_index(begin = var_1464_begin_0, end = var_1464_end_0, end_mask = var_1464_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1464_cast_fp16")]; + tensor var_1468_begin_0 = const()[name = string("op_1468_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_1468_end_0 = const()[name = string("op_1468_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_1468_end_mask_0 = const()[name = string("op_1468_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1468_cast_fp16 = slice_by_index(begin = var_1468_begin_0, end = var_1468_end_0, end_mask = var_1468_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1468_cast_fp16")]; + tensor var_1480_begin_0 = const()[name = string("op_1480_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_1480_end_0 = const()[name = string("op_1480_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_1480_end_mask_0 = const()[name = string("op_1480_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1480_cast_fp16 = slice_by_index(begin = var_1480_begin_0, end = var_1480_end_0, end_mask = var_1480_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1480_cast_fp16")]; + tensor var_1484_begin_0 = const()[name = string("op_1484_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_1484_end_0 = const()[name = string("op_1484_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_1484_end_mask_0 = const()[name = string("op_1484_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1484_cast_fp16 = slice_by_index(begin = var_1484_begin_0, end = var_1484_end_0, end_mask = var_1484_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1484_cast_fp16")]; + tensor var_1496_begin_0 = const()[name = string("op_1496_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_1496_end_0 = const()[name = string("op_1496_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_1496_end_mask_0 = const()[name = string("op_1496_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1496_cast_fp16 = slice_by_index(begin = var_1496_begin_0, end = var_1496_end_0, end_mask = var_1496_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1496_cast_fp16")]; + tensor var_1500_begin_0 = const()[name = string("op_1500_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_1500_end_0 = const()[name = string("op_1500_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_1500_end_mask_0 = const()[name = string("op_1500_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1500_cast_fp16 = slice_by_index(begin = var_1500_begin_0, end = var_1500_end_0, end_mask = var_1500_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1500_cast_fp16")]; + tensor var_1512_begin_0 = const()[name = string("op_1512_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_1512_end_0 = const()[name = string("op_1512_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_1512_end_mask_0 = const()[name = string("op_1512_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1512_cast_fp16 = slice_by_index(begin = var_1512_begin_0, end = var_1512_end_0, end_mask = var_1512_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1512_cast_fp16")]; + tensor var_1516_begin_0 = const()[name = string("op_1516_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_1516_end_0 = const()[name = string("op_1516_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_1516_end_mask_0 = const()[name = string("op_1516_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1516_cast_fp16 = slice_by_index(begin = var_1516_begin_0, end = var_1516_end_0, end_mask = var_1516_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1516_cast_fp16")]; + tensor var_1528_begin_0 = const()[name = string("op_1528_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_1528_end_0 = const()[name = string("op_1528_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_1528_end_mask_0 = const()[name = string("op_1528_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1528_cast_fp16 = slice_by_index(begin = var_1528_begin_0, end = var_1528_end_0, end_mask = var_1528_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1528_cast_fp16")]; + tensor var_1532_begin_0 = const()[name = string("op_1532_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_1532_end_0 = const()[name = string("op_1532_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_1532_end_mask_0 = const()[name = string("op_1532_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1532_cast_fp16 = slice_by_index(begin = var_1532_begin_0, end = var_1532_end_0, end_mask = var_1532_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1532_cast_fp16")]; + tensor var_1544_begin_0 = const()[name = string("op_1544_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_1544_end_0 = const()[name = string("op_1544_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_1544_end_mask_0 = const()[name = string("op_1544_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1544_cast_fp16 = slice_by_index(begin = var_1544_begin_0, end = var_1544_end_0, end_mask = var_1544_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1544_cast_fp16")]; + tensor var_1548_begin_0 = const()[name = string("op_1548_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_1548_end_0 = const()[name = string("op_1548_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_1548_end_mask_0 = const()[name = string("op_1548_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1548_cast_fp16 = slice_by_index(begin = var_1548_begin_0, end = var_1548_end_0, end_mask = var_1548_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1548_cast_fp16")]; + tensor var_1560_begin_0 = const()[name = string("op_1560_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_1560_end_0 = const()[name = string("op_1560_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1560_end_mask_0 = const()[name = string("op_1560_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1560_cast_fp16 = slice_by_index(begin = var_1560_begin_0, end = var_1560_end_0, end_mask = var_1560_end_mask_0, x = key_heads_13_cast_fp16)[name = string("op_1560_cast_fp16")]; + tensor var_1564_begin_0 = const()[name = string("op_1564_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_1564_end_0 = const()[name = string("op_1564_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1564_end_mask_0 = const()[name = string("op_1564_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1564_cast_fp16 = slice_by_index(begin = var_1564_begin_0, end = var_1564_end_0, end_mask = var_1564_end_mask_0, x = value_heads_13_cast_fp16)[name = string("op_1564_cast_fp16")]; + bool key_heads_15_interleave_0 = const()[name = string("key_heads_15_interleave_0"), val = bool(false)]; + tensor key_heads_15_cast_fp16 = concat(axis = var_1290, interleave = key_heads_15_interleave_0, values = (var_1448_cast_fp16, var_1448_cast_fp16, var_1464_cast_fp16, var_1464_cast_fp16, var_1480_cast_fp16, var_1480_cast_fp16, var_1496_cast_fp16, var_1496_cast_fp16, var_1512_cast_fp16, var_1512_cast_fp16, var_1528_cast_fp16, var_1528_cast_fp16, var_1544_cast_fp16, var_1544_cast_fp16, var_1560_cast_fp16, var_1560_cast_fp16))[name = string("key_heads_15_cast_fp16")]; + bool value_heads_15_interleave_0 = const()[name = string("value_heads_15_interleave_0"), val = bool(false)]; + tensor value_heads_15_cast_fp16 = concat(axis = var_1290, interleave = value_heads_15_interleave_0, values = (var_1452_cast_fp16, var_1452_cast_fp16, var_1468_cast_fp16, var_1468_cast_fp16, var_1484_cast_fp16, var_1484_cast_fp16, var_1500_cast_fp16, var_1500_cast_fp16, var_1516_cast_fp16, var_1516_cast_fp16, var_1532_cast_fp16, var_1532_cast_fp16, var_1548_cast_fp16, var_1548_cast_fp16, var_1564_cast_fp16, var_1564_cast_fp16))[name = string("value_heads_15_cast_fp16")]; + fp16 var_1587_to_fp16 = const()[name = string("op_1587_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_1588_cast_fp16 = mul(x = mh_q_21_cast_fp16, y = var_1587_to_fp16)[name = string("op_1588_cast_fp16")]; + bool mh_w_13_transpose_x_0 = const()[name = string("mh_w_13_transpose_x_0"), val = bool(true)]; + bool mh_w_13_transpose_y_0 = const()[name = string("mh_w_13_transpose_y_0"), val = bool(false)]; + tensor mh_w_13_cast_fp16 = matmul(transpose_x = mh_w_13_transpose_x_0, transpose_y = mh_w_13_transpose_y_0, x = var_1588_cast_fp16, y = key_heads_15_cast_fp16)[name = string("mh_w_13_cast_fp16")]; + tensor mh_w_15_cast_fp16 = add(x = mh_w_13_cast_fp16, y = var_436_cast_fp16)[name = string("mh_w_15_cast_fp16")]; + tensor var_1600_cast_fp16 = softmax(axis = var_1272, x = mh_w_15_cast_fp16)[name = string("op_1600_cast_fp16")]; + bool attn_7_transpose_x_0 = const()[name = string("attn_7_transpose_x_0"), val = bool(false)]; + bool attn_7_transpose_y_0 = const()[name = string("attn_7_transpose_y_0"), val = bool(true)]; + tensor attn_7_cast_fp16 = matmul(transpose_x = attn_7_transpose_x_0, transpose_y = attn_7_transpose_y_0, x = value_heads_15_cast_fp16, y = var_1600_cast_fp16)[name = string("attn_7_cast_fp16")]; + tensor var_1605 = const()[name = string("op_1605"), val = tensor([1, -1, 1, 1])]; + tensor input_25_cast_fp16 = reshape(shape = var_1605, x = attn_7_cast_fp16)[name = string("input_25_cast_fp16")]; + string obj_35_pad_type_0 = const()[name = string("obj_35_pad_type_0"), val = string("valid")]; + tensor obj_35_strides_0 = const()[name = string("obj_35_strides_0"), val = tensor([1, 1])]; + tensor obj_35_pad_0 = const()[name = string("obj_35_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_35_dilations_0 = const()[name = string("obj_35_dilations_0"), val = tensor([1, 1])]; + int32 obj_35_groups_0 = const()[name = string("obj_35_groups_0"), val = int32(1)]; + tensor layers_3_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53527488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55624704))))[name = string("layers_3_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_35_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_35_dilations_0, groups = obj_35_groups_0, pad = obj_35_pad_0, pad_type = obj_35_pad_type_0, strides = obj_35_strides_0, weight = layers_3_self_attn_o_proj_weight_to_fp16_palettized, x = input_25_cast_fp16)[name = string("obj_35_cast_fp16")]; + tensor inputs_31_cast_fp16 = add(x = inputs_25_cast_fp16, y = obj_35_cast_fp16)[name = string("inputs_31_cast_fp16")]; + tensor inputs_sq_31_cast_fp16 = mul(x = inputs_31_cast_fp16, y = inputs_31_cast_fp16)[name = string("inputs_sq_31_cast_fp16")]; + tensor variance_31_axes_0 = const()[name = string("variance_31_axes_0"), val = tensor([1])]; + bool variance_31_keep_dims_0 = const()[name = string("variance_31_keep_dims_0"), val = bool(true)]; + tensor variance_31_cast_fp16 = reduce_mean(axes = variance_31_axes_0, keep_dims = variance_31_keep_dims_0, x = inputs_sq_31_cast_fp16)[name = string("variance_31_cast_fp16")]; + fp16 var_1623_to_fp16 = const()[name = string("op_1623_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1624_cast_fp16 = add(x = variance_31_cast_fp16, y = var_1623_to_fp16)[name = string("op_1624_cast_fp16")]; + fp32 var_1625_epsilon_0 = const()[name = string("op_1625_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1625_cast_fp16 = rsqrt(epsilon = var_1625_epsilon_0, x = var_1624_cast_fp16)[name = string("op_1625_cast_fp16")]; + tensor hidden_states_37_cast_fp16 = mul(x = inputs_31_cast_fp16, y = var_1625_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; + tensor w_31_to_fp16 = const()[name = string("w_31_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55625280)))]; + tensor input_27_cast_fp16 = mul(x = w_31_to_fp16, y = hidden_states_37_cast_fp16)[name = string("input_27_cast_fp16")]; + string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; + tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; + tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; + int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; + tensor layers_3_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55627392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(58773184))))[name = string("layers_3_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_29_cast_fp16 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16_palettized, x = input_27_cast_fp16)[name = string("input_29_cast_fp16")]; + tensor var_1639_cast_fp16 = silu(x = input_29_cast_fp16)[name = string("op_1639_cast_fp16")]; + string var_1645_pad_type_0 = const()[name = string("op_1645_pad_type_0"), val = string("valid")]; + tensor var_1645_strides_0 = const()[name = string("op_1645_strides_0"), val = tensor([1, 1])]; + tensor var_1645_pad_0 = const()[name = string("op_1645_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1645_dilations_0 = const()[name = string("op_1645_dilations_0"), val = tensor([1, 1])]; + int32 var_1645_groups_0 = const()[name = string("op_1645_groups_0"), val = int32(1)]; + tensor layers_3_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(58773760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61919552))))[name = string("layers_3_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_1645_cast_fp16 = conv(dilations = var_1645_dilations_0, groups = var_1645_groups_0, pad = var_1645_pad_0, pad_type = var_1645_pad_type_0, strides = var_1645_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16_palettized, x = input_27_cast_fp16)[name = string("op_1645_cast_fp16")]; + tensor input_31_cast_fp16 = mul(x = var_1639_cast_fp16, y = var_1645_cast_fp16)[name = string("input_31_cast_fp16")]; + string hidden_states_39_pad_type_0 = const()[name = string("hidden_states_39_pad_type_0"), val = string("valid")]; + tensor hidden_states_39_strides_0 = const()[name = string("hidden_states_39_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_39_pad_0 = const()[name = string("hidden_states_39_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_39_dilations_0 = const()[name = string("hidden_states_39_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_39_groups_0 = const()[name = string("hidden_states_39_groups_0"), val = int32(1)]; + tensor layers_3_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61920128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65065920))))[name = string("layers_3_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_39_cast_fp16 = conv(dilations = hidden_states_39_dilations_0, groups = hidden_states_39_groups_0, pad = hidden_states_39_pad_0, pad_type = hidden_states_39_pad_type_0, strides = hidden_states_39_strides_0, weight = layers_3_mlp_down_proj_weight_to_fp16_palettized, x = input_31_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; + tensor inputs_33_cast_fp16 = add(x = inputs_31_cast_fp16, y = hidden_states_39_cast_fp16)[name = string("inputs_33_cast_fp16")]; + int32 var_1659 = const()[name = string("op_1659"), val = int32(3)]; + int32 var_1669 = const()[name = string("op_1669"), val = int32(-2)]; + int32 var_1677 = const()[name = string("op_1677"), val = int32(1)]; + tensor inputs_sq_33_cast_fp16 = mul(x = inputs_33_cast_fp16, y = inputs_33_cast_fp16)[name = string("inputs_sq_33_cast_fp16")]; + tensor variance_33_axes_0 = const()[name = string("variance_33_axes_0"), val = tensor([1])]; + bool variance_33_keep_dims_0 = const()[name = string("variance_33_keep_dims_0"), val = bool(true)]; + tensor variance_33_cast_fp16 = reduce_mean(axes = variance_33_axes_0, keep_dims = variance_33_keep_dims_0, x = inputs_sq_33_cast_fp16)[name = string("variance_33_cast_fp16")]; + fp16 var_1689_to_fp16 = const()[name = string("op_1689_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1690_cast_fp16 = add(x = variance_33_cast_fp16, y = var_1689_to_fp16)[name = string("op_1690_cast_fp16")]; + fp32 var_1691_epsilon_0 = const()[name = string("op_1691_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1691_cast_fp16 = rsqrt(epsilon = var_1691_epsilon_0, x = var_1690_cast_fp16)[name = string("op_1691_cast_fp16")]; + tensor hidden_states_41_cast_fp16 = mul(x = inputs_33_cast_fp16, y = var_1691_cast_fp16)[name = string("hidden_states_41_cast_fp16")]; + tensor w_33_to_fp16 = const()[name = string("w_33_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65066496)))]; + tensor obj_37_cast_fp16 = mul(x = w_33_to_fp16, y = hidden_states_41_cast_fp16)[name = string("obj_37_cast_fp16")]; + string query_25_pad_type_0 = const()[name = string("query_25_pad_type_0"), val = string("valid")]; + tensor query_25_strides_0 = const()[name = string("query_25_strides_0"), val = tensor([1, 1])]; + tensor query_25_pad_0 = const()[name = string("query_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_25_dilations_0 = const()[name = string("query_25_dilations_0"), val = tensor([1, 1])]; + int32 query_25_groups_0 = const()[name = string("query_25_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_q_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(65068608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67165824))))[name = string("layers_4_self_attn_q_proj_weight_to_fp16_palettized")]; + tensor query_25_cast_fp16 = conv(bias = layers_0_self_attn_q_proj_bias_to_fp16, dilations = query_25_dilations_0, groups = query_25_groups_0, pad = query_25_pad_0, pad_type = query_25_pad_type_0, strides = query_25_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16_palettized, x = obj_37_cast_fp16)[name = string("query_25_cast_fp16")]; + string current_key_17_pad_type_0 = const()[name = string("current_key_17_pad_type_0"), val = string("valid")]; + tensor current_key_17_strides_0 = const()[name = string("current_key_17_strides_0"), val = tensor([1, 1])]; + tensor current_key_17_pad_0 = const()[name = string("current_key_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_key_17_dilations_0 = const()[name = string("current_key_17_dilations_0"), val = tensor([1, 1])]; + int32 current_key_17_groups_0 = const()[name = string("current_key_17_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_k_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67166400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68215040))))[name = string("layers_4_self_attn_k_proj_weight_to_fp16_palettized")]; + tensor current_key_17_cast_fp16 = conv(dilations = current_key_17_dilations_0, groups = current_key_17_groups_0, pad = current_key_17_pad_0, pad_type = current_key_17_pad_type_0, strides = current_key_17_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16_palettized, x = obj_37_cast_fp16)[name = string("current_key_17_cast_fp16")]; + string current_value_pad_type_0 = const()[name = string("current_value_pad_type_0"), val = string("valid")]; + tensor current_value_strides_0 = const()[name = string("current_value_strides_0"), val = tensor([1, 1])]; + tensor current_value_pad_0 = const()[name = string("current_value_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor current_value_dilations_0 = const()[name = string("current_value_dilations_0"), val = tensor([1, 1])]; + int32 current_value_groups_0 = const()[name = string("current_value_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_v_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68215616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69264256))))[name = string("layers_4_self_attn_v_proj_weight_to_fp16_palettized")]; + tensor current_value_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = current_value_dilations_0, groups = current_value_groups_0, pad = current_value_pad_0, pad_type = current_value_pad_type_0, strides = current_value_strides_0, weight = layers_4_self_attn_v_proj_weight_to_fp16_palettized, x = obj_37_cast_fp16)[name = string("current_value_cast_fp16")]; + tensor var_1728 = const()[name = string("op_1728"), val = tensor([16, 128, 1, 1])]; + tensor inputs_35_cast_fp16 = reshape(shape = var_1728, x = query_25_cast_fp16)[name = string("inputs_35_cast_fp16")]; + tensor inputs_sq_35_cast_fp16 = mul(x = inputs_35_cast_fp16, y = inputs_35_cast_fp16)[name = string("inputs_sq_35_cast_fp16")]; + tensor variance_35_axes_0 = const()[name = string("variance_35_axes_0"), val = tensor([1])]; + bool variance_35_keep_dims_0 = const()[name = string("variance_35_keep_dims_0"), val = bool(true)]; + tensor variance_35_cast_fp16 = reduce_mean(axes = variance_35_axes_0, keep_dims = variance_35_keep_dims_0, x = inputs_sq_35_cast_fp16)[name = string("variance_35_cast_fp16")]; + fp16 var_1734_to_fp16 = const()[name = string("op_1734_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1735_cast_fp16 = add(x = variance_35_cast_fp16, y = var_1734_to_fp16)[name = string("op_1735_cast_fp16")]; + fp32 var_1736_epsilon_0 = const()[name = string("op_1736_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1736_cast_fp16 = rsqrt(epsilon = var_1736_epsilon_0, x = var_1735_cast_fp16)[name = string("op_1736_cast_fp16")]; + tensor hidden_states_43_cast_fp16 = mul(x = inputs_35_cast_fp16, y = var_1736_cast_fp16)[name = string("hidden_states_43_cast_fp16")]; + tensor w_35_to_fp16 = const()[name = string("w_35_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69264832)))]; + tensor query_normed_cast_fp16 = mul(x = w_35_to_fp16, y = hidden_states_43_cast_fp16)[name = string("query_normed_cast_fp16")]; + tensor var_1744 = const()[name = string("op_1744"), val = tensor([8, 128, 1, 1])]; + tensor inputs_37_cast_fp16 = reshape(shape = var_1744, x = current_key_17_cast_fp16)[name = string("inputs_37_cast_fp16")]; + tensor inputs_sq_37_cast_fp16 = mul(x = inputs_37_cast_fp16, y = inputs_37_cast_fp16)[name = string("inputs_sq_37_cast_fp16")]; + tensor variance_37_axes_0 = const()[name = string("variance_37_axes_0"), val = tensor([1])]; + bool variance_37_keep_dims_0 = const()[name = string("variance_37_keep_dims_0"), val = bool(true)]; + tensor variance_37_cast_fp16 = reduce_mean(axes = variance_37_axes_0, keep_dims = variance_37_keep_dims_0, x = inputs_sq_37_cast_fp16)[name = string("variance_37_cast_fp16")]; + fp16 var_1750_to_fp16 = const()[name = string("op_1750_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_1751_cast_fp16 = add(x = variance_37_cast_fp16, y = var_1750_to_fp16)[name = string("op_1751_cast_fp16")]; + fp32 var_1752_epsilon_0 = const()[name = string("op_1752_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_1752_cast_fp16 = rsqrt(epsilon = var_1752_epsilon_0, x = var_1751_cast_fp16)[name = string("op_1752_cast_fp16")]; + tensor hidden_states_45_cast_fp16 = mul(x = inputs_37_cast_fp16, y = var_1752_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; + tensor w_37_to_fp16 = const()[name = string("w_37_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69265152)))]; + tensor current_key_normed_cast_fp16 = mul(x = w_37_to_fp16, y = hidden_states_45_cast_fp16)[name = string("current_key_normed_cast_fp16")]; + tensor var_1770 = const()[name = string("op_1770"), val = tensor([1, 16, 128, -1])]; + tensor mh_q_25_cast_fp16 = reshape(shape = var_1770, x = query_normed_cast_fp16)[name = string("mh_q_25_cast_fp16")]; + tensor var_1772 = const()[name = string("op_1772"), val = tensor([1, 8, 128, -1])]; + tensor mh_k_17_cast_fp16 = reshape(shape = var_1772, x = current_key_normed_cast_fp16)[name = string("mh_k_17_cast_fp16")]; + tensor var_1776_cast_fp16 = mul(x = mh_q_25_cast_fp16, y = cos_1_cast_fp16)[name = string("op_1776_cast_fp16")]; + tensor var_1781_begin_0 = const()[name = string("op_1781_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1781_end_0 = const()[name = string("op_1781_end_0"), val = tensor([1, 16, 64, 1])]; + tensor var_1781_end_mask_0 = const()[name = string("op_1781_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1781_cast_fp16 = slice_by_index(begin = var_1781_begin_0, end = var_1781_end_0, end_mask = var_1781_end_mask_0, x = mh_q_25_cast_fp16)[name = string("op_1781_cast_fp16")]; + tensor var_1787_begin_0 = const()[name = string("op_1787_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1787_end_0 = const()[name = string("op_1787_end_0"), val = tensor([1, 16, 128, 1])]; + tensor var_1787_end_mask_0 = const()[name = string("op_1787_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1787_cast_fp16 = slice_by_index(begin = var_1787_begin_0, end = var_1787_end_0, end_mask = var_1787_end_mask_0, x = mh_q_25_cast_fp16)[name = string("op_1787_cast_fp16")]; + fp16 const_109_promoted_to_fp16 = const()[name = string("const_109_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1789_cast_fp16 = mul(x = var_1787_cast_fp16, y = const_109_promoted_to_fp16)[name = string("op_1789_cast_fp16")]; + bool var_1791_interleave_0 = const()[name = string("op_1791_interleave_0"), val = bool(false)]; + tensor var_1791_cast_fp16 = concat(axis = var_1669, interleave = var_1791_interleave_0, values = (var_1789_cast_fp16, var_1781_cast_fp16))[name = string("op_1791_cast_fp16")]; + tensor var_1792_cast_fp16 = mul(x = var_1791_cast_fp16, y = sin_1_cast_fp16)[name = string("op_1792_cast_fp16")]; + tensor mh_q_27_cast_fp16 = add(x = var_1776_cast_fp16, y = var_1792_cast_fp16)[name = string("mh_q_27_cast_fp16")]; + tensor var_1794_cast_fp16 = mul(x = mh_k_17_cast_fp16, y = cos_1_cast_fp16)[name = string("op_1794_cast_fp16")]; + tensor var_1799_begin_0 = const()[name = string("op_1799_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1799_end_0 = const()[name = string("op_1799_end_0"), val = tensor([1, 8, 64, 1])]; + tensor var_1799_end_mask_0 = const()[name = string("op_1799_end_mask_0"), val = tensor([true, true, false, true])]; + tensor var_1799_cast_fp16 = slice_by_index(begin = var_1799_begin_0, end = var_1799_end_0, end_mask = var_1799_end_mask_0, x = mh_k_17_cast_fp16)[name = string("op_1799_cast_fp16")]; + tensor var_1805_begin_0 = const()[name = string("op_1805_begin_0"), val = tensor([0, 0, 64, 0])]; + tensor var_1805_end_0 = const()[name = string("op_1805_end_0"), val = tensor([1, 8, 128, 1])]; + tensor var_1805_end_mask_0 = const()[name = string("op_1805_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1805_cast_fp16 = slice_by_index(begin = var_1805_begin_0, end = var_1805_end_0, end_mask = var_1805_end_mask_0, x = mh_k_17_cast_fp16)[name = string("op_1805_cast_fp16")]; + fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1807_cast_fp16 = mul(x = var_1805_cast_fp16, y = const_112_promoted_to_fp16)[name = string("op_1807_cast_fp16")]; + bool var_1809_interleave_0 = const()[name = string("op_1809_interleave_0"), val = bool(false)]; + tensor var_1809_cast_fp16 = concat(axis = var_1669, interleave = var_1809_interleave_0, values = (var_1807_cast_fp16, var_1799_cast_fp16))[name = string("op_1809_cast_fp16")]; + tensor var_1810_cast_fp16 = mul(x = var_1809_cast_fp16, y = sin_1_cast_fp16)[name = string("op_1810_cast_fp16")]; + tensor mh_k_cast_fp16 = add(x = var_1794_cast_fp16, y = var_1810_cast_fp16)[name = string("mh_k_cast_fp16")]; + tensor var_1814 = const()[name = string("op_1814"), val = tensor([1, 1024, 1, 1])]; + tensor current_key_cast_fp16 = reshape(shape = var_1814, x = mh_k_cast_fp16)[name = string("current_key_cast_fp16")]; + tensor var_1821_cast_fp16 = mul(x = var_96_cast_fp16_4, y = var_272_cast_fp16)[name = string("op_1821_cast_fp16")]; + tensor var_1822_cast_fp16 = mul(x = current_key_cast_fp16, y = var_270_cast_fp16)[name = string("op_1822_cast_fp16")]; + tensor key_27_cast_fp16 = add(x = var_1821_cast_fp16, y = var_1822_cast_fp16)[name = string("key_27_cast_fp16")]; + tensor var_1825_cast_fp16 = mul(x = var_104_cast_fp16_4, y = var_272_cast_fp16)[name = string("op_1825_cast_fp16")]; + tensor var_1826_cast_fp16 = mul(x = current_value_cast_fp16, y = var_270_cast_fp16)[name = string("op_1826_cast_fp16")]; + tensor value_17_cast_fp16 = add(x = var_1825_cast_fp16, y = var_1826_cast_fp16)[name = string("value_17_cast_fp16")]; + tensor var_1830 = const()[name = string("op_1830"), val = tensor([1, 8, 128, 16])]; + tensor key_heads_17_cast_fp16 = reshape(shape = var_1830, x = key_27_cast_fp16)[name = string("key_heads_17_cast_fp16")]; + tensor var_1832 = const()[name = string("op_1832"), val = tensor([1, 8, 128, 16])]; + tensor value_heads_17_cast_fp16 = reshape(shape = var_1832, x = value_17_cast_fp16)[name = string("value_heads_17_cast_fp16")]; + tensor var_1835_begin_0 = const()[name = string("op_1835_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1835_end_0 = const()[name = string("op_1835_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1835_end_mask_0 = const()[name = string("op_1835_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1835_cast_fp16 = slice_by_index(begin = var_1835_begin_0, end = var_1835_end_0, end_mask = var_1835_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1835_cast_fp16")]; + tensor var_1839_begin_0 = const()[name = string("op_1839_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1839_end_0 = const()[name = string("op_1839_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1839_end_mask_0 = const()[name = string("op_1839_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1839_cast_fp16 = slice_by_index(begin = var_1839_begin_0, end = var_1839_end_0, end_mask = var_1839_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1839_cast_fp16")]; + tensor var_1851_begin_0 = const()[name = string("op_1851_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_1851_end_0 = const()[name = string("op_1851_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_1851_end_mask_0 = const()[name = string("op_1851_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1851_cast_fp16 = slice_by_index(begin = var_1851_begin_0, end = var_1851_end_0, end_mask = var_1851_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1851_cast_fp16")]; + tensor var_1855_begin_0 = const()[name = string("op_1855_begin_0"), val = tensor([0, 1, 0, 0])]; + tensor var_1855_end_0 = const()[name = string("op_1855_end_0"), val = tensor([1, 2, 128, 16])]; + tensor var_1855_end_mask_0 = const()[name = string("op_1855_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1855_cast_fp16 = slice_by_index(begin = var_1855_begin_0, end = var_1855_end_0, end_mask = var_1855_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1855_cast_fp16")]; + tensor var_1867_begin_0 = const()[name = string("op_1867_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_1867_end_0 = const()[name = string("op_1867_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_1867_end_mask_0 = const()[name = string("op_1867_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1867_cast_fp16 = slice_by_index(begin = var_1867_begin_0, end = var_1867_end_0, end_mask = var_1867_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1867_cast_fp16")]; + tensor var_1871_begin_0 = const()[name = string("op_1871_begin_0"), val = tensor([0, 2, 0, 0])]; + tensor var_1871_end_0 = const()[name = string("op_1871_end_0"), val = tensor([1, 3, 128, 16])]; + tensor var_1871_end_mask_0 = const()[name = string("op_1871_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1871_cast_fp16 = slice_by_index(begin = var_1871_begin_0, end = var_1871_end_0, end_mask = var_1871_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1871_cast_fp16")]; + tensor var_1883_begin_0 = const()[name = string("op_1883_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_1883_end_0 = const()[name = string("op_1883_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_1883_end_mask_0 = const()[name = string("op_1883_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1883_cast_fp16 = slice_by_index(begin = var_1883_begin_0, end = var_1883_end_0, end_mask = var_1883_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1883_cast_fp16")]; + tensor var_1887_begin_0 = const()[name = string("op_1887_begin_0"), val = tensor([0, 3, 0, 0])]; + tensor var_1887_end_0 = const()[name = string("op_1887_end_0"), val = tensor([1, 4, 128, 16])]; + tensor var_1887_end_mask_0 = const()[name = string("op_1887_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1887_cast_fp16 = slice_by_index(begin = var_1887_begin_0, end = var_1887_end_0, end_mask = var_1887_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1887_cast_fp16")]; + tensor var_1899_begin_0 = const()[name = string("op_1899_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_1899_end_0 = const()[name = string("op_1899_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_1899_end_mask_0 = const()[name = string("op_1899_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1899_cast_fp16 = slice_by_index(begin = var_1899_begin_0, end = var_1899_end_0, end_mask = var_1899_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1899_cast_fp16")]; + tensor var_1903_begin_0 = const()[name = string("op_1903_begin_0"), val = tensor([0, 4, 0, 0])]; + tensor var_1903_end_0 = const()[name = string("op_1903_end_0"), val = tensor([1, 5, 128, 16])]; + tensor var_1903_end_mask_0 = const()[name = string("op_1903_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1903_cast_fp16 = slice_by_index(begin = var_1903_begin_0, end = var_1903_end_0, end_mask = var_1903_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1903_cast_fp16")]; + tensor var_1915_begin_0 = const()[name = string("op_1915_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_1915_end_0 = const()[name = string("op_1915_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_1915_end_mask_0 = const()[name = string("op_1915_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1915_cast_fp16 = slice_by_index(begin = var_1915_begin_0, end = var_1915_end_0, end_mask = var_1915_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1915_cast_fp16")]; + tensor var_1919_begin_0 = const()[name = string("op_1919_begin_0"), val = tensor([0, 5, 0, 0])]; + tensor var_1919_end_0 = const()[name = string("op_1919_end_0"), val = tensor([1, 6, 128, 16])]; + tensor var_1919_end_mask_0 = const()[name = string("op_1919_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1919_cast_fp16 = slice_by_index(begin = var_1919_begin_0, end = var_1919_end_0, end_mask = var_1919_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1919_cast_fp16")]; + tensor var_1931_begin_0 = const()[name = string("op_1931_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_1931_end_0 = const()[name = string("op_1931_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_1931_end_mask_0 = const()[name = string("op_1931_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1931_cast_fp16 = slice_by_index(begin = var_1931_begin_0, end = var_1931_end_0, end_mask = var_1931_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1931_cast_fp16")]; + tensor var_1935_begin_0 = const()[name = string("op_1935_begin_0"), val = tensor([0, 6, 0, 0])]; + tensor var_1935_end_0 = const()[name = string("op_1935_end_0"), val = tensor([1, 7, 128, 16])]; + tensor var_1935_end_mask_0 = const()[name = string("op_1935_end_mask_0"), val = tensor([true, false, true, true])]; + tensor var_1935_cast_fp16 = slice_by_index(begin = var_1935_begin_0, end = var_1935_end_0, end_mask = var_1935_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1935_cast_fp16")]; + tensor var_1947_begin_0 = const()[name = string("op_1947_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_1947_end_0 = const()[name = string("op_1947_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1947_end_mask_0 = const()[name = string("op_1947_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1947_cast_fp16 = slice_by_index(begin = var_1947_begin_0, end = var_1947_end_0, end_mask = var_1947_end_mask_0, x = key_heads_17_cast_fp16)[name = string("op_1947_cast_fp16")]; + tensor var_1951_begin_0 = const()[name = string("op_1951_begin_0"), val = tensor([0, 7, 0, 0])]; + tensor var_1951_end_0 = const()[name = string("op_1951_end_0"), val = tensor([1, 1, 128, 16])]; + tensor var_1951_end_mask_0 = const()[name = string("op_1951_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_1951_cast_fp16 = slice_by_index(begin = var_1951_begin_0, end = var_1951_end_0, end_mask = var_1951_end_mask_0, x = value_heads_17_cast_fp16)[name = string("op_1951_cast_fp16")]; + bool key_heads_interleave_0 = const()[name = string("key_heads_interleave_0"), val = bool(false)]; + tensor key_heads_cast_fp16 = concat(axis = var_1677, interleave = key_heads_interleave_0, values = (var_1835_cast_fp16, var_1835_cast_fp16, var_1851_cast_fp16, var_1851_cast_fp16, var_1867_cast_fp16, var_1867_cast_fp16, var_1883_cast_fp16, var_1883_cast_fp16, var_1899_cast_fp16, var_1899_cast_fp16, var_1915_cast_fp16, var_1915_cast_fp16, var_1931_cast_fp16, var_1931_cast_fp16, var_1947_cast_fp16, var_1947_cast_fp16))[name = string("key_heads_cast_fp16")]; + bool value_heads_interleave_0 = const()[name = string("value_heads_interleave_0"), val = bool(false)]; + tensor value_heads_cast_fp16 = concat(axis = var_1677, interleave = value_heads_interleave_0, values = (var_1839_cast_fp16, var_1839_cast_fp16, var_1855_cast_fp16, var_1855_cast_fp16, var_1871_cast_fp16, var_1871_cast_fp16, var_1887_cast_fp16, var_1887_cast_fp16, var_1903_cast_fp16, var_1903_cast_fp16, var_1919_cast_fp16, var_1919_cast_fp16, var_1935_cast_fp16, var_1935_cast_fp16, var_1951_cast_fp16, var_1951_cast_fp16))[name = string("value_heads_cast_fp16")]; + fp16 var_1974_to_fp16 = const()[name = string("op_1974_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor var_1975_cast_fp16 = mul(x = mh_q_27_cast_fp16, y = var_1974_to_fp16)[name = string("op_1975_cast_fp16")]; + bool mh_w_17_transpose_x_0 = const()[name = string("mh_w_17_transpose_x_0"), val = bool(true)]; + bool mh_w_17_transpose_y_0 = const()[name = string("mh_w_17_transpose_y_0"), val = bool(false)]; + tensor mh_w_17_cast_fp16 = matmul(transpose_x = mh_w_17_transpose_x_0, transpose_y = mh_w_17_transpose_y_0, x = var_1975_cast_fp16, y = key_heads_cast_fp16)[name = string("mh_w_17_cast_fp16")]; + tensor mh_w_cast_fp16 = add(x = mh_w_17_cast_fp16, y = var_436_cast_fp16)[name = string("mh_w_cast_fp16")]; + tensor var_1987_cast_fp16 = softmax(axis = var_1659, x = mh_w_cast_fp16)[name = string("op_1987_cast_fp16")]; + bool attn_transpose_x_0 = const()[name = string("attn_transpose_x_0"), val = bool(false)]; + bool attn_transpose_y_0 = const()[name = string("attn_transpose_y_0"), val = bool(true)]; + tensor attn_cast_fp16 = matmul(transpose_x = attn_transpose_x_0, transpose_y = attn_transpose_y_0, x = value_heads_cast_fp16, y = var_1987_cast_fp16)[name = string("attn_cast_fp16")]; + tensor var_1992 = const()[name = string("op_1992"), val = tensor([1, -1, 1, 1])]; + tensor input_33_cast_fp16 = reshape(shape = var_1992, x = attn_cast_fp16)[name = string("input_33_cast_fp16")]; + string obj_pad_type_0 = const()[name = string("obj_pad_type_0"), val = string("valid")]; + tensor obj_strides_0 = const()[name = string("obj_strides_0"), val = tensor([1, 1])]; + tensor obj_pad_0 = const()[name = string("obj_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor obj_dilations_0 = const()[name = string("obj_dilations_0"), val = tensor([1, 1])]; + int32 obj_groups_0 = const()[name = string("obj_groups_0"), val = int32(1)]; + tensor layers_4_self_attn_o_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69265472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(71362688))))[name = string("layers_4_self_attn_o_proj_weight_to_fp16_palettized")]; + tensor obj_cast_fp16 = conv(bias = layers_0_self_attn_v_proj_bias_to_fp16, dilations = obj_dilations_0, groups = obj_groups_0, pad = obj_pad_0, pad_type = obj_pad_type_0, strides = obj_strides_0, weight = layers_4_self_attn_o_proj_weight_to_fp16_palettized, x = input_33_cast_fp16)[name = string("obj_cast_fp16")]; + tensor inputs_39_cast_fp16 = add(x = inputs_33_cast_fp16, y = obj_cast_fp16)[name = string("inputs_39_cast_fp16")]; + tensor inputs_sq_39_cast_fp16 = mul(x = inputs_39_cast_fp16, y = inputs_39_cast_fp16)[name = string("inputs_sq_39_cast_fp16")]; + tensor variance_39_axes_0 = const()[name = string("variance_39_axes_0"), val = tensor([1])]; + bool variance_39_keep_dims_0 = const()[name = string("variance_39_keep_dims_0"), val = bool(true)]; + tensor variance_39_cast_fp16 = reduce_mean(axes = variance_39_axes_0, keep_dims = variance_39_keep_dims_0, x = inputs_sq_39_cast_fp16)[name = string("variance_39_cast_fp16")]; + fp16 var_2010_to_fp16 = const()[name = string("op_2010_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2011_cast_fp16 = add(x = variance_39_cast_fp16, y = var_2010_to_fp16)[name = string("op_2011_cast_fp16")]; + fp32 var_2012_epsilon_0 = const()[name = string("op_2012_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2012_cast_fp16 = rsqrt(epsilon = var_2012_epsilon_0, x = var_2011_cast_fp16)[name = string("op_2012_cast_fp16")]; + tensor hidden_states_47_cast_fp16 = mul(x = inputs_39_cast_fp16, y = var_2012_cast_fp16)[name = string("hidden_states_47_cast_fp16")]; + tensor w_39_to_fp16 = const()[name = string("w_39_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(71363264)))]; + tensor input_35_cast_fp16 = mul(x = w_39_to_fp16, y = hidden_states_47_cast_fp16)[name = string("input_35_cast_fp16")]; + string input_37_pad_type_0 = const()[name = string("input_37_pad_type_0"), val = string("valid")]; + tensor input_37_strides_0 = const()[name = string("input_37_strides_0"), val = tensor([1, 1])]; + tensor input_37_pad_0 = const()[name = string("input_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_37_dilations_0 = const()[name = string("input_37_dilations_0"), val = tensor([1, 1])]; + int32 input_37_groups_0 = const()[name = string("input_37_groups_0"), val = int32(1)]; + tensor layers_4_mlp_gate_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(71365376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(74511168))))[name = string("layers_4_mlp_gate_proj_weight_to_fp16_palettized")]; + tensor input_37_cast_fp16 = conv(dilations = input_37_dilations_0, groups = input_37_groups_0, pad = input_37_pad_0, pad_type = input_37_pad_type_0, strides = input_37_strides_0, weight = layers_4_mlp_gate_proj_weight_to_fp16_palettized, x = input_35_cast_fp16)[name = string("input_37_cast_fp16")]; + tensor var_2026_cast_fp16 = silu(x = input_37_cast_fp16)[name = string("op_2026_cast_fp16")]; + string var_2032_pad_type_0 = const()[name = string("op_2032_pad_type_0"), val = string("valid")]; + tensor var_2032_strides_0 = const()[name = string("op_2032_strides_0"), val = tensor([1, 1])]; + tensor var_2032_pad_0 = const()[name = string("op_2032_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2032_dilations_0 = const()[name = string("op_2032_dilations_0"), val = tensor([1, 1])]; + int32 var_2032_groups_0 = const()[name = string("op_2032_groups_0"), val = int32(1)]; + tensor layers_4_mlp_up_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(74511744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(77657536))))[name = string("layers_4_mlp_up_proj_weight_to_fp16_palettized")]; + tensor var_2032_cast_fp16 = conv(dilations = var_2032_dilations_0, groups = var_2032_groups_0, pad = var_2032_pad_0, pad_type = var_2032_pad_type_0, strides = var_2032_strides_0, weight = layers_4_mlp_up_proj_weight_to_fp16_palettized, x = input_35_cast_fp16)[name = string("op_2032_cast_fp16")]; + tensor input_39_cast_fp16 = mul(x = var_2026_cast_fp16, y = var_2032_cast_fp16)[name = string("input_39_cast_fp16")]; + string hidden_states_49_pad_type_0 = const()[name = string("hidden_states_49_pad_type_0"), val = string("valid")]; + tensor hidden_states_49_strides_0 = const()[name = string("hidden_states_49_strides_0"), val = tensor([1, 1])]; + tensor hidden_states_49_pad_0 = const()[name = string("hidden_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor hidden_states_49_dilations_0 = const()[name = string("hidden_states_49_dilations_0"), val = tensor([1, 1])]; + int32 hidden_states_49_groups_0 = const()[name = string("hidden_states_49_groups_0"), val = int32(1)]; + tensor layers_4_mlp_down_proj_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(77658112))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80803904))))[name = string("layers_4_mlp_down_proj_weight_to_fp16_palettized")]; + tensor hidden_states_49_cast_fp16 = conv(dilations = hidden_states_49_dilations_0, groups = hidden_states_49_groups_0, pad = hidden_states_49_pad_0, pad_type = hidden_states_49_pad_type_0, strides = hidden_states_49_strides_0, weight = layers_4_mlp_down_proj_weight_to_fp16_palettized, x = input_39_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; + tensor inputs_cast_fp16 = add(x = inputs_39_cast_fp16, y = hidden_states_49_cast_fp16)[name = string("inputs_cast_fp16")]; + tensor inputs_sq_cast_fp16 = mul(x = inputs_cast_fp16, y = inputs_cast_fp16)[name = string("inputs_sq_cast_fp16")]; + tensor variance_axes_0 = const()[name = string("variance_axes_0"), val = tensor([1])]; + bool variance_keep_dims_0 = const()[name = string("variance_keep_dims_0"), val = bool(true)]; + tensor variance_cast_fp16 = reduce_mean(axes = variance_axes_0, keep_dims = variance_keep_dims_0, x = inputs_sq_cast_fp16)[name = string("variance_cast_fp16")]; + fp16 var_2053_to_fp16 = const()[name = string("op_2053_to_fp16"), val = fp16(0x1.1p-20)]; + tensor var_2054_cast_fp16 = add(x = variance_cast_fp16, y = var_2053_to_fp16)[name = string("op_2054_cast_fp16")]; + fp32 var_2055_epsilon_0 = const()[name = string("op_2055_epsilon_0"), val = fp32(0x1.197998p-40)]; + tensor var_2055_cast_fp16 = rsqrt(epsilon = var_2055_epsilon_0, x = var_2054_cast_fp16)[name = string("op_2055_cast_fp16")]; + tensor hidden_states_cast_fp16 = mul(x = inputs_cast_fp16, y = var_2055_cast_fp16)[name = string("hidden_states_cast_fp16")]; + tensor w_to_fp16 = const()[name = string("w_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80804480)))]; + tensor input_cast_fp16 = mul(x = w_to_fp16, y = hidden_states_cast_fp16)[name = string("input_cast_fp16")]; + string logits_1_pad_type_0 = const()[name = string("logits_1_pad_type_0"), val = string("valid")]; + tensor logits_1_strides_0 = const()[name = string("logits_1_strides_0"), val = tensor([1, 1])]; + tensor logits_1_pad_0 = const()[name = string("logits_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_1_dilations_0 = const()[name = string("logits_1_dilations_0"), val = tensor([1, 1])]; + int32 logits_1_groups_0 = const()[name = string("logits_1_groups_0"), val = int32(1)]; + tensor lm_heads_0_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80806592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82903808))))[name = string("lm_heads_0_weight_to_fp16_palettized")]; + tensor logits_1_cast_fp16 = conv(dilations = logits_1_dilations_0, groups = logits_1_groups_0, pad = logits_1_pad_0, pad_type = logits_1_pad_type_0, strides = logits_1_strides_0, weight = lm_heads_0_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_1_cast_fp16")]; + tensor var_2072_axes_0 = const()[name = string("op_2072_axes_0"), val = tensor([3])]; + tensor var_2072_cast_fp16 = squeeze(axes = var_2072_axes_0, x = logits_1_cast_fp16)[name = string("op_2072_cast_fp16")]; + string logits_3_pad_type_0 = const()[name = string("logits_3_pad_type_0"), val = string("valid")]; + tensor logits_3_strides_0 = const()[name = string("logits_3_strides_0"), val = tensor([1, 1])]; + tensor logits_3_pad_0 = const()[name = string("logits_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_3_dilations_0 = const()[name = string("logits_3_dilations_0"), val = tensor([1, 1])]; + int32 logits_3_groups_0 = const()[name = string("logits_3_groups_0"), val = int32(1)]; + tensor lm_heads_1_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82904384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85001600))))[name = string("lm_heads_1_weight_to_fp16_palettized")]; + tensor logits_3_cast_fp16 = conv(dilations = logits_3_dilations_0, groups = logits_3_groups_0, pad = logits_3_pad_0, pad_type = logits_3_pad_type_0, strides = logits_3_strides_0, weight = lm_heads_1_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_3_cast_fp16")]; + tensor var_2088_axes_0 = const()[name = string("op_2088_axes_0"), val = tensor([3])]; + tensor var_2088_cast_fp16 = squeeze(axes = var_2088_axes_0, x = logits_3_cast_fp16)[name = string("op_2088_cast_fp16")]; + string logits_5_pad_type_0 = const()[name = string("logits_5_pad_type_0"), val = string("valid")]; + tensor logits_5_strides_0 = const()[name = string("logits_5_strides_0"), val = tensor([1, 1])]; + tensor logits_5_pad_0 = const()[name = string("logits_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_5_dilations_0 = const()[name = string("logits_5_dilations_0"), val = tensor([1, 1])]; + int32 logits_5_groups_0 = const()[name = string("logits_5_groups_0"), val = int32(1)]; + tensor lm_heads_2_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85002176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87099392))))[name = string("lm_heads_2_weight_to_fp16_palettized")]; + tensor logits_5_cast_fp16 = conv(dilations = logits_5_dilations_0, groups = logits_5_groups_0, pad = logits_5_pad_0, pad_type = logits_5_pad_type_0, strides = logits_5_strides_0, weight = lm_heads_2_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_5_cast_fp16")]; + tensor var_2104_axes_0 = const()[name = string("op_2104_axes_0"), val = tensor([3])]; + tensor var_2104_cast_fp16 = squeeze(axes = var_2104_axes_0, x = logits_5_cast_fp16)[name = string("op_2104_cast_fp16")]; + string logits_7_pad_type_0 = const()[name = string("logits_7_pad_type_0"), val = string("valid")]; + tensor logits_7_strides_0 = const()[name = string("logits_7_strides_0"), val = tensor([1, 1])]; + tensor logits_7_pad_0 = const()[name = string("logits_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_7_dilations_0 = const()[name = string("logits_7_dilations_0"), val = tensor([1, 1])]; + int32 logits_7_groups_0 = const()[name = string("logits_7_groups_0"), val = int32(1)]; + tensor lm_heads_3_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87099968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89197184))))[name = string("lm_heads_3_weight_to_fp16_palettized")]; + tensor logits_7_cast_fp16 = conv(dilations = logits_7_dilations_0, groups = logits_7_groups_0, pad = logits_7_pad_0, pad_type = logits_7_pad_type_0, strides = logits_7_strides_0, weight = lm_heads_3_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_7_cast_fp16")]; + tensor var_2120_axes_0 = const()[name = string("op_2120_axes_0"), val = tensor([3])]; + tensor var_2120_cast_fp16 = squeeze(axes = var_2120_axes_0, x = logits_7_cast_fp16)[name = string("op_2120_cast_fp16")]; + string logits_9_pad_type_0 = const()[name = string("logits_9_pad_type_0"), val = string("valid")]; + tensor logits_9_strides_0 = const()[name = string("logits_9_strides_0"), val = tensor([1, 1])]; + tensor logits_9_pad_0 = const()[name = string("logits_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_9_dilations_0 = const()[name = string("logits_9_dilations_0"), val = tensor([1, 1])]; + int32 logits_9_groups_0 = const()[name = string("logits_9_groups_0"), val = int32(1)]; + tensor lm_heads_4_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(89197760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91294976))))[name = string("lm_heads_4_weight_to_fp16_palettized")]; + tensor logits_9_cast_fp16 = conv(dilations = logits_9_dilations_0, groups = logits_9_groups_0, pad = logits_9_pad_0, pad_type = logits_9_pad_type_0, strides = logits_9_strides_0, weight = lm_heads_4_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_9_cast_fp16")]; + tensor var_2136_axes_0 = const()[name = string("op_2136_axes_0"), val = tensor([3])]; + tensor var_2136_cast_fp16 = squeeze(axes = var_2136_axes_0, x = logits_9_cast_fp16)[name = string("op_2136_cast_fp16")]; + string logits_11_pad_type_0 = const()[name = string("logits_11_pad_type_0"), val = string("valid")]; + tensor logits_11_strides_0 = const()[name = string("logits_11_strides_0"), val = tensor([1, 1])]; + tensor logits_11_pad_0 = const()[name = string("logits_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_11_dilations_0 = const()[name = string("logits_11_dilations_0"), val = tensor([1, 1])]; + int32 logits_11_groups_0 = const()[name = string("logits_11_groups_0"), val = int32(1)]; + tensor lm_heads_5_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91295552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93392768))))[name = string("lm_heads_5_weight_to_fp16_palettized")]; + tensor logits_11_cast_fp16 = conv(dilations = logits_11_dilations_0, groups = logits_11_groups_0, pad = logits_11_pad_0, pad_type = logits_11_pad_type_0, strides = logits_11_strides_0, weight = lm_heads_5_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_11_cast_fp16")]; + tensor var_2152_axes_0 = const()[name = string("op_2152_axes_0"), val = tensor([3])]; + tensor var_2152_cast_fp16 = squeeze(axes = var_2152_axes_0, x = logits_11_cast_fp16)[name = string("op_2152_cast_fp16")]; + string logits_13_pad_type_0 = const()[name = string("logits_13_pad_type_0"), val = string("valid")]; + tensor logits_13_strides_0 = const()[name = string("logits_13_strides_0"), val = tensor([1, 1])]; + tensor logits_13_pad_0 = const()[name = string("logits_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_13_dilations_0 = const()[name = string("logits_13_dilations_0"), val = tensor([1, 1])]; + int32 logits_13_groups_0 = const()[name = string("logits_13_groups_0"), val = int32(1)]; + tensor lm_heads_6_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93393344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95490560))))[name = string("lm_heads_6_weight_to_fp16_palettized")]; + tensor logits_13_cast_fp16 = conv(dilations = logits_13_dilations_0, groups = logits_13_groups_0, pad = logits_13_pad_0, pad_type = logits_13_pad_type_0, strides = logits_13_strides_0, weight = lm_heads_6_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_13_cast_fp16")]; + tensor var_2168_axes_0 = const()[name = string("op_2168_axes_0"), val = tensor([3])]; + tensor var_2168_cast_fp16 = squeeze(axes = var_2168_axes_0, x = logits_13_cast_fp16)[name = string("op_2168_cast_fp16")]; + string logits_15_pad_type_0 = const()[name = string("logits_15_pad_type_0"), val = string("valid")]; + tensor logits_15_strides_0 = const()[name = string("logits_15_strides_0"), val = tensor([1, 1])]; + tensor logits_15_pad_0 = const()[name = string("logits_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_15_dilations_0 = const()[name = string("logits_15_dilations_0"), val = tensor([1, 1])]; + int32 logits_15_groups_0 = const()[name = string("logits_15_groups_0"), val = int32(1)]; + tensor lm_heads_7_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(95491136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(97588352))))[name = string("lm_heads_7_weight_to_fp16_palettized")]; + tensor logits_15_cast_fp16 = conv(dilations = logits_15_dilations_0, groups = logits_15_groups_0, pad = logits_15_pad_0, pad_type = logits_15_pad_type_0, strides = logits_15_strides_0, weight = lm_heads_7_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_15_cast_fp16")]; + tensor var_2184_axes_0 = const()[name = string("op_2184_axes_0"), val = tensor([3])]; + tensor var_2184_cast_fp16 = squeeze(axes = var_2184_axes_0, x = logits_15_cast_fp16)[name = string("op_2184_cast_fp16")]; + string logits_17_pad_type_0 = const()[name = string("logits_17_pad_type_0"), val = string("valid")]; + tensor logits_17_strides_0 = const()[name = string("logits_17_strides_0"), val = tensor([1, 1])]; + tensor logits_17_pad_0 = const()[name = string("logits_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_17_dilations_0 = const()[name = string("logits_17_dilations_0"), val = tensor([1, 1])]; + int32 logits_17_groups_0 = const()[name = string("logits_17_groups_0"), val = int32(1)]; + tensor lm_heads_8_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(97588928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(99686144))))[name = string("lm_heads_8_weight_to_fp16_palettized")]; + tensor logits_17_cast_fp16 = conv(dilations = logits_17_dilations_0, groups = logits_17_groups_0, pad = logits_17_pad_0, pad_type = logits_17_pad_type_0, strides = logits_17_strides_0, weight = lm_heads_8_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_17_cast_fp16")]; + tensor var_2200_axes_0 = const()[name = string("op_2200_axes_0"), val = tensor([3])]; + tensor var_2200_cast_fp16 = squeeze(axes = var_2200_axes_0, x = logits_17_cast_fp16)[name = string("op_2200_cast_fp16")]; + string logits_19_pad_type_0 = const()[name = string("logits_19_pad_type_0"), val = string("valid")]; + tensor logits_19_strides_0 = const()[name = string("logits_19_strides_0"), val = tensor([1, 1])]; + tensor logits_19_pad_0 = const()[name = string("logits_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_19_dilations_0 = const()[name = string("logits_19_dilations_0"), val = tensor([1, 1])]; + int32 logits_19_groups_0 = const()[name = string("logits_19_groups_0"), val = int32(1)]; + tensor lm_heads_9_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(99686720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101783936))))[name = string("lm_heads_9_weight_to_fp16_palettized")]; + tensor logits_19_cast_fp16 = conv(dilations = logits_19_dilations_0, groups = logits_19_groups_0, pad = logits_19_pad_0, pad_type = logits_19_pad_type_0, strides = logits_19_strides_0, weight = lm_heads_9_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_19_cast_fp16")]; + tensor var_2216_axes_0 = const()[name = string("op_2216_axes_0"), val = tensor([3])]; + tensor var_2216_cast_fp16 = squeeze(axes = var_2216_axes_0, x = logits_19_cast_fp16)[name = string("op_2216_cast_fp16")]; + string logits_21_pad_type_0 = const()[name = string("logits_21_pad_type_0"), val = string("valid")]; + tensor logits_21_strides_0 = const()[name = string("logits_21_strides_0"), val = tensor([1, 1])]; + tensor logits_21_pad_0 = const()[name = string("logits_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_21_dilations_0 = const()[name = string("logits_21_dilations_0"), val = tensor([1, 1])]; + int32 logits_21_groups_0 = const()[name = string("logits_21_groups_0"), val = int32(1)]; + tensor lm_heads_10_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101784512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103881728))))[name = string("lm_heads_10_weight_to_fp16_palettized")]; + tensor logits_21_cast_fp16 = conv(dilations = logits_21_dilations_0, groups = logits_21_groups_0, pad = logits_21_pad_0, pad_type = logits_21_pad_type_0, strides = logits_21_strides_0, weight = lm_heads_10_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_21_cast_fp16")]; + tensor var_2232_axes_0 = const()[name = string("op_2232_axes_0"), val = tensor([3])]; + tensor var_2232_cast_fp16 = squeeze(axes = var_2232_axes_0, x = logits_21_cast_fp16)[name = string("op_2232_cast_fp16")]; + string logits_23_pad_type_0 = const()[name = string("logits_23_pad_type_0"), val = string("valid")]; + tensor logits_23_strides_0 = const()[name = string("logits_23_strides_0"), val = tensor([1, 1])]; + tensor logits_23_pad_0 = const()[name = string("logits_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_23_dilations_0 = const()[name = string("logits_23_dilations_0"), val = tensor([1, 1])]; + int32 logits_23_groups_0 = const()[name = string("logits_23_groups_0"), val = int32(1)]; + tensor lm_heads_11_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103882304))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105979520))))[name = string("lm_heads_11_weight_to_fp16_palettized")]; + tensor logits_23_cast_fp16 = conv(dilations = logits_23_dilations_0, groups = logits_23_groups_0, pad = logits_23_pad_0, pad_type = logits_23_pad_type_0, strides = logits_23_strides_0, weight = lm_heads_11_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_23_cast_fp16")]; + tensor var_2248_axes_0 = const()[name = string("op_2248_axes_0"), val = tensor([3])]; + tensor var_2248_cast_fp16 = squeeze(axes = var_2248_axes_0, x = logits_23_cast_fp16)[name = string("op_2248_cast_fp16")]; + string logits_25_pad_type_0 = const()[name = string("logits_25_pad_type_0"), val = string("valid")]; + tensor logits_25_strides_0 = const()[name = string("logits_25_strides_0"), val = tensor([1, 1])]; + tensor logits_25_pad_0 = const()[name = string("logits_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_25_dilations_0 = const()[name = string("logits_25_dilations_0"), val = tensor([1, 1])]; + int32 logits_25_groups_0 = const()[name = string("logits_25_groups_0"), val = int32(1)]; + tensor lm_heads_12_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105980096))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108077312))))[name = string("lm_heads_12_weight_to_fp16_palettized")]; + tensor logits_25_cast_fp16 = conv(dilations = logits_25_dilations_0, groups = logits_25_groups_0, pad = logits_25_pad_0, pad_type = logits_25_pad_type_0, strides = logits_25_strides_0, weight = lm_heads_12_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_25_cast_fp16")]; + tensor var_2264_axes_0 = const()[name = string("op_2264_axes_0"), val = tensor([3])]; + tensor var_2264_cast_fp16 = squeeze(axes = var_2264_axes_0, x = logits_25_cast_fp16)[name = string("op_2264_cast_fp16")]; + string logits_27_pad_type_0 = const()[name = string("logits_27_pad_type_0"), val = string("valid")]; + tensor logits_27_strides_0 = const()[name = string("logits_27_strides_0"), val = tensor([1, 1])]; + tensor logits_27_pad_0 = const()[name = string("logits_27_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_27_dilations_0 = const()[name = string("logits_27_dilations_0"), val = tensor([1, 1])]; + int32 logits_27_groups_0 = const()[name = string("logits_27_groups_0"), val = int32(1)]; + tensor lm_heads_13_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108077888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110175104))))[name = string("lm_heads_13_weight_to_fp16_palettized")]; + tensor logits_27_cast_fp16 = conv(dilations = logits_27_dilations_0, groups = logits_27_groups_0, pad = logits_27_pad_0, pad_type = logits_27_pad_type_0, strides = logits_27_strides_0, weight = lm_heads_13_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_27_cast_fp16")]; + tensor var_2280_axes_0 = const()[name = string("op_2280_axes_0"), val = tensor([3])]; + tensor var_2280_cast_fp16 = squeeze(axes = var_2280_axes_0, x = logits_27_cast_fp16)[name = string("op_2280_cast_fp16")]; + string logits_29_pad_type_0 = const()[name = string("logits_29_pad_type_0"), val = string("valid")]; + tensor logits_29_strides_0 = const()[name = string("logits_29_strides_0"), val = tensor([1, 1])]; + tensor logits_29_pad_0 = const()[name = string("logits_29_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor logits_29_dilations_0 = const()[name = string("logits_29_dilations_0"), val = tensor([1, 1])]; + int32 logits_29_groups_0 = const()[name = string("logits_29_groups_0"), val = int32(1)]; + tensor lm_heads_14_weight_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(110175680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112272896))))[name = string("lm_heads_14_weight_to_fp16_palettized")]; + tensor logits_29_cast_fp16 = conv(dilations = logits_29_dilations_0, groups = logits_29_groups_0, pad = logits_29_pad_0, pad_type = logits_29_pad_type_0, strides = logits_29_strides_0, weight = lm_heads_14_weight_to_fp16_palettized, x = input_cast_fp16)[name = string("logits_29_cast_fp16")]; + tensor var_2296_axes_0 = const()[name = string("op_2296_axes_0"), val = tensor([3])]; + tensor var_2296_cast_fp16 = squeeze(axes = var_2296_axes_0, x = logits_29_cast_fp16)[name = string("op_2296_cast_fp16")]; + bool var_2302_interleave_0 = const()[name = string("op_2302_interleave_0"), val = bool(false)]; + int32 const_119 = const()[name = string("const_119"), val = int32(2)]; + tensor var_2302_cast_fp16 = concat(axis = const_119, interleave = var_2302_interleave_0, values = (var_2072_cast_fp16, var_2088_cast_fp16, var_2104_cast_fp16, var_2120_cast_fp16, var_2136_cast_fp16, var_2152_cast_fp16, var_2168_cast_fp16, var_2184_cast_fp16, var_2200_cast_fp16, var_2216_cast_fp16, var_2232_cast_fp16, var_2248_cast_fp16, var_2264_cast_fp16, var_2280_cast_fp16, var_2296_cast_fp16))[name = string("op_2302_cast_fp16")]; + int32 var_2304 = const()[name = string("op_2304"), val = int32(1)]; + bool var_2305_interleave_0 = const()[name = string("op_2305_interleave_0"), val = bool(false)]; + tensor key_cache_updates = concat(axis = var_2304, interleave = var_2305_interleave_0, values = (current_key_3_cast_fp16, current_key_7_cast_fp16, current_key_11_cast_fp16, current_key_15_cast_fp16, current_key_cast_fp16))[name = string("op_2305_cast_fp16")]; + int32 var_2307 = const()[name = string("op_2307"), val = int32(1)]; + bool var_2308_interleave_0 = const()[name = string("op_2308_interleave_0"), val = bool(false)]; + tensor value_cache_updates = concat(axis = var_2307, interleave = var_2308_interleave_0, values = (current_value_1_cast_fp16, current_value_3_cast_fp16, current_value_5_cast_fp16, current_value_7_cast_fp16, current_value_cast_fp16))[name = string("op_2308_cast_fp16")]; + tensor transpose_0_perm_0 = const()[name = string("transpose_0_perm_0"), val = tensor([0, 2, 1])]; + tensor all_logits = transpose(perm = transpose_0_perm_0, x = var_2302_cast_fp16)[name = string("transpose_0")]; + } -> (all_logits, key_cache_updates, value_cache_updates); +} \ No newline at end of file diff --git a/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/weights/weight.bin b/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/weights/weight.bin new file mode 100644 index 0000000000000000000000000000000000000000..b216958e44bac86c1db087980bc7d40406cca34c --- /dev/null +++ b/qwen3_tts/multi_code_decoder/12hz-1.7b-customvoice/W8A16-multifunction/MultiCodeDecoder.mlmodelc/weights/weight.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:498cd0cbc7ad298e961325c5b159e89195f5c56dcb4effbcd88c2f3c8509de14 +size 175204768