diff --git "a/leaf.mlmodelc/model.mil" "b/leaf.mlmodelc/model.mil" new file mode 100644--- /dev/null +++ "b/leaf.mlmodelc/model.mil" @@ -0,0 +1,488 @@ +program(1.0) +[buildInfo = dict, tensor>({{"coremlc-component-MIL", "3510.2.1"}, {"coremlc-version", "3500.32.1"}, {"coremltools-component-torch", "2.14.1"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "9.0"}})] +{ + func main(tensor input_ids) [FlexibleShapeInformation = tuple, dict, tensor>>, tuple, dict, dict, tensor>>>>((("DefaultShapes", {{"input_ids", [1, 16]}}), ("EnumeratedShapes", {{"input_ids_1_1_1_1_128_", {{"input_ids", [1, 128]}}}, {"input_ids_1_1_1_1_16_", {{"input_ids", [1, 16]}}}, {"input_ids_1_1_1_1_256_", {{"input_ids", [1, 256]}}}, {"input_ids_1_1_1_1_32_", {{"input_ids", [1, 32]}}}, {"input_ids_1_1_1_1_512_", {{"input_ids", [1, 512]}}}, {"input_ids_1_1_1_1_64_", {{"input_ids", [1, 64]}}}})))] { + tensor m_embeddings_position_ids = const()[name = tensor("m_embeddings_position_ids"), val = tensor([[0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154, 155, 156, 157, 158, 159, 160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 172, 173, 174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192, 193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 209, 210, 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226, 227, 228, 229, 230, 231, 232, 233, 234, 235, 236, 237, 238, 239, 240, 241, 242, 243, 244, 245, 246, 247, 248, 249, 250, 251, 252, 253, 254, 255, 256, 257, 258, 259, 260, 261, 262, 263, 264, 265, 266, 267, 268, 269, 270, 271, 272, 273, 274, 275, 276, 277, 278, 279, 280, 281, 282, 283, 284, 285, 286, 287, 288, 289, 290, 291, 292, 293, 294, 295, 296, 297, 298, 299, 300, 301, 302, 303, 304, 305, 306, 307, 308, 309, 310, 311, 312, 313, 314, 315, 316, 317, 318, 319, 320, 321, 322, 323, 324, 325, 326, 327, 328, 329, 330, 331, 332, 333, 334, 335, 336, 337, 338, 339, 340, 341, 342, 343, 344, 345, 346, 347, 348, 349, 350, 351, 352, 353, 354, 355, 356, 357, 358, 359, 360, 361, 362, 363, 364, 365, 366, 367, 368, 369, 370, 371, 372, 373, 374, 375, 376, 377, 378, 379, 380, 381, 382, 383, 384, 385, 386, 387, 388, 389, 390, 391, 392, 393, 394, 395, 396, 397, 398, 399, 400, 401, 402, 403, 404, 405, 406, 407, 408, 409, 410, 411, 412, 413, 414, 415, 416, 417, 418, 419, 420, 421, 422, 423, 424, 425, 426, 427, 428, 429, 430, 431, 432, 433, 434, 435, 436, 437, 438, 439, 440, 441, 442, 443, 444, 445, 446, 447, 448, 449, 450, 451, 452, 453, 454, 455, 456, 457, 458, 459, 460, 461, 462, 463, 464, 465, 466, 467, 468, 469, 470, 471, 472, 473, 474, 475, 476, 477, 478, 479, 480, 481, 482, 483, 484, 485, 486, 487, 488, 489, 490, 491, 492, 493, 494, 495, 496, 497, 498, 499, 500, 501, 502, 503, 504, 505, 506, 507, 508, 509, 510, 511]])]; + tensor var_5 = const()[name = tensor("op_5"), val = tensor(0)]; + tensor var_6 = not_equal(x = input_ids, y = var_5)[name = tensor("op_6")]; + tensor mask_1_dtype_0 = const()[name = tensor("mask_1_dtype_0"), val = tensor("int32")]; + tensor input_1 = sub(x = input_ids, y = input_ids)[name = tensor("sub_0")]; + tensor var_38 = const()[name = tensor("op_38"), val = tensor(1)]; + tensor var_41_shape = shape(x = input_ids)[name = tensor("op_41_shape")]; + tensor gather_0_axis_0 = const()[name = tensor("gather_0_axis_0"), val = tensor(0)]; + tensor gather_0_batch_dims_0 = const()[name = tensor("gather_0_batch_dims_0"), val = tensor(0)]; + tensor gather_0_validate_indices_0 = const()[name = tensor("gather_0_validate_indices_0"), val = tensor(false)]; + tensor var_41_shape_to_int16_dtype_0 = const()[name = tensor("op_41_shape_to_int16_dtype_0"), val = tensor("int16")]; + tensor gather_0_indices_0_to_uint16 = const()[name = tensor("gather_0_indices_0_to_uint16"), val = tensor(1)]; + tensor var_41_shape_to_int16 = cast(dtype = var_41_shape_to_int16_dtype_0, x = var_41_shape)[name = tensor("cast_47")]; + tensor gather_0_cast_uint16 = gather(axis = gather_0_axis_0, batch_dims = gather_0_batch_dims_0, indices = gather_0_indices_0_to_uint16, validate_indices = gather_0_validate_indices_0, x = var_41_shape_to_int16)[name = tensor("gather_0_cast_uint16")]; + tensor gather_0_cast_uint16_to_int32_dtype_0 = const()[name = tensor("gather_0_cast_uint16_to_int32_dtype_0"), val = tensor("int32")]; + tensor concat_0_values0_0 = const()[name = tensor("concat_0_values0_0"), val = tensor(1)]; + tensor concat_0_axis_0 = const()[name = tensor("concat_0_axis_0"), val = tensor(0)]; + tensor concat_0_interleave_0 = const()[name = tensor("concat_0_interleave_0"), val = tensor(false)]; + tensor gather_0_cast_uint16_to_int32 = cast(dtype = gather_0_cast_uint16_to_int32_dtype_0, x = gather_0_cast_uint16)[name = tensor("cast_46")]; + tensor concat_0 = concat(axis = concat_0_axis_0, interleave = concat_0_interleave_0, values = (concat_0_values0_0, gather_0_cast_uint16_to_int32))[name = tensor("concat_0")]; + tensor input_3_begin_0 = const()[name = tensor("input_3_begin_0"), val = tensor([0, 0])]; + tensor input_3_end_mask_0 = const()[name = tensor("input_3_end_mask_0"), val = tensor([true, false])]; + tensor input_3 = slice_by_index(begin = input_3_begin_0, end = concat_0, end_mask = input_3_end_mask_0, x = m_embeddings_position_ids)[name = tensor("input_3")]; + tensor inputs_embeds_axis_0 = const()[name = tensor("inputs_embeds_axis_0"), val = tensor(0)]; + tensor inputs_embeds_batch_dims_0 = const()[name = tensor("inputs_embeds_batch_dims_0"), val = tensor(0)]; + tensor inputs_embeds_validate_indices_0 = const()[name = tensor("inputs_embeds_validate_indices_0"), val = tensor(false)]; + tensor m_embeddings_word_embeddings_weight_to_fp16 = const()[name = tensor("m_embeddings_word_embeddings_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(64)))]; + tensor input_ids_to_uint16_dtype_0 = const()[name = tensor("input_ids_to_uint16_dtype_0"), val = tensor("uint16")]; + tensor input_ids_to_uint16 = cast(dtype = input_ids_to_uint16_dtype_0, x = input_ids)[name = tensor("cast_45")]; + tensor inputs_embeds_cast_fp16_cast_uint16 = gather(axis = inputs_embeds_axis_0, batch_dims = inputs_embeds_batch_dims_0, indices = input_ids_to_uint16, validate_indices = inputs_embeds_validate_indices_0, x = m_embeddings_word_embeddings_weight_to_fp16)[name = tensor("inputs_embeds_cast_fp16_cast_uint16")]; + tensor token_type_embeddings_1_axis_0 = const()[name = tensor("token_type_embeddings_1_axis_0"), val = tensor(0)]; + tensor token_type_embeddings_1_batch_dims_0 = const()[name = tensor("token_type_embeddings_1_batch_dims_0"), val = tensor(0)]; + tensor token_type_embeddings_1_validate_indices_0 = const()[name = tensor("token_type_embeddings_1_validate_indices_0"), val = tensor(false)]; + tensor m_embeddings_token_type_embeddings_weight_to_fp16 = const()[name = tensor("m_embeddings_token_type_embeddings_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(23441024)))]; + tensor input_1_to_uint16_dtype_0 = const()[name = tensor("input_1_to_uint16_dtype_0"), val = tensor("uint16")]; + tensor input_1_to_uint16 = cast(dtype = input_1_to_uint16_dtype_0, x = input_1)[name = tensor("cast_44")]; + tensor token_type_embeddings_1_cast_fp16_cast_uint16 = gather(axis = token_type_embeddings_1_axis_0, batch_dims = token_type_embeddings_1_batch_dims_0, indices = input_1_to_uint16, validate_indices = token_type_embeddings_1_validate_indices_0, x = m_embeddings_token_type_embeddings_weight_to_fp16)[name = tensor("token_type_embeddings_1_cast_fp16_cast_uint16")]; + tensor embeddings_1_cast_fp16 = add(x = inputs_embeds_cast_fp16_cast_uint16, y = token_type_embeddings_1_cast_fp16_cast_uint16)[name = tensor("embeddings_1_cast_fp16")]; + tensor position_embeddings_1_axis_0 = const()[name = tensor("position_embeddings_1_axis_0"), val = tensor(0)]; + tensor position_embeddings_1_batch_dims_0 = const()[name = tensor("position_embeddings_1_batch_dims_0"), val = tensor(0)]; + tensor position_embeddings_1_validate_indices_0 = const()[name = tensor("position_embeddings_1_validate_indices_0"), val = tensor(false)]; + tensor m_embeddings_position_embeddings_weight_to_fp16 = const()[name = tensor("m_embeddings_position_embeddings_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(23442624)))]; + tensor input_3_to_uint16_dtype_0 = const()[name = tensor("input_3_to_uint16_dtype_0"), val = tensor("uint16")]; + tensor input_3_to_uint16 = cast(dtype = input_3_to_uint16_dtype_0, x = input_3)[name = tensor("cast_43")]; + tensor position_embeddings_1_cast_fp16_cast_uint16 = gather(axis = position_embeddings_1_axis_0, batch_dims = position_embeddings_1_batch_dims_0, indices = input_3_to_uint16, validate_indices = position_embeddings_1_validate_indices_0, x = m_embeddings_position_embeddings_weight_to_fp16)[name = tensor("position_embeddings_1_cast_fp16_cast_uint16")]; + tensor input_5_cast_fp16 = add(x = embeddings_1_cast_fp16, y = position_embeddings_1_cast_fp16_cast_uint16)[name = tensor("input_5_cast_fp16")]; + tensor input_7_axes_0 = const()[name = tensor("input_7_axes_0"), val = tensor([-1])]; + tensor m_embeddings_LayerNorm_weight_to_fp16 = const()[name = tensor("m_embeddings_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(23835904)))]; + tensor m_embeddings_LayerNorm_bias_to_fp16 = const()[name = tensor("m_embeddings_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(23836736)))]; + tensor var_35_to_fp16 = const()[name = tensor("op_35_to_fp16"), val = tensor(0x1p-24)]; + tensor input_7_cast_fp16 = layer_norm(axes = input_7_axes_0, beta = m_embeddings_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_embeddings_LayerNorm_weight_to_fp16, x = input_5_cast_fp16)[name = tensor("input_7_cast_fp16")]; + tensor mask_1 = cast(dtype = mask_1_dtype_0, x = var_6)[name = tensor("cast_48")]; + tensor var_67_shape = shape(x = mask_1)[name = tensor("op_67_shape")]; + tensor gather_2 = const()[name = tensor("gather_2"), val = tensor(1)]; + tensor gather_3_axis_0 = const()[name = tensor("gather_3_axis_0"), val = tensor(0)]; + tensor gather_3_batch_dims_0 = const()[name = tensor("gather_3_batch_dims_0"), val = tensor(0)]; + tensor gather_3_validate_indices_0 = const()[name = tensor("gather_3_validate_indices_0"), val = tensor(false)]; + tensor var_67_shape_to_uint16_dtype_0 = const()[name = tensor("op_67_shape_to_uint16_dtype_0"), val = tensor("uint16")]; + tensor gather_3_indices_0_to_uint16 = const()[name = tensor("gather_3_indices_0_to_uint16"), val = tensor(1)]; + tensor var_67_shape_to_uint16 = cast(dtype = var_67_shape_to_uint16_dtype_0, x = var_67_shape)[name = tensor("cast_42")]; + tensor gather_3_cast_uint16 = gather(axis = gather_3_axis_0, batch_dims = gather_3_batch_dims_0, indices = gather_3_indices_0_to_uint16, validate_indices = gather_3_validate_indices_0, x = var_67_shape_to_uint16)[name = tensor("gather_3_cast_uint16")]; + tensor gather_3_cast_uint16_to_int32_dtype_0 = const()[name = tensor("gather_3_cast_uint16_to_int32_dtype_0"), val = tensor("int32")]; + tensor var_70_axes_0 = const()[name = tensor("op_70_axes_0"), val = tensor([1])]; + tensor var_70 = expand_dims(axes = var_70_axes_0, x = mask_1)[name = tensor("op_70")]; + tensor var_71_axes_0 = const()[name = tensor("op_71_axes_0"), val = tensor([2])]; + tensor var_71 = expand_dims(axes = var_71_axes_0, x = var_70)[name = tensor("op_71")]; + tensor concat_1_axis_0 = const()[name = tensor("concat_1_axis_0"), val = tensor(0)]; + tensor concat_1_interleave_0 = const()[name = tensor("concat_1_interleave_0"), val = tensor(false)]; + tensor gather_3_cast_uint16_to_int32 = cast(dtype = gather_3_cast_uint16_to_int32_dtype_0, x = gather_3_cast_uint16)[name = tensor("cast_41")]; + tensor concat_1 = concat(axis = concat_1_axis_0, interleave = concat_1_interleave_0, values = (gather_2, var_38, gather_0_cast_uint16_to_int32, gather_3_cast_uint16_to_int32))[name = tensor("concat_1")]; + tensor shape_1 = shape(x = var_71)[name = tensor("shape_1")]; + tensor equal_0_y_0 = const()[name = tensor("equal_0_y_0"), val = tensor(-1)]; + tensor equal_0 = equal(x = concat_1, y = equal_0_y_0)[name = tensor("equal_0")]; + tensor select_0 = select(a = shape_1, b = concat_1, cond = equal_0)[name = tensor("select_0")]; + tensor real_div_0 = real_div(x = select_0, y = shape_1)[name = tensor("real_div_0")]; + tensor var_74 = tile(reps = real_div_0, x = var_71)[name = tensor("op_74")]; + tensor const_0_to_fp16 = const()[name = tensor("const_0_to_fp16"), val = tensor(0x1p+0)]; + tensor expanded_mask_to_fp16_dtype_0 = const()[name = tensor("expanded_mask_to_fp16_dtype_0"), val = tensor("fp16")]; + tensor var_74_to_fp16 = cast(dtype = expanded_mask_to_fp16_dtype_0, x = var_74)[name = tensor("cast_40")]; + tensor inverted_mask_cast_fp16 = sub(x = const_0_to_fp16, y = var_74_to_fp16)[name = tensor("inverted_mask_cast_fp16")]; + tensor var_79_dtype_0 = const()[name = tensor("op_79_dtype_0"), val = tensor("bool")]; + tensor var_22_to_fp16 = const()[name = tensor("op_22_to_fp16"), val = tensor(-inf)]; + tensor inverted_mask_cast_fp16_to_bool = cast(dtype = var_79_dtype_0, x = inverted_mask_cast_fp16)[name = tensor("cast_39")]; + tensor attention_mask_cast_fp16 = select(a = var_22_to_fp16, b = inverted_mask_cast_fp16, cond = inverted_mask_cast_fp16_to_bool)[name = tensor("attention_mask_cast_fp16")]; + tensor m_encoder_layer_0_attention_self_query_weight_to_fp16 = const()[name = tensor("m_encoder_layer_0_attention_self_query_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(23837568)))]; + tensor m_encoder_layer_0_attention_self_query_bias_to_fp16 = const()[name = tensor("m_encoder_layer_0_attention_self_query_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(24132544)))]; + tensor linear_0_cast_fp16 = linear(bias = m_encoder_layer_0_attention_self_query_bias_to_fp16, weight = m_encoder_layer_0_attention_self_query_weight_to_fp16, x = input_7_cast_fp16)[name = tensor("linear_0_cast_fp16")]; + tensor var_106 = const()[name = tensor("op_106"), val = tensor([1, -1, 12, 32])]; + tensor var_107_cast_fp16 = reshape(shape = var_106, x = linear_0_cast_fp16)[name = tensor("op_107_cast_fp16")]; + tensor m_encoder_layer_0_attention_self_key_weight_to_fp16 = const()[name = tensor("m_encoder_layer_0_attention_self_key_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(24133376)))]; + tensor m_encoder_layer_0_attention_self_key_bias_to_fp16 = const()[name = tensor("m_encoder_layer_0_attention_self_key_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(24428352)))]; + tensor linear_1_cast_fp16 = linear(bias = m_encoder_layer_0_attention_self_key_bias_to_fp16, weight = m_encoder_layer_0_attention_self_key_weight_to_fp16, x = input_7_cast_fp16)[name = tensor("linear_1_cast_fp16")]; + tensor var_112 = const()[name = tensor("op_112"), val = tensor([1, -1, 12, 32])]; + tensor var_113_cast_fp16 = reshape(shape = var_112, x = linear_1_cast_fp16)[name = tensor("op_113_cast_fp16")]; + tensor m_encoder_layer_0_attention_self_value_weight_to_fp16 = const()[name = tensor("m_encoder_layer_0_attention_self_value_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(24429184)))]; + tensor m_encoder_layer_0_attention_self_value_bias_to_fp16 = const()[name = tensor("m_encoder_layer_0_attention_self_value_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(24724160)))]; + tensor linear_2_cast_fp16 = linear(bias = m_encoder_layer_0_attention_self_value_bias_to_fp16, weight = m_encoder_layer_0_attention_self_value_weight_to_fp16, x = input_7_cast_fp16)[name = tensor("linear_2_cast_fp16")]; + tensor var_118 = const()[name = tensor("op_118"), val = tensor([1, -1, 12, 32])]; + tensor var_119_cast_fp16 = reshape(shape = var_118, x = linear_2_cast_fp16)[name = tensor("op_119_cast_fp16")]; + tensor value_layer_1_perm_0 = const()[name = tensor("value_layer_1_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor mul_0_y_0_to_fp16 = const()[name = tensor("mul_0_y_0_to_fp16"), val = tensor(0x1.6ap-3)]; + tensor mul_0_cast_fp16 = mul(x = var_107_cast_fp16, y = mul_0_y_0_to_fp16)[name = tensor("mul_0_cast_fp16")]; + tensor matmul_0_transpose_y_0 = const()[name = tensor("matmul_0_transpose_y_0"), val = tensor(true)]; + tensor matmul_0_transpose_x_0 = const()[name = tensor("matmul_0_transpose_x_0"), val = tensor(false)]; + tensor transpose_24_perm_0 = const()[name = tensor("transpose_24_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_25_perm_0 = const()[name = tensor("transpose_25_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_25 = transpose(perm = transpose_25_perm_0, x = var_113_cast_fp16)[name = tensor("transpose_57")]; + tensor transpose_24 = transpose(perm = transpose_24_perm_0, x = mul_0_cast_fp16)[name = tensor("transpose_58")]; + tensor matmul_0_cast_fp16 = matmul(transpose_x = matmul_0_transpose_x_0, transpose_y = matmul_0_transpose_y_0, x = transpose_24, y = transpose_25)[name = tensor("matmul_0_cast_fp16")]; + tensor add_0_cast_fp16 = add(x = matmul_0_cast_fp16, y = attention_mask_cast_fp16)[name = tensor("add_0_cast_fp16")]; + tensor softmax_0_axis_0 = const()[name = tensor("softmax_0_axis_0"), val = tensor(-1)]; + tensor softmax_0_cast_fp16 = softmax(axis = softmax_0_axis_0, x = add_0_cast_fp16)[name = tensor("softmax_0_cast_fp16")]; + tensor attn_output_1_transpose_x_0 = const()[name = tensor("attn_output_1_transpose_x_0"), val = tensor(false)]; + tensor attn_output_1_transpose_y_0 = const()[name = tensor("attn_output_1_transpose_y_0"), val = tensor(false)]; + tensor value_layer_1_cast_fp16 = transpose(perm = value_layer_1_perm_0, x = var_119_cast_fp16)[name = tensor("transpose_59")]; + tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = softmax_0_cast_fp16, y = value_layer_1_cast_fp16)[name = tensor("attn_output_1_cast_fp16")]; + tensor attn_output_3_perm_0 = const()[name = tensor("attn_output_3_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_2x = const()[name = tensor("concat_2x"), val = tensor([1, -1, 384])]; + tensor attn_output_3_cast_fp16 = transpose(perm = attn_output_3_perm_0, x = attn_output_1_cast_fp16)[name = tensor("transpose_56")]; + tensor input_9_cast_fp16 = reshape(shape = concat_2x, x = attn_output_3_cast_fp16)[name = tensor("input_9_cast_fp16")]; + tensor m_encoder_layer_0_attention_output_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_0_attention_output_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(24724992)))]; + tensor m_encoder_layer_0_attention_output_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_0_attention_output_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(25019968)))]; + tensor linear_3_cast_fp16 = linear(bias = m_encoder_layer_0_attention_output_dense_bias_to_fp16, weight = m_encoder_layer_0_attention_output_dense_weight_to_fp16, x = input_9_cast_fp16)[name = tensor("linear_3_cast_fp16")]; + tensor input_13_cast_fp16 = add(x = linear_3_cast_fp16, y = input_7_cast_fp16)[name = tensor("input_13_cast_fp16")]; + tensor input_15_axes_0 = const()[name = tensor("input_15_axes_0"), val = tensor([-1])]; + tensor m_encoder_layer_0_attention_output_LayerNorm_weight_to_fp16 = const()[name = tensor("m_encoder_layer_0_attention_output_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(25020800)))]; + tensor m_encoder_layer_0_attention_output_LayerNorm_bias_to_fp16 = const()[name = tensor("m_encoder_layer_0_attention_output_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(25021632)))]; + tensor input_15_cast_fp16 = layer_norm(axes = input_15_axes_0, beta = m_encoder_layer_0_attention_output_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_encoder_layer_0_attention_output_LayerNorm_weight_to_fp16, x = input_13_cast_fp16)[name = tensor("input_15_cast_fp16")]; + tensor m_encoder_layer_0_intermediate_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_0_intermediate_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(25022464)))]; + tensor m_encoder_layer_0_intermediate_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_0_intermediate_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(26202176)))]; + tensor linear_4_cast_fp16 = linear(bias = m_encoder_layer_0_intermediate_dense_bias_to_fp16, weight = m_encoder_layer_0_intermediate_dense_weight_to_fp16, x = input_15_cast_fp16)[name = tensor("linear_4_cast_fp16")]; + tensor input_19_mode_0 = const()[name = tensor("input_19_mode_0"), val = tensor("EXACT")]; + tensor input_19_cast_fp16 = gelu(mode = input_19_mode_0, x = linear_4_cast_fp16)[name = tensor("input_19_cast_fp16")]; + tensor m_encoder_layer_0_output_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_0_output_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(26205312)))]; + tensor m_encoder_layer_0_output_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_0_output_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(27385024)))]; + tensor linear_5_cast_fp16 = linear(bias = m_encoder_layer_0_output_dense_bias_to_fp16, weight = m_encoder_layer_0_output_dense_weight_to_fp16, x = input_19_cast_fp16)[name = tensor("linear_5_cast_fp16")]; + tensor input_23_cast_fp16 = add(x = linear_5_cast_fp16, y = input_15_cast_fp16)[name = tensor("input_23_cast_fp16")]; + tensor hidden_states_7_axes_0 = const()[name = tensor("hidden_states_7_axes_0"), val = tensor([-1])]; + tensor m_encoder_layer_0_output_LayerNorm_weight_to_fp16 = const()[name = tensor("m_encoder_layer_0_output_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(27385856)))]; + tensor m_encoder_layer_0_output_LayerNorm_bias_to_fp16 = const()[name = tensor("m_encoder_layer_0_output_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(27386688)))]; + tensor hidden_states_7_cast_fp16 = layer_norm(axes = hidden_states_7_axes_0, beta = m_encoder_layer_0_output_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_encoder_layer_0_output_LayerNorm_weight_to_fp16, x = input_23_cast_fp16)[name = tensor("hidden_states_7_cast_fp16")]; + tensor m_encoder_layer_1_attention_self_query_weight_to_fp16 = const()[name = tensor("m_encoder_layer_1_attention_self_query_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(27387520)))]; + tensor m_encoder_layer_1_attention_self_query_bias_to_fp16 = const()[name = tensor("m_encoder_layer_1_attention_self_query_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(27682496)))]; + tensor linear_6_cast_fp16 = linear(bias = m_encoder_layer_1_attention_self_query_bias_to_fp16, weight = m_encoder_layer_1_attention_self_query_weight_to_fp16, x = hidden_states_7_cast_fp16)[name = tensor("linear_6_cast_fp16")]; + tensor var_165 = const()[name = tensor("op_165"), val = tensor([1, -1, 12, 32])]; + tensor var_166_cast_fp16 = reshape(shape = var_165, x = linear_6_cast_fp16)[name = tensor("op_166_cast_fp16")]; + tensor m_encoder_layer_1_attention_self_key_weight_to_fp16 = const()[name = tensor("m_encoder_layer_1_attention_self_key_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(27683328)))]; + tensor m_encoder_layer_1_attention_self_key_bias_to_fp16 = const()[name = tensor("m_encoder_layer_1_attention_self_key_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(27978304)))]; + tensor linear_7_cast_fp16 = linear(bias = m_encoder_layer_1_attention_self_key_bias_to_fp16, weight = m_encoder_layer_1_attention_self_key_weight_to_fp16, x = hidden_states_7_cast_fp16)[name = tensor("linear_7_cast_fp16")]; + tensor var_171 = const()[name = tensor("op_171"), val = tensor([1, -1, 12, 32])]; + tensor var_172_cast_fp16 = reshape(shape = var_171, x = linear_7_cast_fp16)[name = tensor("op_172_cast_fp16")]; + tensor m_encoder_layer_1_attention_self_value_weight_to_fp16 = const()[name = tensor("m_encoder_layer_1_attention_self_value_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(27979136)))]; + tensor m_encoder_layer_1_attention_self_value_bias_to_fp16 = const()[name = tensor("m_encoder_layer_1_attention_self_value_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(28274112)))]; + tensor linear_8_cast_fp16 = linear(bias = m_encoder_layer_1_attention_self_value_bias_to_fp16, weight = m_encoder_layer_1_attention_self_value_weight_to_fp16, x = hidden_states_7_cast_fp16)[name = tensor("linear_8_cast_fp16")]; + tensor var_177 = const()[name = tensor("op_177"), val = tensor([1, -1, 12, 32])]; + tensor var_178_cast_fp16 = reshape(shape = var_177, x = linear_8_cast_fp16)[name = tensor("op_178_cast_fp16")]; + tensor value_layer_3_perm_0 = const()[name = tensor("value_layer_3_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor mul_1_y_0_to_fp16 = const()[name = tensor("mul_1_y_0_to_fp16"), val = tensor(0x1.6ap-3)]; + tensor mul_1_cast_fp16 = mul(x = var_166_cast_fp16, y = mul_1_y_0_to_fp16)[name = tensor("mul_1_cast_fp16")]; + tensor matmul_1_transpose_y_0 = const()[name = tensor("matmul_1_transpose_y_0"), val = tensor(true)]; + tensor matmul_1_transpose_x_0 = const()[name = tensor("matmul_1_transpose_x_0"), val = tensor(false)]; + tensor transpose_26_perm_0 = const()[name = tensor("transpose_26_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_27_perm_0 = const()[name = tensor("transpose_27_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_27 = transpose(perm = transpose_27_perm_0, x = var_172_cast_fp16)[name = tensor("transpose_53")]; + tensor transpose_26 = transpose(perm = transpose_26_perm_0, x = mul_1_cast_fp16)[name = tensor("transpose_54")]; + tensor matmul_1_cast_fp16 = matmul(transpose_x = matmul_1_transpose_x_0, transpose_y = matmul_1_transpose_y_0, x = transpose_26, y = transpose_27)[name = tensor("matmul_1_cast_fp16")]; + tensor add_1_cast_fp16 = add(x = matmul_1_cast_fp16, y = attention_mask_cast_fp16)[name = tensor("add_1_cast_fp16")]; + tensor softmax_1_axis_0 = const()[name = tensor("softmax_1_axis_0"), val = tensor(-1)]; + tensor softmax_1_cast_fp16 = softmax(axis = softmax_1_axis_0, x = add_1_cast_fp16)[name = tensor("softmax_1_cast_fp16")]; + tensor attn_output_5_transpose_x_0 = const()[name = tensor("attn_output_5_transpose_x_0"), val = tensor(false)]; + tensor attn_output_5_transpose_y_0 = const()[name = tensor("attn_output_5_transpose_y_0"), val = tensor(false)]; + tensor value_layer_3_cast_fp16 = transpose(perm = value_layer_3_perm_0, x = var_178_cast_fp16)[name = tensor("transpose_55")]; + tensor attn_output_5_cast_fp16 = matmul(transpose_x = attn_output_5_transpose_x_0, transpose_y = attn_output_5_transpose_y_0, x = softmax_1_cast_fp16, y = value_layer_3_cast_fp16)[name = tensor("attn_output_5_cast_fp16")]; + tensor attn_output_7_perm_0 = const()[name = tensor("attn_output_7_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_3x = const()[name = tensor("concat_3x"), val = tensor([1, -1, 384])]; + tensor attn_output_7_cast_fp16 = transpose(perm = attn_output_7_perm_0, x = attn_output_5_cast_fp16)[name = tensor("transpose_52")]; + tensor input_25_cast_fp16 = reshape(shape = concat_3x, x = attn_output_7_cast_fp16)[name = tensor("input_25_cast_fp16")]; + tensor m_encoder_layer_1_attention_output_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_1_attention_output_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(28274944)))]; + tensor m_encoder_layer_1_attention_output_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_1_attention_output_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(28569920)))]; + tensor linear_9_cast_fp16 = linear(bias = m_encoder_layer_1_attention_output_dense_bias_to_fp16, weight = m_encoder_layer_1_attention_output_dense_weight_to_fp16, x = input_25_cast_fp16)[name = tensor("linear_9_cast_fp16")]; + tensor input_29_cast_fp16 = add(x = linear_9_cast_fp16, y = hidden_states_7_cast_fp16)[name = tensor("input_29_cast_fp16")]; + tensor input_31_axes_0 = const()[name = tensor("input_31_axes_0"), val = tensor([-1])]; + tensor m_encoder_layer_1_attention_output_LayerNorm_weight_to_fp16 = const()[name = tensor("m_encoder_layer_1_attention_output_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(28570752)))]; + tensor m_encoder_layer_1_attention_output_LayerNorm_bias_to_fp16 = const()[name = tensor("m_encoder_layer_1_attention_output_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(28571584)))]; + tensor input_31_cast_fp16 = layer_norm(axes = input_31_axes_0, beta = m_encoder_layer_1_attention_output_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_encoder_layer_1_attention_output_LayerNorm_weight_to_fp16, x = input_29_cast_fp16)[name = tensor("input_31_cast_fp16")]; + tensor m_encoder_layer_1_intermediate_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_1_intermediate_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(28572416)))]; + tensor m_encoder_layer_1_intermediate_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_1_intermediate_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(29752128)))]; + tensor linear_10_cast_fp16 = linear(bias = m_encoder_layer_1_intermediate_dense_bias_to_fp16, weight = m_encoder_layer_1_intermediate_dense_weight_to_fp16, x = input_31_cast_fp16)[name = tensor("linear_10_cast_fp16")]; + tensor input_35_mode_0 = const()[name = tensor("input_35_mode_0"), val = tensor("EXACT")]; + tensor input_35_cast_fp16 = gelu(mode = input_35_mode_0, x = linear_10_cast_fp16)[name = tensor("input_35_cast_fp16")]; + tensor m_encoder_layer_1_output_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_1_output_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(29755264)))]; + tensor m_encoder_layer_1_output_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_1_output_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(30934976)))]; + tensor linear_11_cast_fp16 = linear(bias = m_encoder_layer_1_output_dense_bias_to_fp16, weight = m_encoder_layer_1_output_dense_weight_to_fp16, x = input_35_cast_fp16)[name = tensor("linear_11_cast_fp16")]; + tensor input_39_cast_fp16 = add(x = linear_11_cast_fp16, y = input_31_cast_fp16)[name = tensor("input_39_cast_fp16")]; + tensor hidden_states_13_axes_0 = const()[name = tensor("hidden_states_13_axes_0"), val = tensor([-1])]; + tensor m_encoder_layer_1_output_LayerNorm_weight_to_fp16 = const()[name = tensor("m_encoder_layer_1_output_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(30935808)))]; + tensor m_encoder_layer_1_output_LayerNorm_bias_to_fp16 = const()[name = tensor("m_encoder_layer_1_output_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(30936640)))]; + tensor hidden_states_13_cast_fp16 = layer_norm(axes = hidden_states_13_axes_0, beta = m_encoder_layer_1_output_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_encoder_layer_1_output_LayerNorm_weight_to_fp16, x = input_39_cast_fp16)[name = tensor("hidden_states_13_cast_fp16")]; + tensor m_encoder_layer_2_attention_self_query_weight_to_fp16 = const()[name = tensor("m_encoder_layer_2_attention_self_query_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(30937472)))]; + tensor m_encoder_layer_2_attention_self_query_bias_to_fp16 = const()[name = tensor("m_encoder_layer_2_attention_self_query_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(31232448)))]; + tensor linear_12_cast_fp16 = linear(bias = m_encoder_layer_2_attention_self_query_bias_to_fp16, weight = m_encoder_layer_2_attention_self_query_weight_to_fp16, x = hidden_states_13_cast_fp16)[name = tensor("linear_12_cast_fp16")]; + tensor var_224 = const()[name = tensor("op_224"), val = tensor([1, -1, 12, 32])]; + tensor var_225_cast_fp16 = reshape(shape = var_224, x = linear_12_cast_fp16)[name = tensor("op_225_cast_fp16")]; + tensor m_encoder_layer_2_attention_self_key_weight_to_fp16 = const()[name = tensor("m_encoder_layer_2_attention_self_key_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(31233280)))]; + tensor m_encoder_layer_2_attention_self_key_bias_to_fp16 = const()[name = tensor("m_encoder_layer_2_attention_self_key_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(31528256)))]; + tensor linear_13_cast_fp16 = linear(bias = m_encoder_layer_2_attention_self_key_bias_to_fp16, weight = m_encoder_layer_2_attention_self_key_weight_to_fp16, x = hidden_states_13_cast_fp16)[name = tensor("linear_13_cast_fp16")]; + tensor var_230 = const()[name = tensor("op_230"), val = tensor([1, -1, 12, 32])]; + tensor var_231_cast_fp16 = reshape(shape = var_230, x = linear_13_cast_fp16)[name = tensor("op_231_cast_fp16")]; + tensor m_encoder_layer_2_attention_self_value_weight_to_fp16 = const()[name = tensor("m_encoder_layer_2_attention_self_value_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(31529088)))]; + tensor m_encoder_layer_2_attention_self_value_bias_to_fp16 = const()[name = tensor("m_encoder_layer_2_attention_self_value_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(31824064)))]; + tensor linear_14_cast_fp16 = linear(bias = m_encoder_layer_2_attention_self_value_bias_to_fp16, weight = m_encoder_layer_2_attention_self_value_weight_to_fp16, x = hidden_states_13_cast_fp16)[name = tensor("linear_14_cast_fp16")]; + tensor var_236 = const()[name = tensor("op_236"), val = tensor([1, -1, 12, 32])]; + tensor var_237_cast_fp16 = reshape(shape = var_236, x = linear_14_cast_fp16)[name = tensor("op_237_cast_fp16")]; + tensor value_layer_5_perm_0 = const()[name = tensor("value_layer_5_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor mul_2_y_0_to_fp16 = const()[name = tensor("mul_2_y_0_to_fp16"), val = tensor(0x1.6ap-3)]; + tensor mul_2_cast_fp16 = mul(x = var_225_cast_fp16, y = mul_2_y_0_to_fp16)[name = tensor("mul_2_cast_fp16")]; + tensor matmul_2_transpose_y_0 = const()[name = tensor("matmul_2_transpose_y_0"), val = tensor(true)]; + tensor matmul_2_transpose_x_0 = const()[name = tensor("matmul_2_transpose_x_0"), val = tensor(false)]; + tensor transpose_28_perm_0 = const()[name = tensor("transpose_28_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_29_perm_0 = const()[name = tensor("transpose_29_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_29 = transpose(perm = transpose_29_perm_0, x = var_231_cast_fp16)[name = tensor("transpose_49")]; + tensor transpose_28 = transpose(perm = transpose_28_perm_0, x = mul_2_cast_fp16)[name = tensor("transpose_50")]; + tensor matmul_2_cast_fp16 = matmul(transpose_x = matmul_2_transpose_x_0, transpose_y = matmul_2_transpose_y_0, x = transpose_28, y = transpose_29)[name = tensor("matmul_2_cast_fp16")]; + tensor add_2_cast_fp16 = add(x = matmul_2_cast_fp16, y = attention_mask_cast_fp16)[name = tensor("add_2_cast_fp16")]; + tensor softmax_2_axis_0 = const()[name = tensor("softmax_2_axis_0"), val = tensor(-1)]; + tensor softmax_2_cast_fp16 = softmax(axis = softmax_2_axis_0, x = add_2_cast_fp16)[name = tensor("softmax_2_cast_fp16")]; + tensor attn_output_9_transpose_x_0 = const()[name = tensor("attn_output_9_transpose_x_0"), val = tensor(false)]; + tensor attn_output_9_transpose_y_0 = const()[name = tensor("attn_output_9_transpose_y_0"), val = tensor(false)]; + tensor value_layer_5_cast_fp16 = transpose(perm = value_layer_5_perm_0, x = var_237_cast_fp16)[name = tensor("transpose_51")]; + tensor attn_output_9_cast_fp16 = matmul(transpose_x = attn_output_9_transpose_x_0, transpose_y = attn_output_9_transpose_y_0, x = softmax_2_cast_fp16, y = value_layer_5_cast_fp16)[name = tensor("attn_output_9_cast_fp16")]; + tensor attn_output_11_perm_0 = const()[name = tensor("attn_output_11_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_4x = const()[name = tensor("concat_4x"), val = tensor([1, -1, 384])]; + tensor attn_output_11_cast_fp16 = transpose(perm = attn_output_11_perm_0, x = attn_output_9_cast_fp16)[name = tensor("transpose_48")]; + tensor input_41_cast_fp16 = reshape(shape = concat_4x, x = attn_output_11_cast_fp16)[name = tensor("input_41_cast_fp16")]; + tensor m_encoder_layer_2_attention_output_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_2_attention_output_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(31824896)))]; + tensor m_encoder_layer_2_attention_output_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_2_attention_output_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(32119872)))]; + tensor linear_15_cast_fp16 = linear(bias = m_encoder_layer_2_attention_output_dense_bias_to_fp16, weight = m_encoder_layer_2_attention_output_dense_weight_to_fp16, x = input_41_cast_fp16)[name = tensor("linear_15_cast_fp16")]; + tensor input_45_cast_fp16 = add(x = linear_15_cast_fp16, y = hidden_states_13_cast_fp16)[name = tensor("input_45_cast_fp16")]; + tensor input_47_axes_0 = const()[name = tensor("input_47_axes_0"), val = tensor([-1])]; + tensor m_encoder_layer_2_attention_output_LayerNorm_weight_to_fp16 = const()[name = tensor("m_encoder_layer_2_attention_output_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(32120704)))]; + tensor m_encoder_layer_2_attention_output_LayerNorm_bias_to_fp16 = const()[name = tensor("m_encoder_layer_2_attention_output_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(32121536)))]; + tensor input_47_cast_fp16 = layer_norm(axes = input_47_axes_0, beta = m_encoder_layer_2_attention_output_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_encoder_layer_2_attention_output_LayerNorm_weight_to_fp16, x = input_45_cast_fp16)[name = tensor("input_47_cast_fp16")]; + tensor m_encoder_layer_2_intermediate_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_2_intermediate_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(32122368)))]; + tensor m_encoder_layer_2_intermediate_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_2_intermediate_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(33302080)))]; + tensor linear_16_cast_fp16 = linear(bias = m_encoder_layer_2_intermediate_dense_bias_to_fp16, weight = m_encoder_layer_2_intermediate_dense_weight_to_fp16, x = input_47_cast_fp16)[name = tensor("linear_16_cast_fp16")]; + tensor input_51_mode_0 = const()[name = tensor("input_51_mode_0"), val = tensor("EXACT")]; + tensor input_51_cast_fp16 = gelu(mode = input_51_mode_0, x = linear_16_cast_fp16)[name = tensor("input_51_cast_fp16")]; + tensor m_encoder_layer_2_output_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_2_output_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(33305216)))]; + tensor m_encoder_layer_2_output_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_2_output_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(34484928)))]; + tensor linear_17_cast_fp16 = linear(bias = m_encoder_layer_2_output_dense_bias_to_fp16, weight = m_encoder_layer_2_output_dense_weight_to_fp16, x = input_51_cast_fp16)[name = tensor("linear_17_cast_fp16")]; + tensor input_55_cast_fp16 = add(x = linear_17_cast_fp16, y = input_47_cast_fp16)[name = tensor("input_55_cast_fp16")]; + tensor hidden_states_19_axes_0 = const()[name = tensor("hidden_states_19_axes_0"), val = tensor([-1])]; + tensor m_encoder_layer_2_output_LayerNorm_weight_to_fp16 = const()[name = tensor("m_encoder_layer_2_output_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(34485760)))]; + tensor m_encoder_layer_2_output_LayerNorm_bias_to_fp16 = const()[name = tensor("m_encoder_layer_2_output_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(34486592)))]; + tensor hidden_states_19_cast_fp16 = layer_norm(axes = hidden_states_19_axes_0, beta = m_encoder_layer_2_output_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_encoder_layer_2_output_LayerNorm_weight_to_fp16, x = input_55_cast_fp16)[name = tensor("hidden_states_19_cast_fp16")]; + tensor m_encoder_layer_3_attention_self_query_weight_to_fp16 = const()[name = tensor("m_encoder_layer_3_attention_self_query_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(34487424)))]; + tensor m_encoder_layer_3_attention_self_query_bias_to_fp16 = const()[name = tensor("m_encoder_layer_3_attention_self_query_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(34782400)))]; + tensor linear_18_cast_fp16 = linear(bias = m_encoder_layer_3_attention_self_query_bias_to_fp16, weight = m_encoder_layer_3_attention_self_query_weight_to_fp16, x = hidden_states_19_cast_fp16)[name = tensor("linear_18_cast_fp16")]; + tensor var_283 = const()[name = tensor("op_283"), val = tensor([1, -1, 12, 32])]; + tensor var_284_cast_fp16 = reshape(shape = var_283, x = linear_18_cast_fp16)[name = tensor("op_284_cast_fp16")]; + tensor m_encoder_layer_3_attention_self_key_weight_to_fp16 = const()[name = tensor("m_encoder_layer_3_attention_self_key_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(34783232)))]; + tensor m_encoder_layer_3_attention_self_key_bias_to_fp16 = const()[name = tensor("m_encoder_layer_3_attention_self_key_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35078208)))]; + tensor linear_19_cast_fp16 = linear(bias = m_encoder_layer_3_attention_self_key_bias_to_fp16, weight = m_encoder_layer_3_attention_self_key_weight_to_fp16, x = hidden_states_19_cast_fp16)[name = tensor("linear_19_cast_fp16")]; + tensor var_289 = const()[name = tensor("op_289"), val = tensor([1, -1, 12, 32])]; + tensor var_290_cast_fp16 = reshape(shape = var_289, x = linear_19_cast_fp16)[name = tensor("op_290_cast_fp16")]; + tensor m_encoder_layer_3_attention_self_value_weight_to_fp16 = const()[name = tensor("m_encoder_layer_3_attention_self_value_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35079040)))]; + tensor m_encoder_layer_3_attention_self_value_bias_to_fp16 = const()[name = tensor("m_encoder_layer_3_attention_self_value_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35374016)))]; + tensor linear_20_cast_fp16 = linear(bias = m_encoder_layer_3_attention_self_value_bias_to_fp16, weight = m_encoder_layer_3_attention_self_value_weight_to_fp16, x = hidden_states_19_cast_fp16)[name = tensor("linear_20_cast_fp16")]; + tensor var_295 = const()[name = tensor("op_295"), val = tensor([1, -1, 12, 32])]; + tensor var_296_cast_fp16 = reshape(shape = var_295, x = linear_20_cast_fp16)[name = tensor("op_296_cast_fp16")]; + tensor value_layer_7_perm_0 = const()[name = tensor("value_layer_7_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor mul_3_y_0_to_fp16 = const()[name = tensor("mul_3_y_0_to_fp16"), val = tensor(0x1.6ap-3)]; + tensor mul_3_cast_fp16 = mul(x = var_284_cast_fp16, y = mul_3_y_0_to_fp16)[name = tensor("mul_3_cast_fp16")]; + tensor matmul_3_transpose_y_0 = const()[name = tensor("matmul_3_transpose_y_0"), val = tensor(true)]; + tensor matmul_3_transpose_x_0 = const()[name = tensor("matmul_3_transpose_x_0"), val = tensor(false)]; + tensor transpose_30_perm_0 = const()[name = tensor("transpose_30_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_31_perm_0 = const()[name = tensor("transpose_31_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_31 = transpose(perm = transpose_31_perm_0, x = var_290_cast_fp16)[name = tensor("transpose_45")]; + tensor transpose_30 = transpose(perm = transpose_30_perm_0, x = mul_3_cast_fp16)[name = tensor("transpose_46")]; + tensor matmul_3_cast_fp16 = matmul(transpose_x = matmul_3_transpose_x_0, transpose_y = matmul_3_transpose_y_0, x = transpose_30, y = transpose_31)[name = tensor("matmul_3_cast_fp16")]; + tensor add_3_cast_fp16 = add(x = matmul_3_cast_fp16, y = attention_mask_cast_fp16)[name = tensor("add_3_cast_fp16")]; + tensor softmax_3_axis_0 = const()[name = tensor("softmax_3_axis_0"), val = tensor(-1)]; + tensor softmax_3_cast_fp16 = softmax(axis = softmax_3_axis_0, x = add_3_cast_fp16)[name = tensor("softmax_3_cast_fp16")]; + tensor attn_output_13_transpose_x_0 = const()[name = tensor("attn_output_13_transpose_x_0"), val = tensor(false)]; + tensor attn_output_13_transpose_y_0 = const()[name = tensor("attn_output_13_transpose_y_0"), val = tensor(false)]; + tensor value_layer_7_cast_fp16 = transpose(perm = value_layer_7_perm_0, x = var_296_cast_fp16)[name = tensor("transpose_47")]; + tensor attn_output_13_cast_fp16 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = softmax_3_cast_fp16, y = value_layer_7_cast_fp16)[name = tensor("attn_output_13_cast_fp16")]; + tensor attn_output_15_perm_0 = const()[name = tensor("attn_output_15_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_5x = const()[name = tensor("concat_5x"), val = tensor([1, -1, 384])]; + tensor attn_output_15_cast_fp16 = transpose(perm = attn_output_15_perm_0, x = attn_output_13_cast_fp16)[name = tensor("transpose_44")]; + tensor input_57_cast_fp16 = reshape(shape = concat_5x, x = attn_output_15_cast_fp16)[name = tensor("input_57_cast_fp16")]; + tensor m_encoder_layer_3_attention_output_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_3_attention_output_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35374848)))]; + tensor m_encoder_layer_3_attention_output_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_3_attention_output_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35669824)))]; + tensor linear_21_cast_fp16 = linear(bias = m_encoder_layer_3_attention_output_dense_bias_to_fp16, weight = m_encoder_layer_3_attention_output_dense_weight_to_fp16, x = input_57_cast_fp16)[name = tensor("linear_21_cast_fp16")]; + tensor input_61_cast_fp16 = add(x = linear_21_cast_fp16, y = hidden_states_19_cast_fp16)[name = tensor("input_61_cast_fp16")]; + tensor input_63_axes_0 = const()[name = tensor("input_63_axes_0"), val = tensor([-1])]; + tensor m_encoder_layer_3_attention_output_LayerNorm_weight_to_fp16 = const()[name = tensor("m_encoder_layer_3_attention_output_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35670656)))]; + tensor m_encoder_layer_3_attention_output_LayerNorm_bias_to_fp16 = const()[name = tensor("m_encoder_layer_3_attention_output_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35671488)))]; + tensor input_63_cast_fp16 = layer_norm(axes = input_63_axes_0, beta = m_encoder_layer_3_attention_output_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_encoder_layer_3_attention_output_LayerNorm_weight_to_fp16, x = input_61_cast_fp16)[name = tensor("input_63_cast_fp16")]; + tensor m_encoder_layer_3_intermediate_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_3_intermediate_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(35672320)))]; + tensor m_encoder_layer_3_intermediate_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_3_intermediate_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(36852032)))]; + tensor linear_22_cast_fp16 = linear(bias = m_encoder_layer_3_intermediate_dense_bias_to_fp16, weight = m_encoder_layer_3_intermediate_dense_weight_to_fp16, x = input_63_cast_fp16)[name = tensor("linear_22_cast_fp16")]; + tensor input_67_mode_0 = const()[name = tensor("input_67_mode_0"), val = tensor("EXACT")]; + tensor input_67_cast_fp16 = gelu(mode = input_67_mode_0, x = linear_22_cast_fp16)[name = tensor("input_67_cast_fp16")]; + tensor m_encoder_layer_3_output_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_3_output_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(36855168)))]; + tensor m_encoder_layer_3_output_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_3_output_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(38034880)))]; + tensor linear_23_cast_fp16 = linear(bias = m_encoder_layer_3_output_dense_bias_to_fp16, weight = m_encoder_layer_3_output_dense_weight_to_fp16, x = input_67_cast_fp16)[name = tensor("linear_23_cast_fp16")]; + tensor input_71_cast_fp16 = add(x = linear_23_cast_fp16, y = input_63_cast_fp16)[name = tensor("input_71_cast_fp16")]; + tensor hidden_states_25_axes_0 = const()[name = tensor("hidden_states_25_axes_0"), val = tensor([-1])]; + tensor m_encoder_layer_3_output_LayerNorm_weight_to_fp16 = const()[name = tensor("m_encoder_layer_3_output_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(38035712)))]; + tensor m_encoder_layer_3_output_LayerNorm_bias_to_fp16 = const()[name = tensor("m_encoder_layer_3_output_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(38036544)))]; + tensor hidden_states_25_cast_fp16 = layer_norm(axes = hidden_states_25_axes_0, beta = m_encoder_layer_3_output_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_encoder_layer_3_output_LayerNorm_weight_to_fp16, x = input_71_cast_fp16)[name = tensor("hidden_states_25_cast_fp16")]; + tensor m_encoder_layer_4_attention_self_query_weight_to_fp16 = const()[name = tensor("m_encoder_layer_4_attention_self_query_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(38037376)))]; + tensor m_encoder_layer_4_attention_self_query_bias_to_fp16 = const()[name = tensor("m_encoder_layer_4_attention_self_query_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(38332352)))]; + tensor linear_24_cast_fp16 = linear(bias = m_encoder_layer_4_attention_self_query_bias_to_fp16, weight = m_encoder_layer_4_attention_self_query_weight_to_fp16, x = hidden_states_25_cast_fp16)[name = tensor("linear_24_cast_fp16")]; + tensor var_342 = const()[name = tensor("op_342"), val = tensor([1, -1, 12, 32])]; + tensor var_343_cast_fp16 = reshape(shape = var_342, x = linear_24_cast_fp16)[name = tensor("op_343_cast_fp16")]; + tensor m_encoder_layer_4_attention_self_key_weight_to_fp16 = const()[name = tensor("m_encoder_layer_4_attention_self_key_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(38333184)))]; + tensor m_encoder_layer_4_attention_self_key_bias_to_fp16 = const()[name = tensor("m_encoder_layer_4_attention_self_key_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(38628160)))]; + tensor linear_25_cast_fp16 = linear(bias = m_encoder_layer_4_attention_self_key_bias_to_fp16, weight = m_encoder_layer_4_attention_self_key_weight_to_fp16, x = hidden_states_25_cast_fp16)[name = tensor("linear_25_cast_fp16")]; + tensor var_348 = const()[name = tensor("op_348"), val = tensor([1, -1, 12, 32])]; + tensor var_349_cast_fp16 = reshape(shape = var_348, x = linear_25_cast_fp16)[name = tensor("op_349_cast_fp16")]; + tensor m_encoder_layer_4_attention_self_value_weight_to_fp16 = const()[name = tensor("m_encoder_layer_4_attention_self_value_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(38628992)))]; + tensor m_encoder_layer_4_attention_self_value_bias_to_fp16 = const()[name = tensor("m_encoder_layer_4_attention_self_value_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(38923968)))]; + tensor linear_26_cast_fp16 = linear(bias = m_encoder_layer_4_attention_self_value_bias_to_fp16, weight = m_encoder_layer_4_attention_self_value_weight_to_fp16, x = hidden_states_25_cast_fp16)[name = tensor("linear_26_cast_fp16")]; + tensor var_354 = const()[name = tensor("op_354"), val = tensor([1, -1, 12, 32])]; + tensor var_355_cast_fp16 = reshape(shape = var_354, x = linear_26_cast_fp16)[name = tensor("op_355_cast_fp16")]; + tensor value_layer_9_perm_0 = const()[name = tensor("value_layer_9_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor mul_4_y_0_to_fp16 = const()[name = tensor("mul_4_y_0_to_fp16"), val = tensor(0x1.6ap-3)]; + tensor mul_4_cast_fp16 = mul(x = var_343_cast_fp16, y = mul_4_y_0_to_fp16)[name = tensor("mul_4_cast_fp16")]; + tensor matmul_4_transpose_y_0 = const()[name = tensor("matmul_4_transpose_y_0"), val = tensor(true)]; + tensor matmul_4_transpose_x_0 = const()[name = tensor("matmul_4_transpose_x_0"), val = tensor(false)]; + tensor transpose_32_perm_0 = const()[name = tensor("transpose_32_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_33_perm_0 = const()[name = tensor("transpose_33_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_33 = transpose(perm = transpose_33_perm_0, x = var_349_cast_fp16)[name = tensor("transpose_41")]; + tensor transpose_32 = transpose(perm = transpose_32_perm_0, x = mul_4_cast_fp16)[name = tensor("transpose_42")]; + tensor matmul_4_cast_fp16 = matmul(transpose_x = matmul_4_transpose_x_0, transpose_y = matmul_4_transpose_y_0, x = transpose_32, y = transpose_33)[name = tensor("matmul_4_cast_fp16")]; + tensor add_4_cast_fp16 = add(x = matmul_4_cast_fp16, y = attention_mask_cast_fp16)[name = tensor("add_4_cast_fp16")]; + tensor softmax_4_axis_0 = const()[name = tensor("softmax_4_axis_0"), val = tensor(-1)]; + tensor softmax_4_cast_fp16 = softmax(axis = softmax_4_axis_0, x = add_4_cast_fp16)[name = tensor("softmax_4_cast_fp16")]; + tensor attn_output_17_transpose_x_0 = const()[name = tensor("attn_output_17_transpose_x_0"), val = tensor(false)]; + tensor attn_output_17_transpose_y_0 = const()[name = tensor("attn_output_17_transpose_y_0"), val = tensor(false)]; + tensor value_layer_9_cast_fp16 = transpose(perm = value_layer_9_perm_0, x = var_355_cast_fp16)[name = tensor("transpose_43")]; + tensor attn_output_17_cast_fp16 = matmul(transpose_x = attn_output_17_transpose_x_0, transpose_y = attn_output_17_transpose_y_0, x = softmax_4_cast_fp16, y = value_layer_9_cast_fp16)[name = tensor("attn_output_17_cast_fp16")]; + tensor attn_output_19_perm_0 = const()[name = tensor("attn_output_19_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_6x = const()[name = tensor("concat_6x"), val = tensor([1, -1, 384])]; + tensor attn_output_19_cast_fp16 = transpose(perm = attn_output_19_perm_0, x = attn_output_17_cast_fp16)[name = tensor("transpose_40")]; + tensor input_73_cast_fp16 = reshape(shape = concat_6x, x = attn_output_19_cast_fp16)[name = tensor("input_73_cast_fp16")]; + tensor m_encoder_layer_4_attention_output_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_4_attention_output_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(38924800)))]; + tensor m_encoder_layer_4_attention_output_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_4_attention_output_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(39219776)))]; + tensor linear_27_cast_fp16 = linear(bias = m_encoder_layer_4_attention_output_dense_bias_to_fp16, weight = m_encoder_layer_4_attention_output_dense_weight_to_fp16, x = input_73_cast_fp16)[name = tensor("linear_27_cast_fp16")]; + tensor input_77_cast_fp16 = add(x = linear_27_cast_fp16, y = hidden_states_25_cast_fp16)[name = tensor("input_77_cast_fp16")]; + tensor input_79_axes_0 = const()[name = tensor("input_79_axes_0"), val = tensor([-1])]; + tensor m_encoder_layer_4_attention_output_LayerNorm_weight_to_fp16 = const()[name = tensor("m_encoder_layer_4_attention_output_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(39220608)))]; + tensor m_encoder_layer_4_attention_output_LayerNorm_bias_to_fp16 = const()[name = tensor("m_encoder_layer_4_attention_output_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(39221440)))]; + tensor input_79_cast_fp16 = layer_norm(axes = input_79_axes_0, beta = m_encoder_layer_4_attention_output_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_encoder_layer_4_attention_output_LayerNorm_weight_to_fp16, x = input_77_cast_fp16)[name = tensor("input_79_cast_fp16")]; + tensor m_encoder_layer_4_intermediate_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_4_intermediate_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(39222272)))]; + tensor m_encoder_layer_4_intermediate_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_4_intermediate_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(40401984)))]; + tensor linear_28_cast_fp16 = linear(bias = m_encoder_layer_4_intermediate_dense_bias_to_fp16, weight = m_encoder_layer_4_intermediate_dense_weight_to_fp16, x = input_79_cast_fp16)[name = tensor("linear_28_cast_fp16")]; + tensor input_83_mode_0 = const()[name = tensor("input_83_mode_0"), val = tensor("EXACT")]; + tensor input_83_cast_fp16 = gelu(mode = input_83_mode_0, x = linear_28_cast_fp16)[name = tensor("input_83_cast_fp16")]; + tensor m_encoder_layer_4_output_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_4_output_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(40405120)))]; + tensor m_encoder_layer_4_output_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_4_output_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(41584832)))]; + tensor linear_29_cast_fp16 = linear(bias = m_encoder_layer_4_output_dense_bias_to_fp16, weight = m_encoder_layer_4_output_dense_weight_to_fp16, x = input_83_cast_fp16)[name = tensor("linear_29_cast_fp16")]; + tensor input_87_cast_fp16 = add(x = linear_29_cast_fp16, y = input_79_cast_fp16)[name = tensor("input_87_cast_fp16")]; + tensor hidden_states_31_axes_0 = const()[name = tensor("hidden_states_31_axes_0"), val = tensor([-1])]; + tensor m_encoder_layer_4_output_LayerNorm_weight_to_fp16 = const()[name = tensor("m_encoder_layer_4_output_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(41585664)))]; + tensor m_encoder_layer_4_output_LayerNorm_bias_to_fp16 = const()[name = tensor("m_encoder_layer_4_output_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(41586496)))]; + tensor hidden_states_31_cast_fp16 = layer_norm(axes = hidden_states_31_axes_0, beta = m_encoder_layer_4_output_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_encoder_layer_4_output_LayerNorm_weight_to_fp16, x = input_87_cast_fp16)[name = tensor("hidden_states_31_cast_fp16")]; + tensor m_encoder_layer_5_attention_self_query_weight_to_fp16 = const()[name = tensor("m_encoder_layer_5_attention_self_query_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(41587328)))]; + tensor m_encoder_layer_5_attention_self_query_bias_to_fp16 = const()[name = tensor("m_encoder_layer_5_attention_self_query_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(41882304)))]; + tensor linear_30_cast_fp16 = linear(bias = m_encoder_layer_5_attention_self_query_bias_to_fp16, weight = m_encoder_layer_5_attention_self_query_weight_to_fp16, x = hidden_states_31_cast_fp16)[name = tensor("linear_30_cast_fp16")]; + tensor var_401 = const()[name = tensor("op_401"), val = tensor([1, -1, 12, 32])]; + tensor var_402_cast_fp16 = reshape(shape = var_401, x = linear_30_cast_fp16)[name = tensor("op_402_cast_fp16")]; + tensor m_encoder_layer_5_attention_self_key_weight_to_fp16 = const()[name = tensor("m_encoder_layer_5_attention_self_key_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(41883136)))]; + tensor m_encoder_layer_5_attention_self_key_bias_to_fp16 = const()[name = tensor("m_encoder_layer_5_attention_self_key_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(42178112)))]; + tensor linear_31_cast_fp16 = linear(bias = m_encoder_layer_5_attention_self_key_bias_to_fp16, weight = m_encoder_layer_5_attention_self_key_weight_to_fp16, x = hidden_states_31_cast_fp16)[name = tensor("linear_31_cast_fp16")]; + tensor var_407 = const()[name = tensor("op_407"), val = tensor([1, -1, 12, 32])]; + tensor var_408_cast_fp16 = reshape(shape = var_407, x = linear_31_cast_fp16)[name = tensor("op_408_cast_fp16")]; + tensor m_encoder_layer_5_attention_self_value_weight_to_fp16 = const()[name = tensor("m_encoder_layer_5_attention_self_value_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(42178944)))]; + tensor m_encoder_layer_5_attention_self_value_bias_to_fp16 = const()[name = tensor("m_encoder_layer_5_attention_self_value_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(42473920)))]; + tensor linear_32_cast_fp16 = linear(bias = m_encoder_layer_5_attention_self_value_bias_to_fp16, weight = m_encoder_layer_5_attention_self_value_weight_to_fp16, x = hidden_states_31_cast_fp16)[name = tensor("linear_32_cast_fp16")]; + tensor var_413 = const()[name = tensor("op_413"), val = tensor([1, -1, 12, 32])]; + tensor var_414_cast_fp16 = reshape(shape = var_413, x = linear_32_cast_fp16)[name = tensor("op_414_cast_fp16")]; + tensor value_layer_perm_0 = const()[name = tensor("value_layer_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor mul_5_y_0_to_fp16 = const()[name = tensor("mul_5_y_0_to_fp16"), val = tensor(0x1.6ap-3)]; + tensor mul_5_cast_fp16 = mul(x = var_402_cast_fp16, y = mul_5_y_0_to_fp16)[name = tensor("mul_5_cast_fp16")]; + tensor matmul_5_transpose_y_0 = const()[name = tensor("matmul_5_transpose_y_0"), val = tensor(true)]; + tensor matmul_5_transpose_x_0 = const()[name = tensor("matmul_5_transpose_x_0"), val = tensor(false)]; + tensor transpose_34_perm_0 = const()[name = tensor("transpose_34_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_35_perm_0 = const()[name = tensor("transpose_35_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_35 = transpose(perm = transpose_35_perm_0, x = var_408_cast_fp16)[name = tensor("transpose_37")]; + tensor transpose_34 = transpose(perm = transpose_34_perm_0, x = mul_5_cast_fp16)[name = tensor("transpose_38")]; + tensor matmul_5_cast_fp16 = matmul(transpose_x = matmul_5_transpose_x_0, transpose_y = matmul_5_transpose_y_0, x = transpose_34, y = transpose_35)[name = tensor("matmul_5_cast_fp16")]; + tensor add_5_cast_fp16 = add(x = matmul_5_cast_fp16, y = attention_mask_cast_fp16)[name = tensor("add_5_cast_fp16")]; + tensor softmax_5_axis_0 = const()[name = tensor("softmax_5_axis_0"), val = tensor(-1)]; + tensor softmax_5_cast_fp16 = softmax(axis = softmax_5_axis_0, x = add_5_cast_fp16)[name = tensor("softmax_5_cast_fp16")]; + tensor attn_output_21_transpose_x_0 = const()[name = tensor("attn_output_21_transpose_x_0"), val = tensor(false)]; + tensor attn_output_21_transpose_y_0 = const()[name = tensor("attn_output_21_transpose_y_0"), val = tensor(false)]; + tensor value_layer_cast_fp16 = transpose(perm = value_layer_perm_0, x = var_414_cast_fp16)[name = tensor("transpose_39")]; + tensor attn_output_21_cast_fp16 = matmul(transpose_x = attn_output_21_transpose_x_0, transpose_y = attn_output_21_transpose_y_0, x = softmax_5_cast_fp16, y = value_layer_cast_fp16)[name = tensor("attn_output_21_cast_fp16")]; + tensor attn_output_perm_0 = const()[name = tensor("attn_output_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_7x = const()[name = tensor("concat_7x"), val = tensor([1, -1, 384])]; + tensor attn_output_cast_fp16 = transpose(perm = attn_output_perm_0, x = attn_output_21_cast_fp16)[name = tensor("transpose_36")]; + tensor input_89_cast_fp16 = reshape(shape = concat_7x, x = attn_output_cast_fp16)[name = tensor("input_89_cast_fp16")]; + tensor m_encoder_layer_5_attention_output_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_5_attention_output_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(42474752)))]; + tensor m_encoder_layer_5_attention_output_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_5_attention_output_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(42769728)))]; + tensor linear_33_cast_fp16 = linear(bias = m_encoder_layer_5_attention_output_dense_bias_to_fp16, weight = m_encoder_layer_5_attention_output_dense_weight_to_fp16, x = input_89_cast_fp16)[name = tensor("linear_33_cast_fp16")]; + tensor input_93_cast_fp16 = add(x = linear_33_cast_fp16, y = hidden_states_31_cast_fp16)[name = tensor("input_93_cast_fp16")]; + tensor input_95_axes_0 = const()[name = tensor("input_95_axes_0"), val = tensor([-1])]; + tensor m_encoder_layer_5_attention_output_LayerNorm_weight_to_fp16 = const()[name = tensor("m_encoder_layer_5_attention_output_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(42770560)))]; + tensor m_encoder_layer_5_attention_output_LayerNorm_bias_to_fp16 = const()[name = tensor("m_encoder_layer_5_attention_output_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(42771392)))]; + tensor input_95_cast_fp16 = layer_norm(axes = input_95_axes_0, beta = m_encoder_layer_5_attention_output_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_encoder_layer_5_attention_output_LayerNorm_weight_to_fp16, x = input_93_cast_fp16)[name = tensor("input_95_cast_fp16")]; + tensor m_encoder_layer_5_intermediate_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_5_intermediate_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(42772224)))]; + tensor m_encoder_layer_5_intermediate_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_5_intermediate_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(43951936)))]; + tensor linear_34_cast_fp16 = linear(bias = m_encoder_layer_5_intermediate_dense_bias_to_fp16, weight = m_encoder_layer_5_intermediate_dense_weight_to_fp16, x = input_95_cast_fp16)[name = tensor("linear_34_cast_fp16")]; + tensor input_99_mode_0 = const()[name = tensor("input_99_mode_0"), val = tensor("EXACT")]; + tensor input_99_cast_fp16 = gelu(mode = input_99_mode_0, x = linear_34_cast_fp16)[name = tensor("input_99_cast_fp16")]; + tensor m_encoder_layer_5_output_dense_weight_to_fp16 = const()[name = tensor("m_encoder_layer_5_output_dense_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(43955072)))]; + tensor m_encoder_layer_5_output_dense_bias_to_fp16 = const()[name = tensor("m_encoder_layer_5_output_dense_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(45134784)))]; + tensor linear_35_cast_fp16 = linear(bias = m_encoder_layer_5_output_dense_bias_to_fp16, weight = m_encoder_layer_5_output_dense_weight_to_fp16, x = input_99_cast_fp16)[name = tensor("linear_35_cast_fp16")]; + tensor input_103_cast_fp16 = add(x = linear_35_cast_fp16, y = input_95_cast_fp16)[name = tensor("input_103_cast_fp16")]; + tensor token_embeddings_axes_0 = const()[name = tensor("token_embeddings_axes_0"), val = tensor([-1])]; + tensor m_encoder_layer_5_output_LayerNorm_weight_to_fp16 = const()[name = tensor("m_encoder_layer_5_output_LayerNorm_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(45135616)))]; + tensor m_encoder_layer_5_output_LayerNorm_bias_to_fp16 = const()[name = tensor("m_encoder_layer_5_output_LayerNorm_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(45136448)))]; + tensor token_embeddings_cast_fp16 = layer_norm(axes = token_embeddings_axes_0, beta = m_encoder_layer_5_output_LayerNorm_bias_to_fp16, epsilon = var_35_to_fp16, gamma = m_encoder_layer_5_output_LayerNorm_weight_to_fp16, x = input_103_cast_fp16)[name = tensor("token_embeddings_cast_fp16")]; + tensor var_452 = const()[name = tensor("op_452"), val = tensor(-1)]; + tensor var_453_axes_0 = const()[name = tensor("op_453_axes_0"), val = tensor([-1])]; + tensor var_453 = expand_dims(axes = var_453_axes_0, x = mask_1)[name = tensor("op_453")]; + tensor shape_2_cast_fp16 = shape(x = token_embeddings_cast_fp16)[name = tensor("shape_2_cast_fp16")]; + tensor shape_3 = shape(x = var_453)[name = tensor("shape_3")]; + tensor equal_1_y_0 = const()[name = tensor("equal_1_y_0"), val = tensor(-1)]; + tensor equal_1 = equal(x = shape_2_cast_fp16, y = equal_1_y_0)[name = tensor("equal_1")]; + tensor select_1 = select(a = shape_3, b = shape_2_cast_fp16, cond = equal_1)[name = tensor("select_1")]; + tensor real_div_1 = real_div(x = select_1, y = shape_3)[name = tensor("real_div_1")]; + tensor var_454 = tile(reps = real_div_1, x = var_453)[name = tensor("op_454")]; + tensor mask_to_fp16_dtype_0 = const()[name = tensor("mask_to_fp16_dtype_0"), val = tensor("fp16")]; + tensor var_454_to_fp16 = cast(dtype = mask_to_fp16_dtype_0, x = var_454)[name = tensor("cast_38")]; + tensor var_456_cast_fp16 = mul(x = token_embeddings_cast_fp16, y = var_454_to_fp16)[name = tensor("op_456_cast_fp16")]; + tensor mean_sum_axes_0 = const()[name = tensor("mean_sum_axes_0"), val = tensor([1])]; + tensor mean_sum_keep_dims_0 = const()[name = tensor("mean_sum_keep_dims_0"), val = tensor(false)]; + tensor mean_sum_cast_fp16 = reduce_sum(axes = mean_sum_axes_0, keep_dims = mean_sum_keep_dims_0, x = var_456_cast_fp16)[name = tensor("mean_sum_cast_fp16")]; + tensor mean_mask_1_axes_0 = const()[name = tensor("mean_mask_1_axes_0"), val = tensor([1])]; + tensor mean_mask_1_keep_dims_0 = const()[name = tensor("mean_mask_1_keep_dims_0"), val = tensor(false)]; + tensor mean_mask_1_cast_fp16 = reduce_sum(axes = mean_mask_1_axes_0, keep_dims = mean_mask_1_keep_dims_0, x = var_454_to_fp16)[name = tensor("mean_mask_1_cast_fp16")]; + tensor var_447_to_fp16 = const()[name = tensor("op_447_to_fp16"), val = tensor(0x1p-24)]; + tensor const_1_to_fp16 = const()[name = tensor("const_1_to_fp16"), val = tensor(inf)]; + tensor clip_0_cast_fp16 = clip(alpha = var_447_to_fp16, beta = const_1_to_fp16, x = mean_mask_1_cast_fp16)[name = tensor("clip_0_cast_fp16")]; + tensor var_462_cast_fp16 = real_div(x = mean_sum_cast_fp16, y = clip_0_cast_fp16)[name = tensor("op_462_cast_fp16")]; + tensor input_105_interleave_0 = const()[name = tensor("input_105_interleave_0"), val = tensor(false)]; + tensor input_105_cast_fp16 = concat(axis = var_452, interleave = input_105_interleave_0, values = var_462_cast_fp16)[name = tensor("input_105_cast_fp16")]; + tensor rest_1_linear_weight_to_fp16 = const()[name = tensor("rest_1_linear_weight_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(45137280)))]; + tensor rest_1_linear_bias_to_fp16 = const()[name = tensor("rest_1_linear_bias_to_fp16"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(45727168)))]; + tensor linear_36_cast_fp16 = linear(bias = rest_1_linear_bias_to_fp16, weight = rest_1_linear_weight_to_fp16, x = input_105_cast_fp16)[name = tensor("linear_36_cast_fp16")]; + tensor var_471 = const()[name = tensor("op_471"), val = tensor(true)]; + tensor var_474 = const()[name = tensor("op_474"), val = tensor([1])]; + tensor var_475_cast_fp16 = reduce_l2_norm(axes = var_474, keep_dims = var_471, x = linear_36_cast_fp16)[name = tensor("op_475_cast_fp16")]; + tensor var_469_to_fp16 = const()[name = tensor("op_469_to_fp16"), val = tensor(0x1p-24)]; + tensor var_476_cast_fp16 = maximum(x = var_475_cast_fp16, y = var_469_to_fp16)[name = tensor("op_476_cast_fp16")]; + tensor shape_4_cast_fp16 = shape(x = linear_36_cast_fp16)[name = tensor("shape_4_cast_fp16")]; + tensor shape_5_cast_fp16 = shape(x = var_476_cast_fp16)[name = tensor("shape_5_cast_fp16")]; + tensor equal_2_y_0 = const()[name = tensor("equal_2_y_0"), val = tensor(-1)]; + tensor equal_2 = equal(x = shape_4_cast_fp16, y = equal_2_y_0)[name = tensor("equal_2")]; + tensor select_2 = select(a = shape_5_cast_fp16, b = shape_4_cast_fp16, cond = equal_2)[name = tensor("select_2")]; + tensor real_div_2 = real_div(x = select_2, y = shape_5_cast_fp16)[name = tensor("real_div_2")]; + tensor denom_1_cast_fp16 = tile(reps = real_div_2, x = var_476_cast_fp16)[name = tensor("denom_1_cast_fp16")]; + tensor input_cast_fp16 = real_div(x = linear_36_cast_fp16, y = denom_1_cast_fp16)[name = tensor("input_cast_fp16")]; + tensor var_481 = const()[name = tensor("op_481"), val = tensor([-1])]; + tensor var_482 = const()[name = tensor("op_482"), val = tensor(true)]; + tensor var_484_cast_fp16 = reduce_l2_norm(axes = var_481, keep_dims = var_482, x = input_cast_fp16)[name = tensor("op_484_cast_fp16")]; + tensor var_485_to_fp16 = const()[name = tensor("op_485_to_fp16"), val = tensor(0x1p-24)]; + tensor var_486_cast_fp16 = maximum(x = var_484_cast_fp16, y = var_485_to_fp16)[name = tensor("op_486_cast_fp16")]; + tensor shape_6_cast_fp16 = shape(x = input_cast_fp16)[name = tensor("shape_6_cast_fp16")]; + tensor shape_7_cast_fp16 = shape(x = var_486_cast_fp16)[name = tensor("shape_7_cast_fp16")]; + tensor equal_3_y_0 = const()[name = tensor("equal_3_y_0"), val = tensor(-1)]; + tensor equal_3 = equal(x = shape_6_cast_fp16, y = equal_3_y_0)[name = tensor("equal_3")]; + tensor select_3 = select(a = shape_7_cast_fp16, b = shape_6_cast_fp16, cond = equal_3)[name = tensor("select_3")]; + tensor real_div_3 = real_div(x = select_3, y = shape_7_cast_fp16)[name = tensor("real_div_3")]; + tensor denom_cast_fp16 = tile(reps = real_div_3, x = var_486_cast_fp16)[name = tensor("denom_cast_fp16")]; + tensor var_488_cast_fp16 = real_div(x = input_cast_fp16, y = denom_cast_fp16)[name = tensor("op_488_cast_fp16")]; + tensor var_488_cast_fp16_to_fp32_dtype_0 = const()[name = tensor("op_488_cast_fp16_to_fp32_dtype_0"), val = tensor("fp32")]; + tensor embedding = cast(dtype = var_488_cast_fp16_to_fp32_dtype_0, x = var_488_cast_fp16)[name = tensor("cast_37")]; + } -> (embedding); +} \ No newline at end of file