program(1.3) [buildInfo = dict({{"coremlc-component-MIL", "3500.14.1"}, {"coremlc-version", "3500.32.1"}})] { func main(tensor causal_mask, tensor cos_f, tensor cos_s, tensor hidden_states, tensor kv13_k, tensor kv13_v, tensor kv14_k, tensor kv14_v, tensor per_layer_combined, tensor sin_f, tensor sin_s) { tensor layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3145856))))[name = string("layers_0_self_attn_q_proj_weight_palettized")]; tensor layers_0_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3150016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12587264))))[name = string("layers_0_mlp_gate_proj_weight_palettized")]; tensor layers_0_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12599616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22036864))))[name = string("layers_0_mlp_up_proj_weight_palettized")]; tensor layers_0_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(22049216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31486464))))[name = string("layers_0_mlp_down_proj_weight_palettized")]; tensor layers_0_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31488064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31684736))))[name = string("layers_0_per_layer_input_gate_weight_palettized")]; tensor layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31685056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33257984))))[name = string("layers_1_self_attn_q_proj_weight_palettized")]; tensor layers_1_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33260096))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(42697344))))[name = string("layers_1_mlp_gate_proj_weight_palettized")]; tensor layers_1_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(42709696))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52146944))))[name = string("layers_1_mlp_up_proj_weight_palettized")]; tensor layers_1_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(52159296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61596544))))[name = string("layers_1_mlp_down_proj_weight_palettized")]; tensor layers_1_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61598144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61794816))))[name = string("layers_1_per_layer_input_gate_weight_palettized")]; tensor layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(61795136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(63368064))))[name = string("layers_2_self_attn_q_proj_weight_palettized")]; tensor layers_2_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(63370176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72807424))))[name = string("layers_2_mlp_gate_proj_weight_palettized")]; tensor layers_2_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72819776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82257024))))[name = string("layers_2_mlp_up_proj_weight_palettized")]; tensor layers_2_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(82269376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91706624))))[name = string("layers_2_mlp_down_proj_weight_palettized")]; tensor layers_2_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91708224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91904896))))[name = string("layers_2_per_layer_input_gate_weight_palettized")]; tensor layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91905216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93478144))))[name = string("layers_3_self_attn_q_proj_weight_palettized")]; tensor layers_3_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(93480256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102917504))))[name = string("layers_3_mlp_gate_proj_weight_palettized")]; tensor layers_3_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102929856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112367104))))[name = string("layers_3_mlp_up_proj_weight_palettized")]; tensor layers_3_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(112379456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(121816704))))[name = string("layers_3_mlp_down_proj_weight_palettized")]; tensor layers_3_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(121818304))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(122014976))))[name = string("layers_3_per_layer_input_gate_weight_palettized")]; tensor layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(122015296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123588224))))[name = string("layers_4_self_attn_q_proj_weight_palettized")]; tensor layers_4_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123590336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133027584))))[name = string("layers_4_mlp_gate_proj_weight_palettized")]; tensor layers_4_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133039936))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(142477184))))[name = string("layers_4_mlp_up_proj_weight_palettized")]; tensor layers_4_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(142489536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151926784))))[name = string("layers_4_mlp_down_proj_weight_palettized")]; tensor layers_4_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151928384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(152125056))))[name = string("layers_4_per_layer_input_gate_weight_palettized")]; tensor layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(152125376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155271168))))[name = string("layers_5_self_attn_q_proj_weight_palettized")]; tensor layers_5_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155275328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(164712576))))[name = string("layers_5_mlp_gate_proj_weight_palettized")]; tensor layers_5_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(164724928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174162176))))[name = string("layers_5_mlp_up_proj_weight_palettized")]; tensor layers_5_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(174174528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(183611776))))[name = string("layers_5_mlp_down_proj_weight_palettized")]; tensor layers_5_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(183613376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(183810048))))[name = string("layers_5_per_layer_input_gate_weight_palettized")]; tensor layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(183810368))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(185383296))))[name = string("layers_6_self_attn_q_proj_weight_palettized")]; tensor layers_6_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(185385408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(194822656))))[name = string("layers_6_mlp_gate_proj_weight_palettized")]; tensor layers_6_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(194835008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204272256))))[name = string("layers_6_mlp_up_proj_weight_palettized")]; tensor layers_6_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204284608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(213721856))))[name = string("layers_6_mlp_down_proj_weight_palettized")]; tensor layers_6_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(213723456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(213920128))))[name = string("layers_6_per_layer_input_gate_weight_palettized")]; tensor layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(213920448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(215493376))))[name = string("layers_7_self_attn_q_proj_weight_palettized")]; tensor layers_7_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(215495488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224932736))))[name = string("layers_7_mlp_gate_proj_weight_palettized")]; tensor layers_7_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(224945088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(234382336))))[name = string("layers_7_mlp_up_proj_weight_palettized")]; tensor layers_7_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(234394688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243831936))))[name = string("layers_7_mlp_down_proj_weight_palettized")]; tensor layers_7_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243833536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(244030208))))[name = string("layers_7_per_layer_input_gate_weight_palettized")]; tensor layers_8_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(244030528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245603456))))[name = string("layers_8_self_attn_q_proj_weight_palettized")]; tensor layers_8_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245605568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(255042816))))[name = string("layers_8_mlp_gate_proj_weight_palettized")]; tensor layers_8_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(255055168))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(264492416))))[name = string("layers_8_mlp_up_proj_weight_palettized")]; tensor layers_8_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(264504768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273942016))))[name = string("layers_8_mlp_down_proj_weight_palettized")]; tensor layers_8_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273943616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274140288))))[name = string("layers_8_per_layer_input_gate_weight_palettized")]; tensor layers_9_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274140608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275713536))))[name = string("layers_9_self_attn_q_proj_weight_palettized")]; tensor layers_9_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275715648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(285152896))))[name = string("layers_9_mlp_gate_proj_weight_palettized")]; tensor layers_9_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(285165248))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294602496))))[name = string("layers_9_mlp_up_proj_weight_palettized")]; tensor layers_9_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(294614848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304052096))))[name = string("layers_9_mlp_down_proj_weight_palettized")]; tensor layers_9_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304053696))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304250368))))[name = string("layers_9_per_layer_input_gate_weight_palettized")]; tensor layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(304250688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307396480))))[name = string("layers_10_self_attn_q_proj_weight_palettized")]; tensor layers_10_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307400640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316837888))))[name = string("layers_10_mlp_gate_proj_weight_palettized")]; tensor layers_10_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(316850240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326287488))))[name = string("layers_10_mlp_up_proj_weight_palettized")]; tensor layers_10_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(326299840))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335737088))))[name = string("layers_10_mlp_down_proj_weight_palettized")]; tensor layers_10_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335738688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335935360))))[name = string("layers_10_per_layer_input_gate_weight_palettized")]; int32 var_546 = const()[name = string("op_546"), val = int32(-1)]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_552_cast_fp16 = mul(x = hidden_states, y = const_0_promoted_to_fp16)[name = string("op_552_cast_fp16")]; bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; tensor input_1_cast_fp16 = concat(axis = var_546, interleave = input_1_interleave_0, values = (hidden_states, var_552_cast_fp16))[name = string("input_1_cast_fp16")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_544_to_fp16 = const()[name = string("op_544_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_544_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; tensor var_557_split_sizes_0 = const()[name = string("op_557_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_557_axis_0 = const()[name = string("op_557_axis_0"), val = int32(-1)]; tensor var_557_cast_fp16_0, tensor var_557_cast_fp16_1 = split(axis = var_557_axis_0, split_sizes = var_557_split_sizes_0, x = normed_1_cast_fp16)[name = string("op_557_cast_fp16")]; tensor const_1_to_fp16 = const()[name = string("const_1_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335935680)))]; tensor var_560_cast_fp16 = mul(x = var_557_cast_fp16_0, y = const_1_to_fp16)[name = string("op_560_cast_fp16")]; tensor var_565 = const()[name = string("op_565"), val = tensor([0, 2, 1])]; tensor var_568_axes_0 = const()[name = string("op_568_axes_0"), val = tensor([2])]; tensor var_566 = transpose(perm = var_565, x = var_560_cast_fp16)[name = string("transpose_90")]; tensor var_568 = expand_dims(axes = var_568_axes_0, x = var_566)[name = string("op_568")]; string var_584_pad_type_0 = const()[name = string("op_584_pad_type_0"), val = string("valid")]; tensor var_584_strides_0 = const()[name = string("op_584_strides_0"), val = tensor([1, 1])]; tensor var_584_pad_0 = const()[name = string("op_584_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_584_dilations_0 = const()[name = string("op_584_dilations_0"), val = tensor([1, 1])]; int32 var_584_groups_0 = const()[name = string("op_584_groups_0"), val = int32(1)]; tensor var_584 = conv(dilations = var_584_dilations_0, groups = var_584_groups_0, pad = var_584_pad_0, pad_type = var_584_pad_type_0, strides = var_584_strides_0, weight = layers_0_self_attn_q_proj_weight_palettized, x = var_568)[name = string("op_584")]; tensor var_589 = const()[name = string("op_589"), val = tensor([1, 8, 512, 1])]; tensor var_590 = reshape(shape = var_589, x = var_584)[name = string("op_590")]; tensor var_595 = const()[name = string("op_595"), val = tensor([0, 1, 3, 2])]; tensor var_605 = const()[name = string("op_605"), val = tensor([1, 8, 512])]; tensor var_596 = transpose(perm = var_595, x = var_590)[name = string("transpose_89")]; tensor x_3 = reshape(shape = var_605, x = var_596)[name = string("x_3")]; int32 var_611 = const()[name = string("op_611"), val = int32(-1)]; fp16 const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_617_cast_fp16 = mul(x = x_3, y = const_2_promoted_to_fp16)[name = string("op_617_cast_fp16")]; bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; tensor input_5_cast_fp16 = concat(axis = var_611, interleave = input_5_interleave_0, values = (x_3, var_617_cast_fp16))[name = string("input_5_cast_fp16")]; tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; fp16 var_609_to_fp16 = const()[name = string("op_609_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_609_to_fp16, x = input_5_cast_fp16)[name = string("normed_5_cast_fp16")]; tensor var_622_split_sizes_0 = const()[name = string("op_622_split_sizes_0"), val = tensor([512, 512])]; int32 var_622_axis_0 = const()[name = string("op_622_axis_0"), val = int32(-1)]; tensor var_622_cast_fp16_0, tensor var_622_cast_fp16_1 = split(axis = var_622_axis_0, split_sizes = var_622_split_sizes_0, x = normed_5_cast_fp16)[name = string("op_622_cast_fp16")]; tensor const_3_to_fp16 = const()[name = string("const_3_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335938816)))]; tensor var_625_cast_fp16 = mul(x = var_622_cast_fp16_0, y = const_3_to_fp16)[name = string("op_625_cast_fp16")]; tensor var_631 = const()[name = string("op_631"), val = tensor([1, 8, 1, 512])]; tensor q_3 = reshape(shape = var_631, x = var_625_cast_fp16)[name = string("q_3")]; tensor var_633_cast_fp16 = mul(x = q_3, y = cos_f)[name = string("op_633_cast_fp16")]; tensor var_634_split_sizes_0 = const()[name = string("op_634_split_sizes_0"), val = tensor([256, 256])]; int32 var_634_axis_0 = const()[name = string("op_634_axis_0"), val = int32(-1)]; tensor var_634_0, tensor var_634_1 = split(axis = var_634_axis_0, split_sizes = var_634_split_sizes_0, x = q_3)[name = string("op_634")]; fp16 const_4_promoted = const()[name = string("const_4_promoted"), val = fp16(-0x1p+0)]; tensor var_636 = mul(x = var_634_1, y = const_4_promoted)[name = string("op_636")]; int32 var_638 = const()[name = string("op_638"), val = int32(-1)]; bool var_639_interleave_0 = const()[name = string("op_639_interleave_0"), val = bool(false)]; tensor var_639 = concat(axis = var_638, interleave = var_639_interleave_0, values = (var_636, var_634_0))[name = string("op_639")]; tensor var_640_cast_fp16 = mul(x = var_639, y = sin_f)[name = string("op_640_cast_fp16")]; tensor q_5_cast_fp16 = add(x = var_633_cast_fp16, y = var_640_cast_fp16)[name = string("q_5_cast_fp16")]; tensor transpose_0_perm_0 = const()[name = string("transpose_0_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_0_reps_0 = const()[name = string("tile_0_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_0_cast_fp16 = transpose(perm = transpose_0_perm_0, x = kv14_k)[name = string("transpose_88")]; tensor tile_0_cast_fp16 = tile(reps = tile_0_reps_0, x = transpose_0_cast_fp16)[name = string("tile_0_cast_fp16")]; tensor concat_0 = const()[name = string("concat_0"), val = tensor([8, 1, 1, 512, 512])]; tensor reshape_0_cast_fp16 = reshape(shape = concat_0, x = tile_0_cast_fp16)[name = string("reshape_0_cast_fp16")]; tensor transpose_1_perm_0 = const()[name = string("transpose_1_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_1 = const()[name = string("concat_1"), val = tensor([-1, 1, 512, 512])]; tensor transpose_1_cast_fp16 = transpose(perm = transpose_1_perm_0, x = reshape_0_cast_fp16)[name = string("transpose_87")]; tensor reshape_1_cast_fp16 = reshape(shape = concat_1, x = transpose_1_cast_fp16)[name = string("reshape_1_cast_fp16")]; tensor transpose_44_perm_0 = const()[name = string("transpose_44_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_2_perm_0 = const()[name = string("transpose_2_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_1_reps_0 = const()[name = string("tile_1_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_2_cast_fp16 = transpose(perm = transpose_2_perm_0, x = kv14_v)[name = string("transpose_86")]; tensor tile_1_cast_fp16 = tile(reps = tile_1_reps_0, x = transpose_2_cast_fp16)[name = string("tile_1_cast_fp16")]; tensor concat_2 = const()[name = string("concat_2"), val = tensor([8, 1, 1, 512, 512])]; tensor reshape_2_cast_fp16 = reshape(shape = concat_2, x = tile_1_cast_fp16)[name = string("reshape_2_cast_fp16")]; tensor transpose_3_perm_0 = const()[name = string("transpose_3_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_3 = const()[name = string("concat_3"), val = tensor([-1, 1, 512, 512])]; tensor transpose_3_cast_fp16 = transpose(perm = transpose_3_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_85")]; tensor reshape_3_cast_fp16 = reshape(shape = concat_3, x = transpose_3_cast_fp16)[name = string("reshape_3_cast_fp16")]; tensor Ve_1_perm_0 = const()[name = string("Ve_1_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_664_transpose_x_0 = const()[name = string("op_664_transpose_x_0"), val = bool(false)]; bool var_664_transpose_y_0 = const()[name = string("op_664_transpose_y_0"), val = bool(false)]; tensor transpose_44_cast_fp16 = transpose(perm = transpose_44_perm_0, x = reshape_1_cast_fp16)[name = string("transpose_84")]; tensor var_664_cast_fp16 = matmul(transpose_x = var_664_transpose_x_0, transpose_y = var_664_transpose_y_0, x = q_5_cast_fp16, y = transpose_44_cast_fp16)[name = string("op_664_cast_fp16")]; tensor var_671_cast_fp16 = add(x = var_664_cast_fp16, y = causal_mask)[name = string("op_671_cast_fp16")]; int32 var_672 = const()[name = string("op_672"), val = int32(-1)]; tensor var_674_cast_fp16 = softmax(axis = var_672, x = var_671_cast_fp16)[name = string("op_674_cast_fp16")]; bool var_690_transpose_x_0 = const()[name = string("op_690_transpose_x_0"), val = bool(false)]; bool var_690_transpose_y_0 = const()[name = string("op_690_transpose_y_0"), val = bool(false)]; tensor Ve_1_cast_fp16 = transpose(perm = Ve_1_perm_0, x = reshape_3_cast_fp16)[name = string("transpose_83")]; tensor var_690_cast_fp16 = matmul(transpose_x = var_690_transpose_x_0, transpose_y = var_690_transpose_y_0, x = var_674_cast_fp16, y = Ve_1_cast_fp16)[name = string("op_690_cast_fp16")]; tensor var_700 = const()[name = string("op_700"), val = tensor([0, 2, 1, 3])]; tensor var_707 = const()[name = string("op_707"), val = tensor([1, 1, -1])]; tensor var_701 = transpose(perm = var_700, x = var_690_cast_fp16)[name = string("transpose_82")]; tensor var_708 = reshape(shape = var_707, x = var_701)[name = string("op_708")]; tensor var_712 = const()[name = string("op_712"), val = tensor([0, 2, 1])]; tensor squeeze_0_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335939904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339085696))))[name = string("squeeze_0_palettized")]; string var_728_pad_type_0 = const()[name = string("op_728_pad_type_0"), val = string("valid")]; int32 var_728_groups_0 = const()[name = string("op_728_groups_0"), val = int32(1)]; tensor var_728_strides_0 = const()[name = string("op_728_strides_0"), val = tensor([1])]; tensor var_728_pad_0 = const()[name = string("op_728_pad_0"), val = tensor([0, 0])]; tensor var_728_dilations_0 = const()[name = string("op_728_dilations_0"), val = tensor([1])]; tensor var_713 = transpose(perm = var_712, x = var_708)[name = string("transpose_81")]; tensor var_728 = conv(dilations = var_728_dilations_0, groups = var_728_groups_0, pad = var_728_pad_0, pad_type = var_728_pad_type_0, strides = var_728_strides_0, weight = squeeze_0_palettized, x = var_713)[name = string("op_728")]; tensor var_732 = const()[name = string("op_732"), val = tensor([0, 2, 1])]; int32 var_738 = const()[name = string("op_738"), val = int32(-1)]; fp16 const_5_promoted_to_fp16 = const()[name = string("const_5_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_7 = transpose(perm = var_732, x = var_728)[name = string("transpose_80")]; tensor var_744_cast_fp16 = mul(x = x_7, y = const_5_promoted_to_fp16)[name = string("op_744_cast_fp16")]; bool input_9_interleave_0 = const()[name = string("input_9_interleave_0"), val = bool(false)]; tensor input_9_cast_fp16 = concat(axis = var_738, interleave = input_9_interleave_0, values = (x_7, var_744_cast_fp16))[name = string("input_9_cast_fp16")]; tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; fp16 var_736_to_fp16 = const()[name = string("op_736_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_736_to_fp16, x = input_9_cast_fp16)[name = string("normed_9_cast_fp16")]; tensor var_749_split_sizes_0 = const()[name = string("op_749_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_749_axis_0 = const()[name = string("op_749_axis_0"), val = int32(-1)]; tensor var_749_cast_fp16_0, tensor var_749_cast_fp16_1 = split(axis = var_749_axis_0, split_sizes = var_749_split_sizes_0, x = normed_9_cast_fp16)[name = string("op_749_cast_fp16")]; tensor const_6_to_fp16 = const()[name = string("const_6_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339087296)))]; tensor var_752_cast_fp16 = mul(x = var_749_cast_fp16_0, y = const_6_to_fp16)[name = string("op_752_cast_fp16")]; tensor x_11_cast_fp16 = add(x = hidden_states, y = var_752_cast_fp16)[name = string("x_11_cast_fp16")]; int32 var_760 = const()[name = string("op_760"), val = int32(-1)]; fp16 const_7_promoted_to_fp16 = const()[name = string("const_7_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_766_cast_fp16 = mul(x = x_11_cast_fp16, y = const_7_promoted_to_fp16)[name = string("op_766_cast_fp16")]; bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; tensor input_11_cast_fp16 = concat(axis = var_760, interleave = input_11_interleave_0, values = (x_11_cast_fp16, var_766_cast_fp16))[name = string("input_11_cast_fp16")]; tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; fp16 var_758_to_fp16 = const()[name = string("op_758_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_758_to_fp16, x = input_11_cast_fp16)[name = string("normed_13_cast_fp16")]; tensor var_771_split_sizes_0 = const()[name = string("op_771_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_771_axis_0 = const()[name = string("op_771_axis_0"), val = int32(-1)]; tensor var_771_cast_fp16_0, tensor var_771_cast_fp16_1 = split(axis = var_771_axis_0, split_sizes = var_771_split_sizes_0, x = normed_13_cast_fp16)[name = string("op_771_cast_fp16")]; tensor const_8_to_fp16 = const()[name = string("const_8_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339090432)))]; tensor var_774_cast_fp16 = mul(x = var_771_cast_fp16_0, y = const_8_to_fp16)[name = string("op_774_cast_fp16")]; tensor var_784 = const()[name = string("op_784"), val = tensor([0, 2, 1])]; tensor input_13_axes_0 = const()[name = string("input_13_axes_0"), val = tensor([2])]; tensor var_785 = transpose(perm = var_784, x = var_774_cast_fp16)[name = string("transpose_79")]; tensor input_13 = expand_dims(axes = input_13_axes_0, x = var_785)[name = string("input_13")]; string var_798_pad_type_0 = const()[name = string("op_798_pad_type_0"), val = string("valid")]; tensor var_798_strides_0 = const()[name = string("op_798_strides_0"), val = tensor([1, 1])]; tensor var_798_pad_0 = const()[name = string("op_798_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_798_dilations_0 = const()[name = string("op_798_dilations_0"), val = tensor([1, 1])]; int32 var_798_groups_0 = const()[name = string("op_798_groups_0"), val = int32(1)]; tensor var_798 = conv(dilations = var_798_dilations_0, groups = var_798_groups_0, pad = var_798_pad_0, pad_type = var_798_pad_type_0, strides = var_798_strides_0, weight = layers_0_mlp_gate_proj_weight_palettized, x = input_13)[name = string("op_798")]; string var_800_mode_0 = const()[name = string("op_800_mode_0"), val = string("TANH_APPROXIMATION")]; tensor var_800 = gelu(mode = var_800_mode_0, x = var_798)[name = string("op_800")]; string var_811_pad_type_0 = const()[name = string("op_811_pad_type_0"), val = string("valid")]; tensor var_811_strides_0 = const()[name = string("op_811_strides_0"), val = tensor([1, 1])]; tensor var_811_pad_0 = const()[name = string("op_811_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_811_dilations_0 = const()[name = string("op_811_dilations_0"), val = tensor([1, 1])]; int32 var_811_groups_0 = const()[name = string("op_811_groups_0"), val = int32(1)]; tensor var_811 = conv(dilations = var_811_dilations_0, groups = var_811_groups_0, pad = var_811_pad_0, pad_type = var_811_pad_type_0, strides = var_811_strides_0, weight = layers_0_mlp_up_proj_weight_palettized, x = input_13)[name = string("op_811")]; tensor input_15 = mul(x = var_800, y = var_811)[name = string("input_15")]; string var_823_pad_type_0 = const()[name = string("op_823_pad_type_0"), val = string("valid")]; tensor var_823_strides_0 = const()[name = string("op_823_strides_0"), val = tensor([1, 1])]; tensor var_823_pad_0 = const()[name = string("op_823_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_823_dilations_0 = const()[name = string("op_823_dilations_0"), val = tensor([1, 1])]; int32 var_823_groups_0 = const()[name = string("op_823_groups_0"), val = int32(1)]; tensor var_823 = conv(dilations = var_823_dilations_0, groups = var_823_groups_0, pad = var_823_pad_0, pad_type = var_823_pad_type_0, strides = var_823_strides_0, weight = layers_0_mlp_down_proj_weight_palettized, x = input_15)[name = string("op_823")]; tensor var_825_axes_0 = const()[name = string("op_825_axes_0"), val = tensor([2])]; tensor var_825 = squeeze(axes = var_825_axes_0, x = var_823)[name = string("op_825")]; tensor var_829 = const()[name = string("op_829"), val = tensor([0, 2, 1])]; int32 var_835 = const()[name = string("op_835"), val = int32(-1)]; fp16 const_9_promoted_to_fp16 = const()[name = string("const_9_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_15 = transpose(perm = var_829, x = var_825)[name = string("transpose_78")]; tensor var_841_cast_fp16 = mul(x = x_15, y = const_9_promoted_to_fp16)[name = string("op_841_cast_fp16")]; bool input_17_interleave_0 = const()[name = string("input_17_interleave_0"), val = bool(false)]; tensor input_17_cast_fp16 = concat(axis = var_835, interleave = input_17_interleave_0, values = (x_15, var_841_cast_fp16))[name = string("input_17_cast_fp16")]; tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; fp16 var_833_to_fp16 = const()[name = string("op_833_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_833_to_fp16, x = input_17_cast_fp16)[name = string("normed_17_cast_fp16")]; tensor var_846_split_sizes_0 = const()[name = string("op_846_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_846_axis_0 = const()[name = string("op_846_axis_0"), val = int32(-1)]; tensor var_846_cast_fp16_0, tensor var_846_cast_fp16_1 = split(axis = var_846_axis_0, split_sizes = var_846_split_sizes_0, x = normed_17_cast_fp16)[name = string("op_846_cast_fp16")]; tensor const_10_to_fp16 = const()[name = string("const_10_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339093568)))]; tensor var_849_cast_fp16 = mul(x = var_846_cast_fp16_0, y = const_10_to_fp16)[name = string("op_849_cast_fp16")]; tensor hidden_states_13_cast_fp16 = add(x = x_11_cast_fp16, y = var_849_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; tensor linear_0_bias_0 = const()[name = string("linear_0_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339096704)))]; tensor var_860 = linear(bias = linear_0_bias_0, weight = layers_0_per_layer_input_gate_weight_palettized, x = hidden_states_13_cast_fp16)[name = string("linear_0")]; string gated_1_mode_0 = const()[name = string("gated_1_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_1 = gelu(mode = gated_1_mode_0, x = var_860)[name = string("gated_1")]; tensor var_877_begin_0 = const()[name = string("op_877_begin_0"), val = tensor([0, 0, 6144])]; tensor var_877_end_0 = const()[name = string("op_877_end_0"), val = tensor([1, 1, 6400])]; tensor var_877_end_mask_0 = const()[name = string("op_877_end_mask_0"), val = tensor([true, true, false])]; tensor var_877_cast_fp16 = slice_by_index(begin = var_877_begin_0, end = var_877_end_0, end_mask = var_877_end_mask_0, x = per_layer_combined)[name = string("op_877_cast_fp16")]; tensor input_21_cast_fp16 = mul(x = gated_1, y = var_877_cast_fp16)[name = string("input_21_cast_fp16")]; tensor layers_0_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339097280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339293952))))[name = string("layers_0_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_1_bias_0_to_fp16 = const()[name = string("linear_1_bias_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339295552)))]; tensor linear_1_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_0_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_21_cast_fp16)[name = string("linear_1_cast_fp16")]; int32 var_886 = const()[name = string("op_886"), val = int32(-1)]; fp16 const_11_promoted_to_fp16 = const()[name = string("const_11_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_892_cast_fp16 = mul(x = linear_1_cast_fp16, y = const_11_promoted_to_fp16)[name = string("op_892_cast_fp16")]; bool input_23_interleave_0 = const()[name = string("input_23_interleave_0"), val = bool(false)]; tensor input_23_cast_fp16 = concat(axis = var_886, interleave = input_23_interleave_0, values = (linear_1_cast_fp16, var_892_cast_fp16))[name = string("input_23_cast_fp16")]; tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; fp16 var_884_to_fp16 = const()[name = string("op_884_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_884_to_fp16, x = input_23_cast_fp16)[name = string("normed_21_cast_fp16")]; tensor var_897_split_sizes_0 = const()[name = string("op_897_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_897_axis_0 = const()[name = string("op_897_axis_0"), val = int32(-1)]; tensor var_897_cast_fp16_0, tensor var_897_cast_fp16_1 = split(axis = var_897_axis_0, split_sizes = var_897_split_sizes_0, x = normed_21_cast_fp16)[name = string("op_897_cast_fp16")]; tensor const_12_to_fp16 = const()[name = string("const_12_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339298688)))]; tensor var_900_cast_fp16 = mul(x = var_897_cast_fp16_0, y = const_12_to_fp16)[name = string("op_900_cast_fp16")]; tensor hidden_states_17 = add(x = hidden_states_13_cast_fp16, y = var_900_cast_fp16)[name = string("hidden_states_17")]; tensor layers_0_layer_scalar_to_fp16 = const()[name = string("layers_0_layer_scalar_to_fp16"), val = tensor([0x1.cp-2])]; tensor x_23_cast_fp16 = mul(x = hidden_states_17, y = layers_0_layer_scalar_to_fp16)[name = string("x_23_cast_fp16")]; int32 var_908 = const()[name = string("op_908"), val = int32(-1)]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_914_cast_fp16 = mul(x = x_23_cast_fp16, y = const_13_promoted_to_fp16)[name = string("op_914_cast_fp16")]; bool input_25_interleave_0 = const()[name = string("input_25_interleave_0"), val = bool(false)]; tensor input_25_cast_fp16 = concat(axis = var_908, interleave = input_25_interleave_0, values = (x_23_cast_fp16, var_914_cast_fp16))[name = string("input_25_cast_fp16")]; tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; fp16 var_906_to_fp16 = const()[name = string("op_906_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_906_to_fp16, x = input_25_cast_fp16)[name = string("normed_25_cast_fp16")]; tensor var_919_split_sizes_0 = const()[name = string("op_919_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_919_axis_0 = const()[name = string("op_919_axis_0"), val = int32(-1)]; tensor var_919_cast_fp16_0, tensor var_919_cast_fp16_1 = split(axis = var_919_axis_0, split_sizes = var_919_split_sizes_0, x = normed_25_cast_fp16)[name = string("op_919_cast_fp16")]; tensor const_14_to_fp16 = const()[name = string("const_14_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339301824)))]; tensor var_922_cast_fp16 = mul(x = var_919_cast_fp16_0, y = const_14_to_fp16)[name = string("op_922_cast_fp16")]; tensor var_930 = const()[name = string("op_930"), val = tensor([0, 2, 1])]; tensor var_933_axes_0 = const()[name = string("op_933_axes_0"), val = tensor([2])]; tensor var_931_cast_fp16 = transpose(perm = var_930, x = var_922_cast_fp16)[name = string("transpose_77")]; tensor var_933_cast_fp16 = expand_dims(axes = var_933_axes_0, x = var_931_cast_fp16)[name = string("op_933_cast_fp16")]; string var_949_pad_type_0 = const()[name = string("op_949_pad_type_0"), val = string("valid")]; tensor var_949_strides_0 = const()[name = string("op_949_strides_0"), val = tensor([1, 1])]; tensor var_949_pad_0 = const()[name = string("op_949_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_949_dilations_0 = const()[name = string("op_949_dilations_0"), val = tensor([1, 1])]; int32 var_949_groups_0 = const()[name = string("op_949_groups_0"), val = int32(1)]; tensor var_949 = conv(dilations = var_949_dilations_0, groups = var_949_groups_0, pad = var_949_pad_0, pad_type = var_949_pad_type_0, strides = var_949_strides_0, weight = layers_1_self_attn_q_proj_weight_palettized, x = var_933_cast_fp16)[name = string("op_949")]; tensor var_954 = const()[name = string("op_954"), val = tensor([1, 8, 256, 1])]; tensor var_955 = reshape(shape = var_954, x = var_949)[name = string("op_955")]; tensor var_960 = const()[name = string("op_960"), val = tensor([0, 1, 3, 2])]; tensor var_970 = const()[name = string("op_970"), val = tensor([1, 8, 256])]; tensor var_961 = transpose(perm = var_960, x = var_955)[name = string("transpose_76")]; tensor x_27 = reshape(shape = var_970, x = var_961)[name = string("x_27")]; int32 var_976 = const()[name = string("op_976"), val = int32(-1)]; fp16 const_15_promoted_to_fp16 = const()[name = string("const_15_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_982_cast_fp16 = mul(x = x_27, y = const_15_promoted_to_fp16)[name = string("op_982_cast_fp16")]; bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; tensor input_29_cast_fp16 = concat(axis = var_976, interleave = input_29_interleave_0, values = (x_27, var_982_cast_fp16))[name = string("input_29_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; fp16 var_974_to_fp16 = const()[name = string("op_974_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_974_to_fp16, x = input_29_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor var_987_split_sizes_0 = const()[name = string("op_987_split_sizes_0"), val = tensor([256, 256])]; int32 var_987_axis_0 = const()[name = string("op_987_axis_0"), val = int32(-1)]; tensor var_987_cast_fp16_0, tensor var_987_cast_fp16_1 = split(axis = var_987_axis_0, split_sizes = var_987_split_sizes_0, x = normed_29_cast_fp16)[name = string("op_987_cast_fp16")]; tensor const_16_to_fp16 = const()[name = string("const_16_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339304960)))]; tensor var_990_cast_fp16 = mul(x = var_987_cast_fp16_0, y = const_16_to_fp16)[name = string("op_990_cast_fp16")]; tensor var_996 = const()[name = string("op_996"), val = tensor([1, 8, 1, 256])]; tensor q_9 = reshape(shape = var_996, x = var_990_cast_fp16)[name = string("q_9")]; tensor var_998_cast_fp16 = mul(x = q_9, y = cos_s)[name = string("op_998_cast_fp16")]; tensor var_999_split_sizes_0 = const()[name = string("op_999_split_sizes_0"), val = tensor([128, 128])]; int32 var_999_axis_0 = const()[name = string("op_999_axis_0"), val = int32(-1)]; tensor var_999_0, tensor var_999_1 = split(axis = var_999_axis_0, split_sizes = var_999_split_sizes_0, x = q_9)[name = string("op_999")]; fp16 const_17_promoted = const()[name = string("const_17_promoted"), val = fp16(-0x1p+0)]; tensor var_1001 = mul(x = var_999_1, y = const_17_promoted)[name = string("op_1001")]; int32 var_1003 = const()[name = string("op_1003"), val = int32(-1)]; bool var_1004_interleave_0 = const()[name = string("op_1004_interleave_0"), val = bool(false)]; tensor var_1004 = concat(axis = var_1003, interleave = var_1004_interleave_0, values = (var_1001, var_999_0))[name = string("op_1004")]; tensor var_1005_cast_fp16 = mul(x = var_1004, y = sin_s)[name = string("op_1005_cast_fp16")]; tensor q_11_cast_fp16 = add(x = var_998_cast_fp16, y = var_1005_cast_fp16)[name = string("q_11_cast_fp16")]; tensor transpose_4_perm_0 = const()[name = string("transpose_4_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_2_reps_0 = const()[name = string("tile_2_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_4_cast_fp16 = transpose(perm = transpose_4_perm_0, x = kv13_k)[name = string("transpose_75")]; tensor tile_2_cast_fp16 = tile(reps = tile_2_reps_0, x = transpose_4_cast_fp16)[name = string("tile_2_cast_fp16")]; tensor concat_4 = const()[name = string("concat_4"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_4_cast_fp16 = reshape(shape = concat_4, x = tile_2_cast_fp16)[name = string("reshape_4_cast_fp16")]; tensor transpose_5_perm_0 = const()[name = string("transpose_5_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_5 = const()[name = string("concat_5"), val = tensor([-1, 1, 512, 256])]; tensor transpose_5_cast_fp16 = transpose(perm = transpose_5_perm_0, x = reshape_4_cast_fp16)[name = string("transpose_74")]; tensor reshape_5_cast_fp16 = reshape(shape = concat_5, x = transpose_5_cast_fp16)[name = string("reshape_5_cast_fp16")]; tensor transpose_45_perm_0 = const()[name = string("transpose_45_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_6_perm_0 = const()[name = string("transpose_6_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_3_reps_0 = const()[name = string("tile_3_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_6_cast_fp16 = transpose(perm = transpose_6_perm_0, x = kv13_v)[name = string("transpose_73")]; tensor tile_3_cast_fp16 = tile(reps = tile_3_reps_0, x = transpose_6_cast_fp16)[name = string("tile_3_cast_fp16")]; tensor concat_6 = const()[name = string("concat_6"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_6_cast_fp16 = reshape(shape = concat_6, x = tile_3_cast_fp16)[name = string("reshape_6_cast_fp16")]; tensor transpose_7_perm_0 = const()[name = string("transpose_7_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_7 = const()[name = string("concat_7"), val = tensor([-1, 1, 512, 256])]; tensor transpose_7_cast_fp16 = transpose(perm = transpose_7_perm_0, x = reshape_6_cast_fp16)[name = string("transpose_72")]; tensor reshape_7_cast_fp16 = reshape(shape = concat_7, x = transpose_7_cast_fp16)[name = string("reshape_7_cast_fp16")]; tensor Ve_3_perm_0 = const()[name = string("Ve_3_perm_0"), val = tensor([1, 0, -2, -1])]; bool var_1029_transpose_x_0 = const()[name = string("op_1029_transpose_x_0"), val = bool(false)]; bool var_1029_transpose_y_0 = const()[name = string("op_1029_transpose_y_0"), val = bool(false)]; tensor transpose_45_cast_fp16 = transpose(perm = transpose_45_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_71")]; tensor var_1029_cast_fp16 = matmul(transpose_x = var_1029_transpose_x_0, transpose_y = var_1029_transpose_y_0, x = q_11_cast_fp16, y = transpose_45_cast_fp16)[name = string("op_1029_cast_fp16")]; tensor var_1036_cast_fp16 = add(x = var_1029_cast_fp16, y = causal_mask)[name = string("op_1036_cast_fp16")]; int32 var_1037 = const()[name = string("op_1037"), val = int32(-1)]; tensor var_1039_cast_fp16 = softmax(axis = var_1037, x = var_1036_cast_fp16)[name = string("op_1039_cast_fp16")]; bool var_1055_transpose_x_0 = const()[name = string("op_1055_transpose_x_0"), val = bool(false)]; bool var_1055_transpose_y_0 = const()[name = string("op_1055_transpose_y_0"), val = bool(false)]; tensor Ve_3_cast_fp16 = transpose(perm = Ve_3_perm_0, x = reshape_7_cast_fp16)[name = string("transpose_70")]; tensor var_1055_cast_fp16 = matmul(transpose_x = var_1055_transpose_x_0, transpose_y = var_1055_transpose_y_0, x = var_1039_cast_fp16, y = Ve_3_cast_fp16)[name = string("op_1055_cast_fp16")]; tensor var_1065 = const()[name = string("op_1065"), val = tensor([0, 2, 1, 3])]; tensor var_1072 = const()[name = string("op_1072"), val = tensor([1, 1, -1])]; tensor var_1066 = transpose(perm = var_1065, x = var_1055_cast_fp16)[name = string("transpose_69")]; tensor var_1073 = reshape(shape = var_1072, x = var_1066)[name = string("op_1073")]; tensor var_1077 = const()[name = string("op_1077"), val = tensor([0, 2, 1])]; tensor squeeze_1_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339305536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(340878464))))[name = string("squeeze_1_palettized")]; string var_1093_pad_type_0 = const()[name = string("op_1093_pad_type_0"), val = string("valid")]; int32 var_1093_groups_0 = const()[name = string("op_1093_groups_0"), val = int32(1)]; tensor var_1093_strides_0 = const()[name = string("op_1093_strides_0"), val = tensor([1])]; tensor var_1093_pad_0 = const()[name = string("op_1093_pad_0"), val = tensor([0, 0])]; tensor var_1093_dilations_0 = const()[name = string("op_1093_dilations_0"), val = tensor([1])]; tensor var_1078 = transpose(perm = var_1077, x = var_1073)[name = string("transpose_68")]; tensor var_1093 = conv(dilations = var_1093_dilations_0, groups = var_1093_groups_0, pad = var_1093_pad_0, pad_type = var_1093_pad_type_0, strides = var_1093_strides_0, weight = squeeze_1_palettized, x = var_1078)[name = string("op_1093")]; tensor var_1097 = const()[name = string("op_1097"), val = tensor([0, 2, 1])]; int32 var_1103 = const()[name = string("op_1103"), val = int32(-1)]; fp16 const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_31 = transpose(perm = var_1097, x = var_1093)[name = string("transpose_67")]; tensor var_1109_cast_fp16 = mul(x = x_31, y = const_18_promoted_to_fp16)[name = string("op_1109_cast_fp16")]; bool input_33_interleave_0 = const()[name = string("input_33_interleave_0"), val = bool(false)]; tensor input_33_cast_fp16 = concat(axis = var_1103, interleave = input_33_interleave_0, values = (x_31, var_1109_cast_fp16))[name = string("input_33_cast_fp16")]; tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; fp16 var_1101_to_fp16 = const()[name = string("op_1101_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_1101_to_fp16, x = input_33_cast_fp16)[name = string("normed_33_cast_fp16")]; tensor var_1114_split_sizes_0 = const()[name = string("op_1114_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1114_axis_0 = const()[name = string("op_1114_axis_0"), val = int32(-1)]; tensor var_1114_cast_fp16_0, tensor var_1114_cast_fp16_1 = split(axis = var_1114_axis_0, split_sizes = var_1114_split_sizes_0, x = normed_33_cast_fp16)[name = string("op_1114_cast_fp16")]; tensor const_19_to_fp16 = const()[name = string("const_19_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(340880064)))]; tensor var_1117_cast_fp16 = mul(x = var_1114_cast_fp16_0, y = const_19_to_fp16)[name = string("op_1117_cast_fp16")]; tensor x_35_cast_fp16 = add(x = x_23_cast_fp16, y = var_1117_cast_fp16)[name = string("x_35_cast_fp16")]; int32 var_1124 = const()[name = string("op_1124"), val = int32(-1)]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1130_cast_fp16 = mul(x = x_35_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1130_cast_fp16")]; bool input_35_interleave_0 = const()[name = string("input_35_interleave_0"), val = bool(false)]; tensor input_35_cast_fp16 = concat(axis = var_1124, interleave = input_35_interleave_0, values = (x_35_cast_fp16, var_1130_cast_fp16))[name = string("input_35_cast_fp16")]; tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; fp16 var_1122_to_fp16 = const()[name = string("op_1122_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_1122_to_fp16, x = input_35_cast_fp16)[name = string("normed_37_cast_fp16")]; tensor var_1135_split_sizes_0 = const()[name = string("op_1135_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1135_axis_0 = const()[name = string("op_1135_axis_0"), val = int32(-1)]; tensor var_1135_cast_fp16_0, tensor var_1135_cast_fp16_1 = split(axis = var_1135_axis_0, split_sizes = var_1135_split_sizes_0, x = normed_37_cast_fp16)[name = string("op_1135_cast_fp16")]; tensor const_21_to_fp16 = const()[name = string("const_21_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(340883200)))]; tensor var_1138_cast_fp16 = mul(x = var_1135_cast_fp16_0, y = const_21_to_fp16)[name = string("op_1138_cast_fp16")]; tensor var_1151 = const()[name = string("op_1151"), val = tensor([0, 2, 1])]; tensor input_37_axes_0 = const()[name = string("input_37_axes_0"), val = tensor([2])]; tensor var_1152 = transpose(perm = var_1151, x = var_1138_cast_fp16)[name = string("transpose_66")]; tensor input_37 = expand_dims(axes = input_37_axes_0, x = var_1152)[name = string("input_37")]; string var_1165_pad_type_0 = const()[name = string("op_1165_pad_type_0"), val = string("valid")]; tensor var_1165_strides_0 = const()[name = string("op_1165_strides_0"), val = tensor([1, 1])]; tensor var_1165_pad_0 = const()[name = string("op_1165_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1165_dilations_0 = const()[name = string("op_1165_dilations_0"), val = tensor([1, 1])]; int32 var_1165_groups_0 = const()[name = string("op_1165_groups_0"), val = int32(1)]; tensor var_1165 = conv(dilations = var_1165_dilations_0, groups = var_1165_groups_0, pad = var_1165_pad_0, pad_type = var_1165_pad_type_0, strides = var_1165_strides_0, weight = layers_1_mlp_gate_proj_weight_palettized, x = input_37)[name = string("op_1165")]; string var_1167_mode_0 = const()[name = string("op_1167_mode_0"), val = string("TANH_APPROXIMATION")]; tensor var_1167 = gelu(mode = var_1167_mode_0, x = var_1165)[name = string("op_1167")]; string var_1178_pad_type_0 = const()[name = string("op_1178_pad_type_0"), val = string("valid")]; tensor var_1178_strides_0 = const()[name = string("op_1178_strides_0"), val = tensor([1, 1])]; tensor var_1178_pad_0 = const()[name = string("op_1178_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1178_dilations_0 = const()[name = string("op_1178_dilations_0"), val = tensor([1, 1])]; int32 var_1178_groups_0 = const()[name = string("op_1178_groups_0"), val = int32(1)]; tensor var_1178 = conv(dilations = var_1178_dilations_0, groups = var_1178_groups_0, pad = var_1178_pad_0, pad_type = var_1178_pad_type_0, strides = var_1178_strides_0, weight = layers_1_mlp_up_proj_weight_palettized, x = input_37)[name = string("op_1178")]; tensor input_39 = mul(x = var_1167, y = var_1178)[name = string("input_39")]; string var_1190_pad_type_0 = const()[name = string("op_1190_pad_type_0"), val = string("valid")]; tensor var_1190_strides_0 = const()[name = string("op_1190_strides_0"), val = tensor([1, 1])]; tensor var_1190_pad_0 = const()[name = string("op_1190_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1190_dilations_0 = const()[name = string("op_1190_dilations_0"), val = tensor([1, 1])]; int32 var_1190_groups_0 = const()[name = string("op_1190_groups_0"), val = int32(1)]; tensor var_1190 = conv(dilations = var_1190_dilations_0, groups = var_1190_groups_0, pad = var_1190_pad_0, pad_type = var_1190_pad_type_0, strides = var_1190_strides_0, weight = layers_1_mlp_down_proj_weight_palettized, x = input_39)[name = string("op_1190")]; tensor var_1192_axes_0 = const()[name = string("op_1192_axes_0"), val = tensor([2])]; tensor var_1192 = squeeze(axes = var_1192_axes_0, x = var_1190)[name = string("op_1192")]; tensor var_1196 = const()[name = string("op_1196"), val = tensor([0, 2, 1])]; int32 var_1202 = const()[name = string("op_1202"), val = int32(-1)]; fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_39 = transpose(perm = var_1196, x = var_1192)[name = string("transpose_65")]; tensor var_1208_cast_fp16 = mul(x = x_39, y = const_22_promoted_to_fp16)[name = string("op_1208_cast_fp16")]; bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; tensor input_41_cast_fp16 = concat(axis = var_1202, interleave = input_41_interleave_0, values = (x_39, var_1208_cast_fp16))[name = string("input_41_cast_fp16")]; tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; fp16 var_1200_to_fp16 = const()[name = string("op_1200_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_1200_to_fp16, x = input_41_cast_fp16)[name = string("normed_41_cast_fp16")]; tensor var_1213_split_sizes_0 = const()[name = string("op_1213_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1213_axis_0 = const()[name = string("op_1213_axis_0"), val = int32(-1)]; tensor var_1213_cast_fp16_0, tensor var_1213_cast_fp16_1 = split(axis = var_1213_axis_0, split_sizes = var_1213_split_sizes_0, x = normed_41_cast_fp16)[name = string("op_1213_cast_fp16")]; tensor const_23_to_fp16 = const()[name = string("const_23_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(340886336)))]; tensor var_1216_cast_fp16 = mul(x = var_1213_cast_fp16_0, y = const_23_to_fp16)[name = string("op_1216_cast_fp16")]; tensor hidden_states_27_cast_fp16 = add(x = x_35_cast_fp16, y = var_1216_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; tensor var_1227 = linear(bias = linear_0_bias_0, weight = layers_1_per_layer_input_gate_weight_palettized, x = hidden_states_27_cast_fp16)[name = string("linear_2")]; string gated_3_mode_0 = const()[name = string("gated_3_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_3 = gelu(mode = gated_3_mode_0, x = var_1227)[name = string("gated_3")]; tensor var_1244_begin_0 = const()[name = string("op_1244_begin_0"), val = tensor([0, 0, 6400])]; tensor var_1244_end_0 = const()[name = string("op_1244_end_0"), val = tensor([1, 1, 6656])]; tensor var_1244_end_mask_0 = const()[name = string("op_1244_end_mask_0"), val = tensor([true, true, false])]; tensor var_1244_cast_fp16 = slice_by_index(begin = var_1244_begin_0, end = var_1244_end_0, end_mask = var_1244_end_mask_0, x = per_layer_combined)[name = string("op_1244_cast_fp16")]; tensor input_45_cast_fp16 = mul(x = gated_3, y = var_1244_cast_fp16)[name = string("input_45_cast_fp16")]; tensor layers_1_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(340889472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(341086144))))[name = string("layers_1_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_3_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_1_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_45_cast_fp16)[name = string("linear_3_cast_fp16")]; int32 var_1253 = const()[name = string("op_1253"), val = int32(-1)]; fp16 const_24_promoted_to_fp16 = const()[name = string("const_24_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1259_cast_fp16 = mul(x = linear_3_cast_fp16, y = const_24_promoted_to_fp16)[name = string("op_1259_cast_fp16")]; bool input_47_interleave_0 = const()[name = string("input_47_interleave_0"), val = bool(false)]; tensor input_47_cast_fp16 = concat(axis = var_1253, interleave = input_47_interleave_0, values = (linear_3_cast_fp16, var_1259_cast_fp16))[name = string("input_47_cast_fp16")]; tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; fp16 var_1251_to_fp16 = const()[name = string("op_1251_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_1251_to_fp16, x = input_47_cast_fp16)[name = string("normed_45_cast_fp16")]; tensor var_1264_split_sizes_0 = const()[name = string("op_1264_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1264_axis_0 = const()[name = string("op_1264_axis_0"), val = int32(-1)]; tensor var_1264_cast_fp16_0, tensor var_1264_cast_fp16_1 = split(axis = var_1264_axis_0, split_sizes = var_1264_split_sizes_0, x = normed_45_cast_fp16)[name = string("op_1264_cast_fp16")]; tensor const_25_to_fp16 = const()[name = string("const_25_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(341087744)))]; tensor var_1267_cast_fp16 = mul(x = var_1264_cast_fp16_0, y = const_25_to_fp16)[name = string("op_1267_cast_fp16")]; tensor hidden_states_31_cast_fp16 = add(x = hidden_states_27_cast_fp16, y = var_1267_cast_fp16)[name = string("hidden_states_31_cast_fp16")]; tensor layers_1_layer_scalar_to_fp16 = const()[name = string("layers_1_layer_scalar_to_fp16"), val = tensor([0x1.92p-1])]; tensor x_47_cast_fp16 = mul(x = hidden_states_31_cast_fp16, y = layers_1_layer_scalar_to_fp16)[name = string("x_47_cast_fp16")]; int32 var_1275 = const()[name = string("op_1275"), val = int32(-1)]; fp16 const_26_promoted_to_fp16 = const()[name = string("const_26_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1281_cast_fp16 = mul(x = x_47_cast_fp16, y = const_26_promoted_to_fp16)[name = string("op_1281_cast_fp16")]; bool input_49_interleave_0 = const()[name = string("input_49_interleave_0"), val = bool(false)]; tensor input_49_cast_fp16 = concat(axis = var_1275, interleave = input_49_interleave_0, values = (x_47_cast_fp16, var_1281_cast_fp16))[name = string("input_49_cast_fp16")]; tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; fp16 var_1273_to_fp16 = const()[name = string("op_1273_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_1273_to_fp16, x = input_49_cast_fp16)[name = string("normed_49_cast_fp16")]; tensor var_1286_split_sizes_0 = const()[name = string("op_1286_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1286_axis_0 = const()[name = string("op_1286_axis_0"), val = int32(-1)]; tensor var_1286_cast_fp16_0, tensor var_1286_cast_fp16_1 = split(axis = var_1286_axis_0, split_sizes = var_1286_split_sizes_0, x = normed_49_cast_fp16)[name = string("op_1286_cast_fp16")]; tensor const_27_to_fp16 = const()[name = string("const_27_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(341090880)))]; tensor var_1289_cast_fp16 = mul(x = var_1286_cast_fp16_0, y = const_27_to_fp16)[name = string("op_1289_cast_fp16")]; tensor var_1297 = const()[name = string("op_1297"), val = tensor([0, 2, 1])]; tensor var_1300_axes_0 = const()[name = string("op_1300_axes_0"), val = tensor([2])]; tensor var_1298_cast_fp16 = transpose(perm = var_1297, x = var_1289_cast_fp16)[name = string("transpose_64")]; tensor var_1300_cast_fp16 = expand_dims(axes = var_1300_axes_0, x = var_1298_cast_fp16)[name = string("op_1300_cast_fp16")]; string var_1316_pad_type_0 = const()[name = string("op_1316_pad_type_0"), val = string("valid")]; tensor var_1316_strides_0 = const()[name = string("op_1316_strides_0"), val = tensor([1, 1])]; tensor var_1316_pad_0 = const()[name = string("op_1316_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1316_dilations_0 = const()[name = string("op_1316_dilations_0"), val = tensor([1, 1])]; int32 var_1316_groups_0 = const()[name = string("op_1316_groups_0"), val = int32(1)]; tensor var_1316 = conv(dilations = var_1316_dilations_0, groups = var_1316_groups_0, pad = var_1316_pad_0, pad_type = var_1316_pad_type_0, strides = var_1316_strides_0, weight = layers_2_self_attn_q_proj_weight_palettized, x = var_1300_cast_fp16)[name = string("op_1316")]; tensor var_1321 = const()[name = string("op_1321"), val = tensor([1, 8, 256, 1])]; tensor var_1322 = reshape(shape = var_1321, x = var_1316)[name = string("op_1322")]; tensor var_1327 = const()[name = string("op_1327"), val = tensor([0, 1, 3, 2])]; tensor var_1337 = const()[name = string("op_1337"), val = tensor([1, 8, 256])]; tensor var_1328 = transpose(perm = var_1327, x = var_1322)[name = string("transpose_63")]; tensor x_51 = reshape(shape = var_1337, x = var_1328)[name = string("x_51")]; int32 var_1343 = const()[name = string("op_1343"), val = int32(-1)]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1349_cast_fp16 = mul(x = x_51, y = const_28_promoted_to_fp16)[name = string("op_1349_cast_fp16")]; bool input_53_interleave_0 = const()[name = string("input_53_interleave_0"), val = bool(false)]; tensor input_53_cast_fp16 = concat(axis = var_1343, interleave = input_53_interleave_0, values = (x_51, var_1349_cast_fp16))[name = string("input_53_cast_fp16")]; tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; fp16 var_1341_to_fp16 = const()[name = string("op_1341_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_1341_to_fp16, x = input_53_cast_fp16)[name = string("normed_53_cast_fp16")]; tensor var_1354_split_sizes_0 = const()[name = string("op_1354_split_sizes_0"), val = tensor([256, 256])]; int32 var_1354_axis_0 = const()[name = string("op_1354_axis_0"), val = int32(-1)]; tensor var_1354_cast_fp16_0, tensor var_1354_cast_fp16_1 = split(axis = var_1354_axis_0, split_sizes = var_1354_split_sizes_0, x = normed_53_cast_fp16)[name = string("op_1354_cast_fp16")]; tensor var_1357_cast_fp16 = mul(x = var_1354_cast_fp16_0, y = const_16_to_fp16)[name = string("op_1357_cast_fp16")]; tensor var_1363 = const()[name = string("op_1363"), val = tensor([1, 8, 1, 256])]; tensor q_15 = reshape(shape = var_1363, x = var_1357_cast_fp16)[name = string("q_15")]; tensor var_1365_cast_fp16 = mul(x = q_15, y = cos_s)[name = string("op_1365_cast_fp16")]; tensor var_1366_split_sizes_0 = const()[name = string("op_1366_split_sizes_0"), val = tensor([128, 128])]; int32 var_1366_axis_0 = const()[name = string("op_1366_axis_0"), val = int32(-1)]; tensor var_1366_0, tensor var_1366_1 = split(axis = var_1366_axis_0, split_sizes = var_1366_split_sizes_0, x = q_15)[name = string("op_1366")]; fp16 const_30_promoted = const()[name = string("const_30_promoted"), val = fp16(-0x1p+0)]; tensor var_1368 = mul(x = var_1366_1, y = const_30_promoted)[name = string("op_1368")]; int32 var_1370 = const()[name = string("op_1370"), val = int32(-1)]; bool var_1371_interleave_0 = const()[name = string("op_1371_interleave_0"), val = bool(false)]; tensor var_1371 = concat(axis = var_1370, interleave = var_1371_interleave_0, values = (var_1368, var_1366_0))[name = string("op_1371")]; tensor var_1372_cast_fp16 = mul(x = var_1371, y = sin_s)[name = string("op_1372_cast_fp16")]; tensor q_17_cast_fp16 = add(x = var_1365_cast_fp16, y = var_1372_cast_fp16)[name = string("q_17_cast_fp16")]; bool var_1396_transpose_x_0 = const()[name = string("op_1396_transpose_x_0"), val = bool(false)]; bool var_1396_transpose_y_0 = const()[name = string("op_1396_transpose_y_0"), val = bool(false)]; tensor var_1396_cast_fp16 = matmul(transpose_x = var_1396_transpose_x_0, transpose_y = var_1396_transpose_y_0, x = q_17_cast_fp16, y = transpose_45_cast_fp16)[name = string("op_1396_cast_fp16")]; tensor var_1403_cast_fp16 = add(x = var_1396_cast_fp16, y = causal_mask)[name = string("op_1403_cast_fp16")]; int32 var_1404 = const()[name = string("op_1404"), val = int32(-1)]; tensor var_1406_cast_fp16 = softmax(axis = var_1404, x = var_1403_cast_fp16)[name = string("op_1406_cast_fp16")]; bool var_1422_transpose_x_0 = const()[name = string("op_1422_transpose_x_0"), val = bool(false)]; bool var_1422_transpose_y_0 = const()[name = string("op_1422_transpose_y_0"), val = bool(false)]; tensor var_1422_cast_fp16 = matmul(transpose_x = var_1422_transpose_x_0, transpose_y = var_1422_transpose_y_0, x = var_1406_cast_fp16, y = Ve_3_cast_fp16)[name = string("op_1422_cast_fp16")]; tensor var_1432 = const()[name = string("op_1432"), val = tensor([0, 2, 1, 3])]; tensor var_1439 = const()[name = string("op_1439"), val = tensor([1, 1, -1])]; tensor var_1433 = transpose(perm = var_1432, x = var_1422_cast_fp16)[name = string("transpose_62")]; tensor var_1440 = reshape(shape = var_1439, x = var_1433)[name = string("op_1440")]; tensor var_1444 = const()[name = string("op_1444"), val = tensor([0, 2, 1])]; tensor squeeze_2_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(341094016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342666944))))[name = string("squeeze_2_palettized")]; string var_1460_pad_type_0 = const()[name = string("op_1460_pad_type_0"), val = string("valid")]; int32 var_1460_groups_0 = const()[name = string("op_1460_groups_0"), val = int32(1)]; tensor var_1460_strides_0 = const()[name = string("op_1460_strides_0"), val = tensor([1])]; tensor var_1460_pad_0 = const()[name = string("op_1460_pad_0"), val = tensor([0, 0])]; tensor var_1460_dilations_0 = const()[name = string("op_1460_dilations_0"), val = tensor([1])]; tensor var_1445 = transpose(perm = var_1444, x = var_1440)[name = string("transpose_61")]; tensor var_1460 = conv(dilations = var_1460_dilations_0, groups = var_1460_groups_0, pad = var_1460_pad_0, pad_type = var_1460_pad_type_0, strides = var_1460_strides_0, weight = squeeze_2_palettized, x = var_1445)[name = string("op_1460")]; tensor var_1464 = const()[name = string("op_1464"), val = tensor([0, 2, 1])]; int32 var_1470 = const()[name = string("op_1470"), val = int32(-1)]; fp16 const_31_promoted_to_fp16 = const()[name = string("const_31_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_55 = transpose(perm = var_1464, x = var_1460)[name = string("transpose_60")]; tensor var_1476_cast_fp16 = mul(x = x_55, y = const_31_promoted_to_fp16)[name = string("op_1476_cast_fp16")]; bool input_57_interleave_0 = const()[name = string("input_57_interleave_0"), val = bool(false)]; tensor input_57_cast_fp16 = concat(axis = var_1470, interleave = input_57_interleave_0, values = (x_55, var_1476_cast_fp16))[name = string("input_57_cast_fp16")]; tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; fp16 var_1468_to_fp16 = const()[name = string("op_1468_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_1468_to_fp16, x = input_57_cast_fp16)[name = string("normed_57_cast_fp16")]; tensor var_1481_split_sizes_0 = const()[name = string("op_1481_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1481_axis_0 = const()[name = string("op_1481_axis_0"), val = int32(-1)]; tensor var_1481_cast_fp16_0, tensor var_1481_cast_fp16_1 = split(axis = var_1481_axis_0, split_sizes = var_1481_split_sizes_0, x = normed_57_cast_fp16)[name = string("op_1481_cast_fp16")]; tensor const_32_to_fp16 = const()[name = string("const_32_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342668544)))]; tensor var_1484_cast_fp16 = mul(x = var_1481_cast_fp16_0, y = const_32_to_fp16)[name = string("op_1484_cast_fp16")]; tensor x_59_cast_fp16 = add(x = x_47_cast_fp16, y = var_1484_cast_fp16)[name = string("x_59_cast_fp16")]; int32 var_1491 = const()[name = string("op_1491"), val = int32(-1)]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1497_cast_fp16 = mul(x = x_59_cast_fp16, y = const_33_promoted_to_fp16)[name = string("op_1497_cast_fp16")]; bool input_59_interleave_0 = const()[name = string("input_59_interleave_0"), val = bool(false)]; tensor input_59_cast_fp16 = concat(axis = var_1491, interleave = input_59_interleave_0, values = (x_59_cast_fp16, var_1497_cast_fp16))[name = string("input_59_cast_fp16")]; tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; fp16 var_1489_to_fp16 = const()[name = string("op_1489_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_1489_to_fp16, x = input_59_cast_fp16)[name = string("normed_61_cast_fp16")]; tensor var_1502_split_sizes_0 = const()[name = string("op_1502_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1502_axis_0 = const()[name = string("op_1502_axis_0"), val = int32(-1)]; tensor var_1502_cast_fp16_0, tensor var_1502_cast_fp16_1 = split(axis = var_1502_axis_0, split_sizes = var_1502_split_sizes_0, x = normed_61_cast_fp16)[name = string("op_1502_cast_fp16")]; tensor const_34_to_fp16 = const()[name = string("const_34_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342671680)))]; tensor var_1505_cast_fp16 = mul(x = var_1502_cast_fp16_0, y = const_34_to_fp16)[name = string("op_1505_cast_fp16")]; tensor var_1518 = const()[name = string("op_1518"), val = tensor([0, 2, 1])]; tensor input_61_axes_0 = const()[name = string("input_61_axes_0"), val = tensor([2])]; tensor var_1519 = transpose(perm = var_1518, x = var_1505_cast_fp16)[name = string("transpose_59")]; tensor input_61 = expand_dims(axes = input_61_axes_0, x = var_1519)[name = string("input_61")]; string var_1532_pad_type_0 = const()[name = string("op_1532_pad_type_0"), val = string("valid")]; tensor var_1532_strides_0 = const()[name = string("op_1532_strides_0"), val = tensor([1, 1])]; tensor var_1532_pad_0 = const()[name = string("op_1532_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1532_dilations_0 = const()[name = string("op_1532_dilations_0"), val = tensor([1, 1])]; int32 var_1532_groups_0 = const()[name = string("op_1532_groups_0"), val = int32(1)]; tensor var_1532 = conv(dilations = var_1532_dilations_0, groups = var_1532_groups_0, pad = var_1532_pad_0, pad_type = var_1532_pad_type_0, strides = var_1532_strides_0, weight = layers_2_mlp_gate_proj_weight_palettized, x = input_61)[name = string("op_1532")]; string var_1534_mode_0 = const()[name = string("op_1534_mode_0"), val = string("TANH_APPROXIMATION")]; tensor var_1534 = gelu(mode = var_1534_mode_0, x = var_1532)[name = string("op_1534")]; string var_1545_pad_type_0 = const()[name = string("op_1545_pad_type_0"), val = string("valid")]; tensor var_1545_strides_0 = const()[name = string("op_1545_strides_0"), val = tensor([1, 1])]; tensor var_1545_pad_0 = const()[name = string("op_1545_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1545_dilations_0 = const()[name = string("op_1545_dilations_0"), val = tensor([1, 1])]; int32 var_1545_groups_0 = const()[name = string("op_1545_groups_0"), val = int32(1)]; tensor var_1545 = conv(dilations = var_1545_dilations_0, groups = var_1545_groups_0, pad = var_1545_pad_0, pad_type = var_1545_pad_type_0, strides = var_1545_strides_0, weight = layers_2_mlp_up_proj_weight_palettized, x = input_61)[name = string("op_1545")]; tensor input_63 = mul(x = var_1534, y = var_1545)[name = string("input_63")]; string var_1557_pad_type_0 = const()[name = string("op_1557_pad_type_0"), val = string("valid")]; tensor var_1557_strides_0 = const()[name = string("op_1557_strides_0"), val = tensor([1, 1])]; tensor var_1557_pad_0 = const()[name = string("op_1557_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1557_dilations_0 = const()[name = string("op_1557_dilations_0"), val = tensor([1, 1])]; int32 var_1557_groups_0 = const()[name = string("op_1557_groups_0"), val = int32(1)]; tensor var_1557 = conv(dilations = var_1557_dilations_0, groups = var_1557_groups_0, pad = var_1557_pad_0, pad_type = var_1557_pad_type_0, strides = var_1557_strides_0, weight = layers_2_mlp_down_proj_weight_palettized, x = input_63)[name = string("op_1557")]; tensor var_1559_axes_0 = const()[name = string("op_1559_axes_0"), val = tensor([2])]; tensor var_1559 = squeeze(axes = var_1559_axes_0, x = var_1557)[name = string("op_1559")]; tensor var_1563 = const()[name = string("op_1563"), val = tensor([0, 2, 1])]; int32 var_1569 = const()[name = string("op_1569"), val = int32(-1)]; fp16 const_35_promoted_to_fp16 = const()[name = string("const_35_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_63 = transpose(perm = var_1563, x = var_1559)[name = string("transpose_58")]; tensor var_1575_cast_fp16 = mul(x = x_63, y = const_35_promoted_to_fp16)[name = string("op_1575_cast_fp16")]; bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; tensor input_65_cast_fp16 = concat(axis = var_1569, interleave = input_65_interleave_0, values = (x_63, var_1575_cast_fp16))[name = string("input_65_cast_fp16")]; tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; fp16 var_1567_to_fp16 = const()[name = string("op_1567_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_1567_to_fp16, x = input_65_cast_fp16)[name = string("normed_65_cast_fp16")]; tensor var_1580_split_sizes_0 = const()[name = string("op_1580_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1580_axis_0 = const()[name = string("op_1580_axis_0"), val = int32(-1)]; tensor var_1580_cast_fp16_0, tensor var_1580_cast_fp16_1 = split(axis = var_1580_axis_0, split_sizes = var_1580_split_sizes_0, x = normed_65_cast_fp16)[name = string("op_1580_cast_fp16")]; tensor const_36_to_fp16 = const()[name = string("const_36_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342674816)))]; tensor var_1583_cast_fp16 = mul(x = var_1580_cast_fp16_0, y = const_36_to_fp16)[name = string("op_1583_cast_fp16")]; tensor hidden_states_41_cast_fp16 = add(x = x_59_cast_fp16, y = var_1583_cast_fp16)[name = string("hidden_states_41_cast_fp16")]; tensor var_1594 = linear(bias = linear_0_bias_0, weight = layers_2_per_layer_input_gate_weight_palettized, x = hidden_states_41_cast_fp16)[name = string("linear_4")]; string gated_5_mode_0 = const()[name = string("gated_5_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_5 = gelu(mode = gated_5_mode_0, x = var_1594)[name = string("gated_5")]; tensor var_1611_begin_0 = const()[name = string("op_1611_begin_0"), val = tensor([0, 0, 6656])]; tensor var_1611_end_0 = const()[name = string("op_1611_end_0"), val = tensor([1, 1, 6912])]; tensor var_1611_end_mask_0 = const()[name = string("op_1611_end_mask_0"), val = tensor([true, true, false])]; tensor var_1611_cast_fp16 = slice_by_index(begin = var_1611_begin_0, end = var_1611_end_0, end_mask = var_1611_end_mask_0, x = per_layer_combined)[name = string("op_1611_cast_fp16")]; tensor input_69_cast_fp16 = mul(x = gated_5, y = var_1611_cast_fp16)[name = string("input_69_cast_fp16")]; tensor layers_2_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342677952))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342874624))))[name = string("layers_2_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_5_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_2_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_69_cast_fp16)[name = string("linear_5_cast_fp16")]; int32 var_1620 = const()[name = string("op_1620"), val = int32(-1)]; fp16 const_37_promoted_to_fp16 = const()[name = string("const_37_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1626_cast_fp16 = mul(x = linear_5_cast_fp16, y = const_37_promoted_to_fp16)[name = string("op_1626_cast_fp16")]; bool input_71_interleave_0 = const()[name = string("input_71_interleave_0"), val = bool(false)]; tensor input_71_cast_fp16 = concat(axis = var_1620, interleave = input_71_interleave_0, values = (linear_5_cast_fp16, var_1626_cast_fp16))[name = string("input_71_cast_fp16")]; tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; fp16 var_1618_to_fp16 = const()[name = string("op_1618_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_1618_to_fp16, x = input_71_cast_fp16)[name = string("normed_69_cast_fp16")]; tensor var_1631_split_sizes_0 = const()[name = string("op_1631_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1631_axis_0 = const()[name = string("op_1631_axis_0"), val = int32(-1)]; tensor var_1631_cast_fp16_0, tensor var_1631_cast_fp16_1 = split(axis = var_1631_axis_0, split_sizes = var_1631_split_sizes_0, x = normed_69_cast_fp16)[name = string("op_1631_cast_fp16")]; tensor const_38_to_fp16 = const()[name = string("const_38_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342876224)))]; tensor var_1634_cast_fp16 = mul(x = var_1631_cast_fp16_0, y = const_38_to_fp16)[name = string("op_1634_cast_fp16")]; tensor hidden_states_45_cast_fp16 = add(x = hidden_states_41_cast_fp16, y = var_1634_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; tensor layers_2_layer_scalar_to_fp16 = const()[name = string("layers_2_layer_scalar_to_fp16"), val = tensor([0x1.a6p-1])]; tensor x_71_cast_fp16 = mul(x = hidden_states_45_cast_fp16, y = layers_2_layer_scalar_to_fp16)[name = string("x_71_cast_fp16")]; int32 var_1642 = const()[name = string("op_1642"), val = int32(-1)]; fp16 const_39_promoted_to_fp16 = const()[name = string("const_39_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1648_cast_fp16 = mul(x = x_71_cast_fp16, y = const_39_promoted_to_fp16)[name = string("op_1648_cast_fp16")]; bool input_73_interleave_0 = const()[name = string("input_73_interleave_0"), val = bool(false)]; tensor input_73_cast_fp16 = concat(axis = var_1642, interleave = input_73_interleave_0, values = (x_71_cast_fp16, var_1648_cast_fp16))[name = string("input_73_cast_fp16")]; tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; fp16 var_1640_to_fp16 = const()[name = string("op_1640_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_1640_to_fp16, x = input_73_cast_fp16)[name = string("normed_73_cast_fp16")]; tensor var_1653_split_sizes_0 = const()[name = string("op_1653_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1653_axis_0 = const()[name = string("op_1653_axis_0"), val = int32(-1)]; tensor var_1653_cast_fp16_0, tensor var_1653_cast_fp16_1 = split(axis = var_1653_axis_0, split_sizes = var_1653_split_sizes_0, x = normed_73_cast_fp16)[name = string("op_1653_cast_fp16")]; tensor const_40_to_fp16 = const()[name = string("const_40_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342879360)))]; tensor var_1656_cast_fp16 = mul(x = var_1653_cast_fp16_0, y = const_40_to_fp16)[name = string("op_1656_cast_fp16")]; tensor var_1664 = const()[name = string("op_1664"), val = tensor([0, 2, 1])]; tensor var_1667_axes_0 = const()[name = string("op_1667_axes_0"), val = tensor([2])]; tensor var_1665_cast_fp16 = transpose(perm = var_1664, x = var_1656_cast_fp16)[name = string("transpose_57")]; tensor var_1667_cast_fp16 = expand_dims(axes = var_1667_axes_0, x = var_1665_cast_fp16)[name = string("op_1667_cast_fp16")]; string var_1683_pad_type_0 = const()[name = string("op_1683_pad_type_0"), val = string("valid")]; tensor var_1683_strides_0 = const()[name = string("op_1683_strides_0"), val = tensor([1, 1])]; tensor var_1683_pad_0 = const()[name = string("op_1683_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1683_dilations_0 = const()[name = string("op_1683_dilations_0"), val = tensor([1, 1])]; int32 var_1683_groups_0 = const()[name = string("op_1683_groups_0"), val = int32(1)]; tensor var_1683 = conv(dilations = var_1683_dilations_0, groups = var_1683_groups_0, pad = var_1683_pad_0, pad_type = var_1683_pad_type_0, strides = var_1683_strides_0, weight = layers_3_self_attn_q_proj_weight_palettized, x = var_1667_cast_fp16)[name = string("op_1683")]; tensor var_1688 = const()[name = string("op_1688"), val = tensor([1, 8, 256, 1])]; tensor var_1689 = reshape(shape = var_1688, x = var_1683)[name = string("op_1689")]; tensor var_1694 = const()[name = string("op_1694"), val = tensor([0, 1, 3, 2])]; tensor var_1704 = const()[name = string("op_1704"), val = tensor([1, 8, 256])]; tensor var_1695 = transpose(perm = var_1694, x = var_1689)[name = string("transpose_56")]; tensor x_75 = reshape(shape = var_1704, x = var_1695)[name = string("x_75")]; int32 var_1710 = const()[name = string("op_1710"), val = int32(-1)]; fp16 const_41_promoted_to_fp16 = const()[name = string("const_41_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1716_cast_fp16 = mul(x = x_75, y = const_41_promoted_to_fp16)[name = string("op_1716_cast_fp16")]; bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; tensor input_77_cast_fp16 = concat(axis = var_1710, interleave = input_77_interleave_0, values = (x_75, var_1716_cast_fp16))[name = string("input_77_cast_fp16")]; tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; fp16 var_1708_to_fp16 = const()[name = string("op_1708_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_1708_to_fp16, x = input_77_cast_fp16)[name = string("normed_77_cast_fp16")]; tensor var_1721_split_sizes_0 = const()[name = string("op_1721_split_sizes_0"), val = tensor([256, 256])]; int32 var_1721_axis_0 = const()[name = string("op_1721_axis_0"), val = int32(-1)]; tensor var_1721_cast_fp16_0, tensor var_1721_cast_fp16_1 = split(axis = var_1721_axis_0, split_sizes = var_1721_split_sizes_0, x = normed_77_cast_fp16)[name = string("op_1721_cast_fp16")]; tensor var_1724_cast_fp16 = mul(x = var_1721_cast_fp16_0, y = const_16_to_fp16)[name = string("op_1724_cast_fp16")]; tensor var_1730 = const()[name = string("op_1730"), val = tensor([1, 8, 1, 256])]; tensor q_21 = reshape(shape = var_1730, x = var_1724_cast_fp16)[name = string("q_21")]; tensor var_1732_cast_fp16 = mul(x = q_21, y = cos_s)[name = string("op_1732_cast_fp16")]; tensor var_1733_split_sizes_0 = const()[name = string("op_1733_split_sizes_0"), val = tensor([128, 128])]; int32 var_1733_axis_0 = const()[name = string("op_1733_axis_0"), val = int32(-1)]; tensor var_1733_0, tensor var_1733_1 = split(axis = var_1733_axis_0, split_sizes = var_1733_split_sizes_0, x = q_21)[name = string("op_1733")]; fp16 const_43_promoted = const()[name = string("const_43_promoted"), val = fp16(-0x1p+0)]; tensor var_1735 = mul(x = var_1733_1, y = const_43_promoted)[name = string("op_1735")]; int32 var_1737 = const()[name = string("op_1737"), val = int32(-1)]; bool var_1738_interleave_0 = const()[name = string("op_1738_interleave_0"), val = bool(false)]; tensor var_1738 = concat(axis = var_1737, interleave = var_1738_interleave_0, values = (var_1735, var_1733_0))[name = string("op_1738")]; tensor var_1739_cast_fp16 = mul(x = var_1738, y = sin_s)[name = string("op_1739_cast_fp16")]; tensor q_23_cast_fp16 = add(x = var_1732_cast_fp16, y = var_1739_cast_fp16)[name = string("q_23_cast_fp16")]; bool var_1763_transpose_x_0 = const()[name = string("op_1763_transpose_x_0"), val = bool(false)]; bool var_1763_transpose_y_0 = const()[name = string("op_1763_transpose_y_0"), val = bool(false)]; tensor var_1763_cast_fp16 = matmul(transpose_x = var_1763_transpose_x_0, transpose_y = var_1763_transpose_y_0, x = q_23_cast_fp16, y = transpose_45_cast_fp16)[name = string("op_1763_cast_fp16")]; tensor var_1770_cast_fp16 = add(x = var_1763_cast_fp16, y = causal_mask)[name = string("op_1770_cast_fp16")]; int32 var_1771 = const()[name = string("op_1771"), val = int32(-1)]; tensor var_1773_cast_fp16 = softmax(axis = var_1771, x = var_1770_cast_fp16)[name = string("op_1773_cast_fp16")]; bool var_1789_transpose_x_0 = const()[name = string("op_1789_transpose_x_0"), val = bool(false)]; bool var_1789_transpose_y_0 = const()[name = string("op_1789_transpose_y_0"), val = bool(false)]; tensor var_1789_cast_fp16 = matmul(transpose_x = var_1789_transpose_x_0, transpose_y = var_1789_transpose_y_0, x = var_1773_cast_fp16, y = Ve_3_cast_fp16)[name = string("op_1789_cast_fp16")]; tensor var_1799 = const()[name = string("op_1799"), val = tensor([0, 2, 1, 3])]; tensor var_1806 = const()[name = string("op_1806"), val = tensor([1, 1, -1])]; tensor var_1800 = transpose(perm = var_1799, x = var_1789_cast_fp16)[name = string("transpose_55")]; tensor var_1807 = reshape(shape = var_1806, x = var_1800)[name = string("op_1807")]; tensor var_1811 = const()[name = string("op_1811"), val = tensor([0, 2, 1])]; tensor squeeze_3_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342882496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(344455424))))[name = string("squeeze_3_palettized")]; string var_1827_pad_type_0 = const()[name = string("op_1827_pad_type_0"), val = string("valid")]; int32 var_1827_groups_0 = const()[name = string("op_1827_groups_0"), val = int32(1)]; tensor var_1827_strides_0 = const()[name = string("op_1827_strides_0"), val = tensor([1])]; tensor var_1827_pad_0 = const()[name = string("op_1827_pad_0"), val = tensor([0, 0])]; tensor var_1827_dilations_0 = const()[name = string("op_1827_dilations_0"), val = tensor([1])]; tensor var_1812 = transpose(perm = var_1811, x = var_1807)[name = string("transpose_54")]; tensor var_1827 = conv(dilations = var_1827_dilations_0, groups = var_1827_groups_0, pad = var_1827_pad_0, pad_type = var_1827_pad_type_0, strides = var_1827_strides_0, weight = squeeze_3_palettized, x = var_1812)[name = string("op_1827")]; tensor var_1831 = const()[name = string("op_1831"), val = tensor([0, 2, 1])]; int32 var_1837 = const()[name = string("op_1837"), val = int32(-1)]; fp16 const_44_promoted_to_fp16 = const()[name = string("const_44_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_79 = transpose(perm = var_1831, x = var_1827)[name = string("transpose_53")]; tensor var_1843_cast_fp16 = mul(x = x_79, y = const_44_promoted_to_fp16)[name = string("op_1843_cast_fp16")]; bool input_81_interleave_0 = const()[name = string("input_81_interleave_0"), val = bool(false)]; tensor input_81_cast_fp16 = concat(axis = var_1837, interleave = input_81_interleave_0, values = (x_79, var_1843_cast_fp16))[name = string("input_81_cast_fp16")]; tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; fp16 var_1835_to_fp16 = const()[name = string("op_1835_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_1835_to_fp16, x = input_81_cast_fp16)[name = string("normed_81_cast_fp16")]; tensor var_1848_split_sizes_0 = const()[name = string("op_1848_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1848_axis_0 = const()[name = string("op_1848_axis_0"), val = int32(-1)]; tensor var_1848_cast_fp16_0, tensor var_1848_cast_fp16_1 = split(axis = var_1848_axis_0, split_sizes = var_1848_split_sizes_0, x = normed_81_cast_fp16)[name = string("op_1848_cast_fp16")]; tensor const_45_to_fp16 = const()[name = string("const_45_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(344457024)))]; tensor var_1851_cast_fp16 = mul(x = var_1848_cast_fp16_0, y = const_45_to_fp16)[name = string("op_1851_cast_fp16")]; tensor x_83_cast_fp16 = add(x = x_71_cast_fp16, y = var_1851_cast_fp16)[name = string("x_83_cast_fp16")]; int32 var_1858 = const()[name = string("op_1858"), val = int32(-1)]; fp16 const_46_promoted_to_fp16 = const()[name = string("const_46_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1864_cast_fp16 = mul(x = x_83_cast_fp16, y = const_46_promoted_to_fp16)[name = string("op_1864_cast_fp16")]; bool input_83_interleave_0 = const()[name = string("input_83_interleave_0"), val = bool(false)]; tensor input_83_cast_fp16 = concat(axis = var_1858, interleave = input_83_interleave_0, values = (x_83_cast_fp16, var_1864_cast_fp16))[name = string("input_83_cast_fp16")]; tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; fp16 var_1856_to_fp16 = const()[name = string("op_1856_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_1856_to_fp16, x = input_83_cast_fp16)[name = string("normed_85_cast_fp16")]; tensor var_1869_split_sizes_0 = const()[name = string("op_1869_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1869_axis_0 = const()[name = string("op_1869_axis_0"), val = int32(-1)]; tensor var_1869_cast_fp16_0, tensor var_1869_cast_fp16_1 = split(axis = var_1869_axis_0, split_sizes = var_1869_split_sizes_0, x = normed_85_cast_fp16)[name = string("op_1869_cast_fp16")]; tensor const_47_to_fp16 = const()[name = string("const_47_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(344460160)))]; tensor var_1872_cast_fp16 = mul(x = var_1869_cast_fp16_0, y = const_47_to_fp16)[name = string("op_1872_cast_fp16")]; tensor var_1885 = const()[name = string("op_1885"), val = tensor([0, 2, 1])]; tensor input_85_axes_0 = const()[name = string("input_85_axes_0"), val = tensor([2])]; tensor var_1886 = transpose(perm = var_1885, x = var_1872_cast_fp16)[name = string("transpose_52")]; tensor input_85 = expand_dims(axes = input_85_axes_0, x = var_1886)[name = string("input_85")]; string var_1899_pad_type_0 = const()[name = string("op_1899_pad_type_0"), val = string("valid")]; tensor var_1899_strides_0 = const()[name = string("op_1899_strides_0"), val = tensor([1, 1])]; tensor var_1899_pad_0 = const()[name = string("op_1899_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1899_dilations_0 = const()[name = string("op_1899_dilations_0"), val = tensor([1, 1])]; int32 var_1899_groups_0 = const()[name = string("op_1899_groups_0"), val = int32(1)]; tensor var_1899 = conv(dilations = var_1899_dilations_0, groups = var_1899_groups_0, pad = var_1899_pad_0, pad_type = var_1899_pad_type_0, strides = var_1899_strides_0, weight = layers_3_mlp_gate_proj_weight_palettized, x = input_85)[name = string("op_1899")]; string var_1901_mode_0 = const()[name = string("op_1901_mode_0"), val = string("TANH_APPROXIMATION")]; tensor var_1901 = gelu(mode = var_1901_mode_0, x = var_1899)[name = string("op_1901")]; string var_1912_pad_type_0 = const()[name = string("op_1912_pad_type_0"), val = string("valid")]; tensor var_1912_strides_0 = const()[name = string("op_1912_strides_0"), val = tensor([1, 1])]; tensor var_1912_pad_0 = const()[name = string("op_1912_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1912_dilations_0 = const()[name = string("op_1912_dilations_0"), val = tensor([1, 1])]; int32 var_1912_groups_0 = const()[name = string("op_1912_groups_0"), val = int32(1)]; tensor var_1912 = conv(dilations = var_1912_dilations_0, groups = var_1912_groups_0, pad = var_1912_pad_0, pad_type = var_1912_pad_type_0, strides = var_1912_strides_0, weight = layers_3_mlp_up_proj_weight_palettized, x = input_85)[name = string("op_1912")]; tensor input_87 = mul(x = var_1901, y = var_1912)[name = string("input_87")]; string var_1924_pad_type_0 = const()[name = string("op_1924_pad_type_0"), val = string("valid")]; tensor var_1924_strides_0 = const()[name = string("op_1924_strides_0"), val = tensor([1, 1])]; tensor var_1924_pad_0 = const()[name = string("op_1924_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1924_dilations_0 = const()[name = string("op_1924_dilations_0"), val = tensor([1, 1])]; int32 var_1924_groups_0 = const()[name = string("op_1924_groups_0"), val = int32(1)]; tensor var_1924 = conv(dilations = var_1924_dilations_0, groups = var_1924_groups_0, pad = var_1924_pad_0, pad_type = var_1924_pad_type_0, strides = var_1924_strides_0, weight = layers_3_mlp_down_proj_weight_palettized, x = input_87)[name = string("op_1924")]; tensor var_1926_axes_0 = const()[name = string("op_1926_axes_0"), val = tensor([2])]; tensor var_1926 = squeeze(axes = var_1926_axes_0, x = var_1924)[name = string("op_1926")]; tensor var_1930 = const()[name = string("op_1930"), val = tensor([0, 2, 1])]; int32 var_1936 = const()[name = string("op_1936"), val = int32(-1)]; fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_87 = transpose(perm = var_1930, x = var_1926)[name = string("transpose_51")]; tensor var_1942_cast_fp16 = mul(x = x_87, y = const_48_promoted_to_fp16)[name = string("op_1942_cast_fp16")]; bool input_89_interleave_0 = const()[name = string("input_89_interleave_0"), val = bool(false)]; tensor input_89_cast_fp16 = concat(axis = var_1936, interleave = input_89_interleave_0, values = (x_87, var_1942_cast_fp16))[name = string("input_89_cast_fp16")]; tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; fp16 var_1934_to_fp16 = const()[name = string("op_1934_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_1934_to_fp16, x = input_89_cast_fp16)[name = string("normed_89_cast_fp16")]; tensor var_1947_split_sizes_0 = const()[name = string("op_1947_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1947_axis_0 = const()[name = string("op_1947_axis_0"), val = int32(-1)]; tensor var_1947_cast_fp16_0, tensor var_1947_cast_fp16_1 = split(axis = var_1947_axis_0, split_sizes = var_1947_split_sizes_0, x = normed_89_cast_fp16)[name = string("op_1947_cast_fp16")]; tensor const_49_to_fp16 = const()[name = string("const_49_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(344463296)))]; tensor var_1950_cast_fp16 = mul(x = var_1947_cast_fp16_0, y = const_49_to_fp16)[name = string("op_1950_cast_fp16")]; tensor hidden_states_55_cast_fp16 = add(x = x_83_cast_fp16, y = var_1950_cast_fp16)[name = string("hidden_states_55_cast_fp16")]; tensor var_1961 = linear(bias = linear_0_bias_0, weight = layers_3_per_layer_input_gate_weight_palettized, x = hidden_states_55_cast_fp16)[name = string("linear_6")]; string gated_7_mode_0 = const()[name = string("gated_7_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_7 = gelu(mode = gated_7_mode_0, x = var_1961)[name = string("gated_7")]; tensor var_1978_begin_0 = const()[name = string("op_1978_begin_0"), val = tensor([0, 0, 6912])]; tensor var_1978_end_0 = const()[name = string("op_1978_end_0"), val = tensor([1, 1, 7168])]; tensor var_1978_end_mask_0 = const()[name = string("op_1978_end_mask_0"), val = tensor([true, true, false])]; tensor var_1978_cast_fp16 = slice_by_index(begin = var_1978_begin_0, end = var_1978_end_0, end_mask = var_1978_end_mask_0, x = per_layer_combined)[name = string("op_1978_cast_fp16")]; tensor input_93_cast_fp16 = mul(x = gated_7, y = var_1978_cast_fp16)[name = string("input_93_cast_fp16")]; tensor layers_3_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(344466432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(344663104))))[name = string("layers_3_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_7_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_3_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_93_cast_fp16)[name = string("linear_7_cast_fp16")]; int32 var_1987 = const()[name = string("op_1987"), val = int32(-1)]; fp16 const_50_promoted_to_fp16 = const()[name = string("const_50_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1993_cast_fp16 = mul(x = linear_7_cast_fp16, y = const_50_promoted_to_fp16)[name = string("op_1993_cast_fp16")]; bool input_95_interleave_0 = const()[name = string("input_95_interleave_0"), val = bool(false)]; tensor input_95_cast_fp16 = concat(axis = var_1987, interleave = input_95_interleave_0, values = (linear_7_cast_fp16, var_1993_cast_fp16))[name = string("input_95_cast_fp16")]; tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; fp16 var_1985_to_fp16 = const()[name = string("op_1985_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_1985_to_fp16, x = input_95_cast_fp16)[name = string("normed_93_cast_fp16")]; tensor var_1998_split_sizes_0 = const()[name = string("op_1998_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1998_axis_0 = const()[name = string("op_1998_axis_0"), val = int32(-1)]; tensor var_1998_cast_fp16_0, tensor var_1998_cast_fp16_1 = split(axis = var_1998_axis_0, split_sizes = var_1998_split_sizes_0, x = normed_93_cast_fp16)[name = string("op_1998_cast_fp16")]; tensor const_51_to_fp16 = const()[name = string("const_51_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(344664704)))]; tensor var_2001_cast_fp16 = mul(x = var_1998_cast_fp16_0, y = const_51_to_fp16)[name = string("op_2001_cast_fp16")]; tensor hidden_states_59_cast_fp16 = add(x = hidden_states_55_cast_fp16, y = var_2001_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; tensor layers_3_layer_scalar_to_fp16 = const()[name = string("layers_3_layer_scalar_to_fp16"), val = tensor([0x1.a4p-1])]; tensor x_95_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = layers_3_layer_scalar_to_fp16)[name = string("x_95_cast_fp16")]; int32 var_2009 = const()[name = string("op_2009"), val = int32(-1)]; fp16 const_52_promoted_to_fp16 = const()[name = string("const_52_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2015_cast_fp16 = mul(x = x_95_cast_fp16, y = const_52_promoted_to_fp16)[name = string("op_2015_cast_fp16")]; bool input_97_interleave_0 = const()[name = string("input_97_interleave_0"), val = bool(false)]; tensor input_97_cast_fp16 = concat(axis = var_2009, interleave = input_97_interleave_0, values = (x_95_cast_fp16, var_2015_cast_fp16))[name = string("input_97_cast_fp16")]; tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; fp16 var_2007_to_fp16 = const()[name = string("op_2007_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_2007_to_fp16, x = input_97_cast_fp16)[name = string("normed_97_cast_fp16")]; tensor var_2020_split_sizes_0 = const()[name = string("op_2020_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2020_axis_0 = const()[name = string("op_2020_axis_0"), val = int32(-1)]; tensor var_2020_cast_fp16_0, tensor var_2020_cast_fp16_1 = split(axis = var_2020_axis_0, split_sizes = var_2020_split_sizes_0, x = normed_97_cast_fp16)[name = string("op_2020_cast_fp16")]; tensor const_53_to_fp16 = const()[name = string("const_53_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(344667840)))]; tensor var_2023_cast_fp16 = mul(x = var_2020_cast_fp16_0, y = const_53_to_fp16)[name = string("op_2023_cast_fp16")]; tensor var_2031 = const()[name = string("op_2031"), val = tensor([0, 2, 1])]; tensor var_2034_axes_0 = const()[name = string("op_2034_axes_0"), val = tensor([2])]; tensor var_2032_cast_fp16 = transpose(perm = var_2031, x = var_2023_cast_fp16)[name = string("transpose_50")]; tensor var_2034_cast_fp16 = expand_dims(axes = var_2034_axes_0, x = var_2032_cast_fp16)[name = string("op_2034_cast_fp16")]; string var_2050_pad_type_0 = const()[name = string("op_2050_pad_type_0"), val = string("valid")]; tensor var_2050_strides_0 = const()[name = string("op_2050_strides_0"), val = tensor([1, 1])]; tensor var_2050_pad_0 = const()[name = string("op_2050_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2050_dilations_0 = const()[name = string("op_2050_dilations_0"), val = tensor([1, 1])]; int32 var_2050_groups_0 = const()[name = string("op_2050_groups_0"), val = int32(1)]; tensor var_2050 = conv(dilations = var_2050_dilations_0, groups = var_2050_groups_0, pad = var_2050_pad_0, pad_type = var_2050_pad_type_0, strides = var_2050_strides_0, weight = layers_4_self_attn_q_proj_weight_palettized, x = var_2034_cast_fp16)[name = string("op_2050")]; tensor var_2055 = const()[name = string("op_2055"), val = tensor([1, 8, 256, 1])]; tensor var_2056 = reshape(shape = var_2055, x = var_2050)[name = string("op_2056")]; tensor var_2061 = const()[name = string("op_2061"), val = tensor([0, 1, 3, 2])]; tensor var_2071 = const()[name = string("op_2071"), val = tensor([1, 8, 256])]; tensor var_2062 = transpose(perm = var_2061, x = var_2056)[name = string("transpose_49")]; tensor x_99 = reshape(shape = var_2071, x = var_2062)[name = string("x_99")]; int32 var_2077 = const()[name = string("op_2077"), val = int32(-1)]; fp16 const_54_promoted_to_fp16 = const()[name = string("const_54_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2083_cast_fp16 = mul(x = x_99, y = const_54_promoted_to_fp16)[name = string("op_2083_cast_fp16")]; bool input_101_interleave_0 = const()[name = string("input_101_interleave_0"), val = bool(false)]; tensor input_101_cast_fp16 = concat(axis = var_2077, interleave = input_101_interleave_0, values = (x_99, var_2083_cast_fp16))[name = string("input_101_cast_fp16")]; tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; fp16 var_2075_to_fp16 = const()[name = string("op_2075_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_2075_to_fp16, x = input_101_cast_fp16)[name = string("normed_101_cast_fp16")]; tensor var_2088_split_sizes_0 = const()[name = string("op_2088_split_sizes_0"), val = tensor([256, 256])]; int32 var_2088_axis_0 = const()[name = string("op_2088_axis_0"), val = int32(-1)]; tensor var_2088_cast_fp16_0, tensor var_2088_cast_fp16_1 = split(axis = var_2088_axis_0, split_sizes = var_2088_split_sizes_0, x = normed_101_cast_fp16)[name = string("op_2088_cast_fp16")]; tensor var_2091_cast_fp16 = mul(x = var_2088_cast_fp16_0, y = const_16_to_fp16)[name = string("op_2091_cast_fp16")]; tensor var_2097 = const()[name = string("op_2097"), val = tensor([1, 8, 1, 256])]; tensor q_27 = reshape(shape = var_2097, x = var_2091_cast_fp16)[name = string("q_27")]; tensor var_2099_cast_fp16 = mul(x = q_27, y = cos_s)[name = string("op_2099_cast_fp16")]; tensor var_2100_split_sizes_0 = const()[name = string("op_2100_split_sizes_0"), val = tensor([128, 128])]; int32 var_2100_axis_0 = const()[name = string("op_2100_axis_0"), val = int32(-1)]; tensor var_2100_0, tensor var_2100_1 = split(axis = var_2100_axis_0, split_sizes = var_2100_split_sizes_0, x = q_27)[name = string("op_2100")]; fp16 const_56_promoted = const()[name = string("const_56_promoted"), val = fp16(-0x1p+0)]; tensor var_2102 = mul(x = var_2100_1, y = const_56_promoted)[name = string("op_2102")]; int32 var_2104 = const()[name = string("op_2104"), val = int32(-1)]; bool var_2105_interleave_0 = const()[name = string("op_2105_interleave_0"), val = bool(false)]; tensor var_2105 = concat(axis = var_2104, interleave = var_2105_interleave_0, values = (var_2102, var_2100_0))[name = string("op_2105")]; tensor var_2106_cast_fp16 = mul(x = var_2105, y = sin_s)[name = string("op_2106_cast_fp16")]; tensor q_29_cast_fp16 = add(x = var_2099_cast_fp16, y = var_2106_cast_fp16)[name = string("q_29_cast_fp16")]; bool var_2130_transpose_x_0 = const()[name = string("op_2130_transpose_x_0"), val = bool(false)]; bool var_2130_transpose_y_0 = const()[name = string("op_2130_transpose_y_0"), val = bool(false)]; tensor var_2130_cast_fp16 = matmul(transpose_x = var_2130_transpose_x_0, transpose_y = var_2130_transpose_y_0, x = q_29_cast_fp16, y = transpose_45_cast_fp16)[name = string("op_2130_cast_fp16")]; tensor var_2137_cast_fp16 = add(x = var_2130_cast_fp16, y = causal_mask)[name = string("op_2137_cast_fp16")]; int32 var_2138 = const()[name = string("op_2138"), val = int32(-1)]; tensor var_2140_cast_fp16 = softmax(axis = var_2138, x = var_2137_cast_fp16)[name = string("op_2140_cast_fp16")]; bool var_2156_transpose_x_0 = const()[name = string("op_2156_transpose_x_0"), val = bool(false)]; bool var_2156_transpose_y_0 = const()[name = string("op_2156_transpose_y_0"), val = bool(false)]; tensor var_2156_cast_fp16 = matmul(transpose_x = var_2156_transpose_x_0, transpose_y = var_2156_transpose_y_0, x = var_2140_cast_fp16, y = Ve_3_cast_fp16)[name = string("op_2156_cast_fp16")]; tensor var_2166 = const()[name = string("op_2166"), val = tensor([0, 2, 1, 3])]; tensor var_2173 = const()[name = string("op_2173"), val = tensor([1, 1, -1])]; tensor var_2167 = transpose(perm = var_2166, x = var_2156_cast_fp16)[name = string("transpose_48")]; tensor var_2174 = reshape(shape = var_2173, x = var_2167)[name = string("op_2174")]; tensor var_2178 = const()[name = string("op_2178"), val = tensor([0, 2, 1])]; tensor squeeze_4_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(344670976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(346243904))))[name = string("squeeze_4_palettized")]; string var_2194_pad_type_0 = const()[name = string("op_2194_pad_type_0"), val = string("valid")]; int32 var_2194_groups_0 = const()[name = string("op_2194_groups_0"), val = int32(1)]; tensor var_2194_strides_0 = const()[name = string("op_2194_strides_0"), val = tensor([1])]; tensor var_2194_pad_0 = const()[name = string("op_2194_pad_0"), val = tensor([0, 0])]; tensor var_2194_dilations_0 = const()[name = string("op_2194_dilations_0"), val = tensor([1])]; tensor var_2179 = transpose(perm = var_2178, x = var_2174)[name = string("transpose_47")]; tensor var_2194 = conv(dilations = var_2194_dilations_0, groups = var_2194_groups_0, pad = var_2194_pad_0, pad_type = var_2194_pad_type_0, strides = var_2194_strides_0, weight = squeeze_4_palettized, x = var_2179)[name = string("op_2194")]; tensor var_2198 = const()[name = string("op_2198"), val = tensor([0, 2, 1])]; int32 var_2204 = const()[name = string("op_2204"), val = int32(-1)]; fp16 const_57_promoted_to_fp16 = const()[name = string("const_57_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_103 = transpose(perm = var_2198, x = var_2194)[name = string("transpose_46")]; tensor var_2210_cast_fp16 = mul(x = x_103, y = const_57_promoted_to_fp16)[name = string("op_2210_cast_fp16")]; bool input_105_interleave_0 = const()[name = string("input_105_interleave_0"), val = bool(false)]; tensor input_105_cast_fp16 = concat(axis = var_2204, interleave = input_105_interleave_0, values = (x_103, var_2210_cast_fp16))[name = string("input_105_cast_fp16")]; tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; fp16 var_2202_to_fp16 = const()[name = string("op_2202_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_2202_to_fp16, x = input_105_cast_fp16)[name = string("normed_105_cast_fp16")]; tensor var_2215_split_sizes_0 = const()[name = string("op_2215_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2215_axis_0 = const()[name = string("op_2215_axis_0"), val = int32(-1)]; tensor var_2215_cast_fp16_0, tensor var_2215_cast_fp16_1 = split(axis = var_2215_axis_0, split_sizes = var_2215_split_sizes_0, x = normed_105_cast_fp16)[name = string("op_2215_cast_fp16")]; tensor const_58_to_fp16 = const()[name = string("const_58_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(346245504)))]; tensor var_2218_cast_fp16 = mul(x = var_2215_cast_fp16_0, y = const_58_to_fp16)[name = string("op_2218_cast_fp16")]; tensor x_107_cast_fp16 = add(x = x_95_cast_fp16, y = var_2218_cast_fp16)[name = string("x_107_cast_fp16")]; int32 var_2225 = const()[name = string("op_2225"), val = int32(-1)]; fp16 const_59_promoted_to_fp16 = const()[name = string("const_59_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2231_cast_fp16 = mul(x = x_107_cast_fp16, y = const_59_promoted_to_fp16)[name = string("op_2231_cast_fp16")]; bool input_107_interleave_0 = const()[name = string("input_107_interleave_0"), val = bool(false)]; tensor input_107_cast_fp16 = concat(axis = var_2225, interleave = input_107_interleave_0, values = (x_107_cast_fp16, var_2231_cast_fp16))[name = string("input_107_cast_fp16")]; tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; fp16 var_2223_to_fp16 = const()[name = string("op_2223_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_2223_to_fp16, x = input_107_cast_fp16)[name = string("normed_109_cast_fp16")]; tensor var_2236_split_sizes_0 = const()[name = string("op_2236_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2236_axis_0 = const()[name = string("op_2236_axis_0"), val = int32(-1)]; tensor var_2236_cast_fp16_0, tensor var_2236_cast_fp16_1 = split(axis = var_2236_axis_0, split_sizes = var_2236_split_sizes_0, x = normed_109_cast_fp16)[name = string("op_2236_cast_fp16")]; tensor const_60_to_fp16 = const()[name = string("const_60_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(346248640)))]; tensor var_2239_cast_fp16 = mul(x = var_2236_cast_fp16_0, y = const_60_to_fp16)[name = string("op_2239_cast_fp16")]; tensor var_2252 = const()[name = string("op_2252"), val = tensor([0, 2, 1])]; tensor input_109_axes_0 = const()[name = string("input_109_axes_0"), val = tensor([2])]; tensor var_2253 = transpose(perm = var_2252, x = var_2239_cast_fp16)[name = string("transpose_45")]; tensor input_109 = expand_dims(axes = input_109_axes_0, x = var_2253)[name = string("input_109")]; string var_2266_pad_type_0 = const()[name = string("op_2266_pad_type_0"), val = string("valid")]; tensor var_2266_strides_0 = const()[name = string("op_2266_strides_0"), val = tensor([1, 1])]; tensor var_2266_pad_0 = const()[name = string("op_2266_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2266_dilations_0 = const()[name = string("op_2266_dilations_0"), val = tensor([1, 1])]; int32 var_2266_groups_0 = const()[name = string("op_2266_groups_0"), val = int32(1)]; tensor var_2266 = conv(dilations = var_2266_dilations_0, groups = var_2266_groups_0, pad = var_2266_pad_0, pad_type = var_2266_pad_type_0, strides = var_2266_strides_0, weight = layers_4_mlp_gate_proj_weight_palettized, x = input_109)[name = string("op_2266")]; string var_2268_mode_0 = const()[name = string("op_2268_mode_0"), val = string("TANH_APPROXIMATION")]; tensor var_2268 = gelu(mode = var_2268_mode_0, x = var_2266)[name = string("op_2268")]; string var_2279_pad_type_0 = const()[name = string("op_2279_pad_type_0"), val = string("valid")]; tensor var_2279_strides_0 = const()[name = string("op_2279_strides_0"), val = tensor([1, 1])]; tensor var_2279_pad_0 = const()[name = string("op_2279_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2279_dilations_0 = const()[name = string("op_2279_dilations_0"), val = tensor([1, 1])]; int32 var_2279_groups_0 = const()[name = string("op_2279_groups_0"), val = int32(1)]; tensor var_2279 = conv(dilations = var_2279_dilations_0, groups = var_2279_groups_0, pad = var_2279_pad_0, pad_type = var_2279_pad_type_0, strides = var_2279_strides_0, weight = layers_4_mlp_up_proj_weight_palettized, x = input_109)[name = string("op_2279")]; tensor input_111 = mul(x = var_2268, y = var_2279)[name = string("input_111")]; string var_2291_pad_type_0 = const()[name = string("op_2291_pad_type_0"), val = string("valid")]; tensor var_2291_strides_0 = const()[name = string("op_2291_strides_0"), val = tensor([1, 1])]; tensor var_2291_pad_0 = const()[name = string("op_2291_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2291_dilations_0 = const()[name = string("op_2291_dilations_0"), val = tensor([1, 1])]; int32 var_2291_groups_0 = const()[name = string("op_2291_groups_0"), val = int32(1)]; tensor var_2291 = conv(dilations = var_2291_dilations_0, groups = var_2291_groups_0, pad = var_2291_pad_0, pad_type = var_2291_pad_type_0, strides = var_2291_strides_0, weight = layers_4_mlp_down_proj_weight_palettized, x = input_111)[name = string("op_2291")]; tensor var_2293_axes_0 = const()[name = string("op_2293_axes_0"), val = tensor([2])]; tensor var_2293 = squeeze(axes = var_2293_axes_0, x = var_2291)[name = string("op_2293")]; tensor var_2297 = const()[name = string("op_2297"), val = tensor([0, 2, 1])]; int32 var_2303 = const()[name = string("op_2303"), val = int32(-1)]; fp16 const_61_promoted_to_fp16 = const()[name = string("const_61_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_111 = transpose(perm = var_2297, x = var_2293)[name = string("transpose_44")]; tensor var_2309_cast_fp16 = mul(x = x_111, y = const_61_promoted_to_fp16)[name = string("op_2309_cast_fp16")]; bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; tensor input_113_cast_fp16 = concat(axis = var_2303, interleave = input_113_interleave_0, values = (x_111, var_2309_cast_fp16))[name = string("input_113_cast_fp16")]; tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; fp16 var_2301_to_fp16 = const()[name = string("op_2301_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_2301_to_fp16, x = input_113_cast_fp16)[name = string("normed_113_cast_fp16")]; tensor var_2314_split_sizes_0 = const()[name = string("op_2314_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2314_axis_0 = const()[name = string("op_2314_axis_0"), val = int32(-1)]; tensor var_2314_cast_fp16_0, tensor var_2314_cast_fp16_1 = split(axis = var_2314_axis_0, split_sizes = var_2314_split_sizes_0, x = normed_113_cast_fp16)[name = string("op_2314_cast_fp16")]; tensor const_62_to_fp16 = const()[name = string("const_62_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(346251776)))]; tensor var_2317_cast_fp16 = mul(x = var_2314_cast_fp16_0, y = const_62_to_fp16)[name = string("op_2317_cast_fp16")]; tensor hidden_states_69_cast_fp16 = add(x = x_107_cast_fp16, y = var_2317_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; tensor var_2328 = linear(bias = linear_0_bias_0, weight = layers_4_per_layer_input_gate_weight_palettized, x = hidden_states_69_cast_fp16)[name = string("linear_8")]; string gated_9_mode_0 = const()[name = string("gated_9_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_9 = gelu(mode = gated_9_mode_0, x = var_2328)[name = string("gated_9")]; tensor var_2345_begin_0 = const()[name = string("op_2345_begin_0"), val = tensor([0, 0, 7168])]; tensor var_2345_end_0 = const()[name = string("op_2345_end_0"), val = tensor([1, 1, 7424])]; tensor var_2345_end_mask_0 = const()[name = string("op_2345_end_mask_0"), val = tensor([true, true, false])]; tensor var_2345_cast_fp16 = slice_by_index(begin = var_2345_begin_0, end = var_2345_end_0, end_mask = var_2345_end_mask_0, x = per_layer_combined)[name = string("op_2345_cast_fp16")]; tensor input_117_cast_fp16 = mul(x = gated_9, y = var_2345_cast_fp16)[name = string("input_117_cast_fp16")]; tensor layers_4_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(346254912))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(346451584))))[name = string("layers_4_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_9_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_4_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_117_cast_fp16)[name = string("linear_9_cast_fp16")]; int32 var_2354 = const()[name = string("op_2354"), val = int32(-1)]; fp16 const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2360_cast_fp16 = mul(x = linear_9_cast_fp16, y = const_63_promoted_to_fp16)[name = string("op_2360_cast_fp16")]; bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; tensor input_119_cast_fp16 = concat(axis = var_2354, interleave = input_119_interleave_0, values = (linear_9_cast_fp16, var_2360_cast_fp16))[name = string("input_119_cast_fp16")]; tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; fp16 var_2352_to_fp16 = const()[name = string("op_2352_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_2352_to_fp16, x = input_119_cast_fp16)[name = string("normed_117_cast_fp16")]; tensor var_2365_split_sizes_0 = const()[name = string("op_2365_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2365_axis_0 = const()[name = string("op_2365_axis_0"), val = int32(-1)]; tensor var_2365_cast_fp16_0, tensor var_2365_cast_fp16_1 = split(axis = var_2365_axis_0, split_sizes = var_2365_split_sizes_0, x = normed_117_cast_fp16)[name = string("op_2365_cast_fp16")]; tensor const_64_to_fp16 = const()[name = string("const_64_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(346453184)))]; tensor var_2368_cast_fp16 = mul(x = var_2365_cast_fp16_0, y = const_64_to_fp16)[name = string("op_2368_cast_fp16")]; tensor hidden_states_73_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = var_2368_cast_fp16)[name = string("hidden_states_73_cast_fp16")]; tensor layers_4_layer_scalar_to_fp16 = const()[name = string("layers_4_layer_scalar_to_fp16"), val = tensor([0x1.a4p-1])]; tensor x_119_cast_fp16 = mul(x = hidden_states_73_cast_fp16, y = layers_4_layer_scalar_to_fp16)[name = string("x_119_cast_fp16")]; int32 var_2376 = const()[name = string("op_2376"), val = int32(-1)]; fp16 const_65_promoted_to_fp16 = const()[name = string("const_65_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2382_cast_fp16 = mul(x = x_119_cast_fp16, y = const_65_promoted_to_fp16)[name = string("op_2382_cast_fp16")]; bool input_121_interleave_0 = const()[name = string("input_121_interleave_0"), val = bool(false)]; tensor input_121_cast_fp16 = concat(axis = var_2376, interleave = input_121_interleave_0, values = (x_119_cast_fp16, var_2382_cast_fp16))[name = string("input_121_cast_fp16")]; tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; fp16 var_2374_to_fp16 = const()[name = string("op_2374_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_2374_to_fp16, x = input_121_cast_fp16)[name = string("normed_121_cast_fp16")]; tensor var_2387_split_sizes_0 = const()[name = string("op_2387_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2387_axis_0 = const()[name = string("op_2387_axis_0"), val = int32(-1)]; tensor var_2387_cast_fp16_0, tensor var_2387_cast_fp16_1 = split(axis = var_2387_axis_0, split_sizes = var_2387_split_sizes_0, x = normed_121_cast_fp16)[name = string("op_2387_cast_fp16")]; tensor const_66_to_fp16 = const()[name = string("const_66_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(346456320)))]; tensor var_2390_cast_fp16 = mul(x = var_2387_cast_fp16_0, y = const_66_to_fp16)[name = string("op_2390_cast_fp16")]; tensor var_2398 = const()[name = string("op_2398"), val = tensor([0, 2, 1])]; tensor var_2401_axes_0 = const()[name = string("op_2401_axes_0"), val = tensor([2])]; tensor var_2399_cast_fp16 = transpose(perm = var_2398, x = var_2390_cast_fp16)[name = string("transpose_43")]; tensor var_2401_cast_fp16 = expand_dims(axes = var_2401_axes_0, x = var_2399_cast_fp16)[name = string("op_2401_cast_fp16")]; string var_2417_pad_type_0 = const()[name = string("op_2417_pad_type_0"), val = string("valid")]; tensor var_2417_strides_0 = const()[name = string("op_2417_strides_0"), val = tensor([1, 1])]; tensor var_2417_pad_0 = const()[name = string("op_2417_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2417_dilations_0 = const()[name = string("op_2417_dilations_0"), val = tensor([1, 1])]; int32 var_2417_groups_0 = const()[name = string("op_2417_groups_0"), val = int32(1)]; tensor var_2417 = conv(dilations = var_2417_dilations_0, groups = var_2417_groups_0, pad = var_2417_pad_0, pad_type = var_2417_pad_type_0, strides = var_2417_strides_0, weight = layers_5_self_attn_q_proj_weight_palettized, x = var_2401_cast_fp16)[name = string("op_2417")]; tensor var_2422 = const()[name = string("op_2422"), val = tensor([1, 8, 512, 1])]; tensor var_2423 = reshape(shape = var_2422, x = var_2417)[name = string("op_2423")]; tensor var_2428 = const()[name = string("op_2428"), val = tensor([0, 1, 3, 2])]; tensor var_2438 = const()[name = string("op_2438"), val = tensor([1, 8, 512])]; tensor var_2429 = transpose(perm = var_2428, x = var_2423)[name = string("transpose_42")]; tensor x_123 = reshape(shape = var_2438, x = var_2429)[name = string("x_123")]; int32 var_2444 = const()[name = string("op_2444"), val = int32(-1)]; fp16 const_67_promoted_to_fp16 = const()[name = string("const_67_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2450_cast_fp16 = mul(x = x_123, y = const_67_promoted_to_fp16)[name = string("op_2450_cast_fp16")]; bool input_125_interleave_0 = const()[name = string("input_125_interleave_0"), val = bool(false)]; tensor input_125_cast_fp16 = concat(axis = var_2444, interleave = input_125_interleave_0, values = (x_123, var_2450_cast_fp16))[name = string("input_125_cast_fp16")]; tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; fp16 var_2442_to_fp16 = const()[name = string("op_2442_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_2442_to_fp16, x = input_125_cast_fp16)[name = string("normed_125_cast_fp16")]; tensor var_2455_split_sizes_0 = const()[name = string("op_2455_split_sizes_0"), val = tensor([512, 512])]; int32 var_2455_axis_0 = const()[name = string("op_2455_axis_0"), val = int32(-1)]; tensor var_2455_cast_fp16_0, tensor var_2455_cast_fp16_1 = split(axis = var_2455_axis_0, split_sizes = var_2455_split_sizes_0, x = normed_125_cast_fp16)[name = string("op_2455_cast_fp16")]; tensor var_2458_cast_fp16 = mul(x = var_2455_cast_fp16_0, y = const_3_to_fp16)[name = string("op_2458_cast_fp16")]; tensor var_2464 = const()[name = string("op_2464"), val = tensor([1, 8, 1, 512])]; tensor q_33 = reshape(shape = var_2464, x = var_2458_cast_fp16)[name = string("q_33")]; tensor var_2466_cast_fp16 = mul(x = q_33, y = cos_f)[name = string("op_2466_cast_fp16")]; tensor var_2467_split_sizes_0 = const()[name = string("op_2467_split_sizes_0"), val = tensor([256, 256])]; int32 var_2467_axis_0 = const()[name = string("op_2467_axis_0"), val = int32(-1)]; tensor var_2467_0, tensor var_2467_1 = split(axis = var_2467_axis_0, split_sizes = var_2467_split_sizes_0, x = q_33)[name = string("op_2467")]; fp16 const_69_promoted = const()[name = string("const_69_promoted"), val = fp16(-0x1p+0)]; tensor var_2469 = mul(x = var_2467_1, y = const_69_promoted)[name = string("op_2469")]; int32 var_2471 = const()[name = string("op_2471"), val = int32(-1)]; bool var_2472_interleave_0 = const()[name = string("op_2472_interleave_0"), val = bool(false)]; tensor var_2472 = concat(axis = var_2471, interleave = var_2472_interleave_0, values = (var_2469, var_2467_0))[name = string("op_2472")]; tensor var_2473_cast_fp16 = mul(x = var_2472, y = sin_f)[name = string("op_2473_cast_fp16")]; tensor q_35_cast_fp16 = add(x = var_2466_cast_fp16, y = var_2473_cast_fp16)[name = string("q_35_cast_fp16")]; bool var_2497_transpose_x_0 = const()[name = string("op_2497_transpose_x_0"), val = bool(false)]; bool var_2497_transpose_y_0 = const()[name = string("op_2497_transpose_y_0"), val = bool(false)]; tensor var_2497_cast_fp16 = matmul(transpose_x = var_2497_transpose_x_0, transpose_y = var_2497_transpose_y_0, x = q_35_cast_fp16, y = transpose_44_cast_fp16)[name = string("op_2497_cast_fp16")]; tensor var_2504_cast_fp16 = add(x = var_2497_cast_fp16, y = causal_mask)[name = string("op_2504_cast_fp16")]; int32 var_2505 = const()[name = string("op_2505"), val = int32(-1)]; tensor var_2507_cast_fp16 = softmax(axis = var_2505, x = var_2504_cast_fp16)[name = string("op_2507_cast_fp16")]; bool var_2523_transpose_x_0 = const()[name = string("op_2523_transpose_x_0"), val = bool(false)]; bool var_2523_transpose_y_0 = const()[name = string("op_2523_transpose_y_0"), val = bool(false)]; tensor var_2523_cast_fp16 = matmul(transpose_x = var_2523_transpose_x_0, transpose_y = var_2523_transpose_y_0, x = var_2507_cast_fp16, y = Ve_1_cast_fp16)[name = string("op_2523_cast_fp16")]; tensor var_2533 = const()[name = string("op_2533"), val = tensor([0, 2, 1, 3])]; tensor var_2540 = const()[name = string("op_2540"), val = tensor([1, 1, -1])]; tensor var_2534 = transpose(perm = var_2533, x = var_2523_cast_fp16)[name = string("transpose_41")]; tensor var_2541 = reshape(shape = var_2540, x = var_2534)[name = string("op_2541")]; tensor var_2545 = const()[name = string("op_2545"), val = tensor([0, 2, 1])]; tensor squeeze_5_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(346459456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(349605248))))[name = string("squeeze_5_palettized")]; string var_2561_pad_type_0 = const()[name = string("op_2561_pad_type_0"), val = string("valid")]; int32 var_2561_groups_0 = const()[name = string("op_2561_groups_0"), val = int32(1)]; tensor var_2561_strides_0 = const()[name = string("op_2561_strides_0"), val = tensor([1])]; tensor var_2561_pad_0 = const()[name = string("op_2561_pad_0"), val = tensor([0, 0])]; tensor var_2561_dilations_0 = const()[name = string("op_2561_dilations_0"), val = tensor([1])]; tensor var_2546 = transpose(perm = var_2545, x = var_2541)[name = string("transpose_40")]; tensor var_2561 = conv(dilations = var_2561_dilations_0, groups = var_2561_groups_0, pad = var_2561_pad_0, pad_type = var_2561_pad_type_0, strides = var_2561_strides_0, weight = squeeze_5_palettized, x = var_2546)[name = string("op_2561")]; tensor var_2565 = const()[name = string("op_2565"), val = tensor([0, 2, 1])]; int32 var_2571 = const()[name = string("op_2571"), val = int32(-1)]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_127 = transpose(perm = var_2565, x = var_2561)[name = string("transpose_39")]; tensor var_2577_cast_fp16 = mul(x = x_127, y = const_70_promoted_to_fp16)[name = string("op_2577_cast_fp16")]; bool input_129_interleave_0 = const()[name = string("input_129_interleave_0"), val = bool(false)]; tensor input_129_cast_fp16 = concat(axis = var_2571, interleave = input_129_interleave_0, values = (x_127, var_2577_cast_fp16))[name = string("input_129_cast_fp16")]; tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; fp16 var_2569_to_fp16 = const()[name = string("op_2569_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_2569_to_fp16, x = input_129_cast_fp16)[name = string("normed_129_cast_fp16")]; tensor var_2582_split_sizes_0 = const()[name = string("op_2582_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2582_axis_0 = const()[name = string("op_2582_axis_0"), val = int32(-1)]; tensor var_2582_cast_fp16_0, tensor var_2582_cast_fp16_1 = split(axis = var_2582_axis_0, split_sizes = var_2582_split_sizes_0, x = normed_129_cast_fp16)[name = string("op_2582_cast_fp16")]; tensor const_71_to_fp16 = const()[name = string("const_71_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(349606848)))]; tensor var_2585_cast_fp16 = mul(x = var_2582_cast_fp16_0, y = const_71_to_fp16)[name = string("op_2585_cast_fp16")]; tensor x_131_cast_fp16 = add(x = x_119_cast_fp16, y = var_2585_cast_fp16)[name = string("x_131_cast_fp16")]; int32 var_2592 = const()[name = string("op_2592"), val = int32(-1)]; fp16 const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2598_cast_fp16 = mul(x = x_131_cast_fp16, y = const_72_promoted_to_fp16)[name = string("op_2598_cast_fp16")]; bool input_131_interleave_0 = const()[name = string("input_131_interleave_0"), val = bool(false)]; tensor input_131_cast_fp16 = concat(axis = var_2592, interleave = input_131_interleave_0, values = (x_131_cast_fp16, var_2598_cast_fp16))[name = string("input_131_cast_fp16")]; tensor normed_133_axes_0 = const()[name = string("normed_133_axes_0"), val = tensor([-1])]; fp16 var_2590_to_fp16 = const()[name = string("op_2590_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_133_cast_fp16 = layer_norm(axes = normed_133_axes_0, epsilon = var_2590_to_fp16, x = input_131_cast_fp16)[name = string("normed_133_cast_fp16")]; tensor var_2603_split_sizes_0 = const()[name = string("op_2603_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2603_axis_0 = const()[name = string("op_2603_axis_0"), val = int32(-1)]; tensor var_2603_cast_fp16_0, tensor var_2603_cast_fp16_1 = split(axis = var_2603_axis_0, split_sizes = var_2603_split_sizes_0, x = normed_133_cast_fp16)[name = string("op_2603_cast_fp16")]; tensor const_73_to_fp16 = const()[name = string("const_73_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(349609984)))]; tensor var_2606_cast_fp16 = mul(x = var_2603_cast_fp16_0, y = const_73_to_fp16)[name = string("op_2606_cast_fp16")]; tensor var_2619 = const()[name = string("op_2619"), val = tensor([0, 2, 1])]; tensor input_133_axes_0 = const()[name = string("input_133_axes_0"), val = tensor([2])]; tensor var_2620 = transpose(perm = var_2619, x = var_2606_cast_fp16)[name = string("transpose_38")]; tensor input_133 = expand_dims(axes = input_133_axes_0, x = var_2620)[name = string("input_133")]; string var_2633_pad_type_0 = const()[name = string("op_2633_pad_type_0"), val = string("valid")]; tensor var_2633_strides_0 = const()[name = string("op_2633_strides_0"), val = tensor([1, 1])]; tensor var_2633_pad_0 = const()[name = string("op_2633_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2633_dilations_0 = const()[name = string("op_2633_dilations_0"), val = tensor([1, 1])]; int32 var_2633_groups_0 = const()[name = string("op_2633_groups_0"), val = int32(1)]; tensor var_2633 = conv(dilations = var_2633_dilations_0, groups = var_2633_groups_0, pad = var_2633_pad_0, pad_type = var_2633_pad_type_0, strides = var_2633_strides_0, weight = layers_5_mlp_gate_proj_weight_palettized, x = input_133)[name = string("op_2633")]; string var_2635_mode_0 = const()[name = string("op_2635_mode_0"), val = string("TANH_APPROXIMATION")]; tensor var_2635 = gelu(mode = var_2635_mode_0, x = var_2633)[name = string("op_2635")]; string var_2646_pad_type_0 = const()[name = string("op_2646_pad_type_0"), val = string("valid")]; tensor var_2646_strides_0 = const()[name = string("op_2646_strides_0"), val = tensor([1, 1])]; tensor var_2646_pad_0 = const()[name = string("op_2646_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2646_dilations_0 = const()[name = string("op_2646_dilations_0"), val = tensor([1, 1])]; int32 var_2646_groups_0 = const()[name = string("op_2646_groups_0"), val = int32(1)]; tensor var_2646 = conv(dilations = var_2646_dilations_0, groups = var_2646_groups_0, pad = var_2646_pad_0, pad_type = var_2646_pad_type_0, strides = var_2646_strides_0, weight = layers_5_mlp_up_proj_weight_palettized, x = input_133)[name = string("op_2646")]; tensor input_135 = mul(x = var_2635, y = var_2646)[name = string("input_135")]; string var_2658_pad_type_0 = const()[name = string("op_2658_pad_type_0"), val = string("valid")]; tensor var_2658_strides_0 = const()[name = string("op_2658_strides_0"), val = tensor([1, 1])]; tensor var_2658_pad_0 = const()[name = string("op_2658_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2658_dilations_0 = const()[name = string("op_2658_dilations_0"), val = tensor([1, 1])]; int32 var_2658_groups_0 = const()[name = string("op_2658_groups_0"), val = int32(1)]; tensor var_2658 = conv(dilations = var_2658_dilations_0, groups = var_2658_groups_0, pad = var_2658_pad_0, pad_type = var_2658_pad_type_0, strides = var_2658_strides_0, weight = layers_5_mlp_down_proj_weight_palettized, x = input_135)[name = string("op_2658")]; tensor var_2660_axes_0 = const()[name = string("op_2660_axes_0"), val = tensor([2])]; tensor var_2660 = squeeze(axes = var_2660_axes_0, x = var_2658)[name = string("op_2660")]; tensor var_2664 = const()[name = string("op_2664"), val = tensor([0, 2, 1])]; int32 var_2670 = const()[name = string("op_2670"), val = int32(-1)]; fp16 const_74_promoted_to_fp16 = const()[name = string("const_74_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_135 = transpose(perm = var_2664, x = var_2660)[name = string("transpose_37")]; tensor var_2676_cast_fp16 = mul(x = x_135, y = const_74_promoted_to_fp16)[name = string("op_2676_cast_fp16")]; bool input_137_interleave_0 = const()[name = string("input_137_interleave_0"), val = bool(false)]; tensor input_137_cast_fp16 = concat(axis = var_2670, interleave = input_137_interleave_0, values = (x_135, var_2676_cast_fp16))[name = string("input_137_cast_fp16")]; tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; fp16 var_2668_to_fp16 = const()[name = string("op_2668_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_2668_to_fp16, x = input_137_cast_fp16)[name = string("normed_137_cast_fp16")]; tensor var_2681_split_sizes_0 = const()[name = string("op_2681_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2681_axis_0 = const()[name = string("op_2681_axis_0"), val = int32(-1)]; tensor var_2681_cast_fp16_0, tensor var_2681_cast_fp16_1 = split(axis = var_2681_axis_0, split_sizes = var_2681_split_sizes_0, x = normed_137_cast_fp16)[name = string("op_2681_cast_fp16")]; tensor const_75_to_fp16 = const()[name = string("const_75_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(349613120)))]; tensor var_2684_cast_fp16 = mul(x = var_2681_cast_fp16_0, y = const_75_to_fp16)[name = string("op_2684_cast_fp16")]; tensor hidden_states_83_cast_fp16 = add(x = x_131_cast_fp16, y = var_2684_cast_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor var_2695 = linear(bias = linear_0_bias_0, weight = layers_5_per_layer_input_gate_weight_palettized, x = hidden_states_83_cast_fp16)[name = string("linear_10")]; string gated_11_mode_0 = const()[name = string("gated_11_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_11 = gelu(mode = gated_11_mode_0, x = var_2695)[name = string("gated_11")]; tensor var_2712_begin_0 = const()[name = string("op_2712_begin_0"), val = tensor([0, 0, 7424])]; tensor var_2712_end_0 = const()[name = string("op_2712_end_0"), val = tensor([1, 1, 7680])]; tensor var_2712_end_mask_0 = const()[name = string("op_2712_end_mask_0"), val = tensor([true, true, false])]; tensor var_2712_cast_fp16 = slice_by_index(begin = var_2712_begin_0, end = var_2712_end_0, end_mask = var_2712_end_mask_0, x = per_layer_combined)[name = string("op_2712_cast_fp16")]; tensor input_141_cast_fp16 = mul(x = gated_11, y = var_2712_cast_fp16)[name = string("input_141_cast_fp16")]; tensor layers_5_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(349616256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(349812928))))[name = string("layers_5_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_11_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_5_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_141_cast_fp16)[name = string("linear_11_cast_fp16")]; int32 var_2721 = const()[name = string("op_2721"), val = int32(-1)]; fp16 const_76_promoted_to_fp16 = const()[name = string("const_76_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2727_cast_fp16 = mul(x = linear_11_cast_fp16, y = const_76_promoted_to_fp16)[name = string("op_2727_cast_fp16")]; bool input_143_interleave_0 = const()[name = string("input_143_interleave_0"), val = bool(false)]; tensor input_143_cast_fp16 = concat(axis = var_2721, interleave = input_143_interleave_0, values = (linear_11_cast_fp16, var_2727_cast_fp16))[name = string("input_143_cast_fp16")]; tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; fp16 var_2719_to_fp16 = const()[name = string("op_2719_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_2719_to_fp16, x = input_143_cast_fp16)[name = string("normed_141_cast_fp16")]; tensor var_2732_split_sizes_0 = const()[name = string("op_2732_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2732_axis_0 = const()[name = string("op_2732_axis_0"), val = int32(-1)]; tensor var_2732_cast_fp16_0, tensor var_2732_cast_fp16_1 = split(axis = var_2732_axis_0, split_sizes = var_2732_split_sizes_0, x = normed_141_cast_fp16)[name = string("op_2732_cast_fp16")]; tensor const_77_to_fp16 = const()[name = string("const_77_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(349814528)))]; tensor var_2735_cast_fp16 = mul(x = var_2732_cast_fp16_0, y = const_77_to_fp16)[name = string("op_2735_cast_fp16")]; tensor hidden_states_87_cast_fp16 = add(x = hidden_states_83_cast_fp16, y = var_2735_cast_fp16)[name = string("hidden_states_87_cast_fp16")]; tensor layers_5_layer_scalar_to_fp16 = const()[name = string("layers_5_layer_scalar_to_fp16"), val = tensor([0x1.ap-1])]; tensor x_143_cast_fp16 = mul(x = hidden_states_87_cast_fp16, y = layers_5_layer_scalar_to_fp16)[name = string("x_143_cast_fp16")]; int32 var_2743 = const()[name = string("op_2743"), val = int32(-1)]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2749_cast_fp16 = mul(x = x_143_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_2749_cast_fp16")]; bool input_145_interleave_0 = const()[name = string("input_145_interleave_0"), val = bool(false)]; tensor input_145_cast_fp16 = concat(axis = var_2743, interleave = input_145_interleave_0, values = (x_143_cast_fp16, var_2749_cast_fp16))[name = string("input_145_cast_fp16")]; tensor normed_145_axes_0 = const()[name = string("normed_145_axes_0"), val = tensor([-1])]; fp16 var_2741_to_fp16 = const()[name = string("op_2741_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_145_cast_fp16 = layer_norm(axes = normed_145_axes_0, epsilon = var_2741_to_fp16, x = input_145_cast_fp16)[name = string("normed_145_cast_fp16")]; tensor var_2754_split_sizes_0 = const()[name = string("op_2754_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2754_axis_0 = const()[name = string("op_2754_axis_0"), val = int32(-1)]; tensor var_2754_cast_fp16_0, tensor var_2754_cast_fp16_1 = split(axis = var_2754_axis_0, split_sizes = var_2754_split_sizes_0, x = normed_145_cast_fp16)[name = string("op_2754_cast_fp16")]; tensor const_79_to_fp16 = const()[name = string("const_79_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(349817664)))]; tensor var_2757_cast_fp16 = mul(x = var_2754_cast_fp16_0, y = const_79_to_fp16)[name = string("op_2757_cast_fp16")]; tensor var_2765 = const()[name = string("op_2765"), val = tensor([0, 2, 1])]; tensor var_2768_axes_0 = const()[name = string("op_2768_axes_0"), val = tensor([2])]; tensor var_2766_cast_fp16 = transpose(perm = var_2765, x = var_2757_cast_fp16)[name = string("transpose_36")]; tensor var_2768_cast_fp16 = expand_dims(axes = var_2768_axes_0, x = var_2766_cast_fp16)[name = string("op_2768_cast_fp16")]; string var_2784_pad_type_0 = const()[name = string("op_2784_pad_type_0"), val = string("valid")]; tensor var_2784_strides_0 = const()[name = string("op_2784_strides_0"), val = tensor([1, 1])]; tensor var_2784_pad_0 = const()[name = string("op_2784_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2784_dilations_0 = const()[name = string("op_2784_dilations_0"), val = tensor([1, 1])]; int32 var_2784_groups_0 = const()[name = string("op_2784_groups_0"), val = int32(1)]; tensor var_2784 = conv(dilations = var_2784_dilations_0, groups = var_2784_groups_0, pad = var_2784_pad_0, pad_type = var_2784_pad_type_0, strides = var_2784_strides_0, weight = layers_6_self_attn_q_proj_weight_palettized, x = var_2768_cast_fp16)[name = string("op_2784")]; tensor var_2789 = const()[name = string("op_2789"), val = tensor([1, 8, 256, 1])]; tensor var_2790 = reshape(shape = var_2789, x = var_2784)[name = string("op_2790")]; tensor var_2795 = const()[name = string("op_2795"), val = tensor([0, 1, 3, 2])]; tensor var_2805 = const()[name = string("op_2805"), val = tensor([1, 8, 256])]; tensor var_2796 = transpose(perm = var_2795, x = var_2790)[name = string("transpose_35")]; tensor x_147 = reshape(shape = var_2805, x = var_2796)[name = string("x_147")]; int32 var_2811 = const()[name = string("op_2811"), val = int32(-1)]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2817_cast_fp16 = mul(x = x_147, y = const_80_promoted_to_fp16)[name = string("op_2817_cast_fp16")]; bool input_149_interleave_0 = const()[name = string("input_149_interleave_0"), val = bool(false)]; tensor input_149_cast_fp16 = concat(axis = var_2811, interleave = input_149_interleave_0, values = (x_147, var_2817_cast_fp16))[name = string("input_149_cast_fp16")]; tensor normed_149_axes_0 = const()[name = string("normed_149_axes_0"), val = tensor([-1])]; fp16 var_2809_to_fp16 = const()[name = string("op_2809_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_149_cast_fp16 = layer_norm(axes = normed_149_axes_0, epsilon = var_2809_to_fp16, x = input_149_cast_fp16)[name = string("normed_149_cast_fp16")]; tensor var_2822_split_sizes_0 = const()[name = string("op_2822_split_sizes_0"), val = tensor([256, 256])]; int32 var_2822_axis_0 = const()[name = string("op_2822_axis_0"), val = int32(-1)]; tensor var_2822_cast_fp16_0, tensor var_2822_cast_fp16_1 = split(axis = var_2822_axis_0, split_sizes = var_2822_split_sizes_0, x = normed_149_cast_fp16)[name = string("op_2822_cast_fp16")]; tensor var_2825_cast_fp16 = mul(x = var_2822_cast_fp16_0, y = const_16_to_fp16)[name = string("op_2825_cast_fp16")]; tensor var_2831 = const()[name = string("op_2831"), val = tensor([1, 8, 1, 256])]; tensor q_39 = reshape(shape = var_2831, x = var_2825_cast_fp16)[name = string("q_39")]; tensor var_2833_cast_fp16 = mul(x = q_39, y = cos_s)[name = string("op_2833_cast_fp16")]; tensor var_2834_split_sizes_0 = const()[name = string("op_2834_split_sizes_0"), val = tensor([128, 128])]; int32 var_2834_axis_0 = const()[name = string("op_2834_axis_0"), val = int32(-1)]; tensor var_2834_0, tensor var_2834_1 = split(axis = var_2834_axis_0, split_sizes = var_2834_split_sizes_0, x = q_39)[name = string("op_2834")]; fp16 const_82_promoted = const()[name = string("const_82_promoted"), val = fp16(-0x1p+0)]; tensor var_2836 = mul(x = var_2834_1, y = const_82_promoted)[name = string("op_2836")]; int32 var_2838 = const()[name = string("op_2838"), val = int32(-1)]; bool var_2839_interleave_0 = const()[name = string("op_2839_interleave_0"), val = bool(false)]; tensor var_2839 = concat(axis = var_2838, interleave = var_2839_interleave_0, values = (var_2836, var_2834_0))[name = string("op_2839")]; tensor var_2840_cast_fp16 = mul(x = var_2839, y = sin_s)[name = string("op_2840_cast_fp16")]; tensor q_41_cast_fp16 = add(x = var_2833_cast_fp16, y = var_2840_cast_fp16)[name = string("q_41_cast_fp16")]; bool var_2864_transpose_x_0 = const()[name = string("op_2864_transpose_x_0"), val = bool(false)]; bool var_2864_transpose_y_0 = const()[name = string("op_2864_transpose_y_0"), val = bool(false)]; tensor var_2864_cast_fp16 = matmul(transpose_x = var_2864_transpose_x_0, transpose_y = var_2864_transpose_y_0, x = q_41_cast_fp16, y = transpose_45_cast_fp16)[name = string("op_2864_cast_fp16")]; tensor var_2871_cast_fp16 = add(x = var_2864_cast_fp16, y = causal_mask)[name = string("op_2871_cast_fp16")]; int32 var_2872 = const()[name = string("op_2872"), val = int32(-1)]; tensor var_2874_cast_fp16 = softmax(axis = var_2872, x = var_2871_cast_fp16)[name = string("op_2874_cast_fp16")]; bool var_2890_transpose_x_0 = const()[name = string("op_2890_transpose_x_0"), val = bool(false)]; bool var_2890_transpose_y_0 = const()[name = string("op_2890_transpose_y_0"), val = bool(false)]; tensor var_2890_cast_fp16 = matmul(transpose_x = var_2890_transpose_x_0, transpose_y = var_2890_transpose_y_0, x = var_2874_cast_fp16, y = Ve_3_cast_fp16)[name = string("op_2890_cast_fp16")]; tensor var_2900 = const()[name = string("op_2900"), val = tensor([0, 2, 1, 3])]; tensor var_2907 = const()[name = string("op_2907"), val = tensor([1, 1, -1])]; tensor var_2901 = transpose(perm = var_2900, x = var_2890_cast_fp16)[name = string("transpose_34")]; tensor var_2908 = reshape(shape = var_2907, x = var_2901)[name = string("op_2908")]; tensor var_2912 = const()[name = string("op_2912"), val = tensor([0, 2, 1])]; tensor squeeze_6_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(349820800))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(351393728))))[name = string("squeeze_6_palettized")]; string var_2928_pad_type_0 = const()[name = string("op_2928_pad_type_0"), val = string("valid")]; int32 var_2928_groups_0 = const()[name = string("op_2928_groups_0"), val = int32(1)]; tensor var_2928_strides_0 = const()[name = string("op_2928_strides_0"), val = tensor([1])]; tensor var_2928_pad_0 = const()[name = string("op_2928_pad_0"), val = tensor([0, 0])]; tensor var_2928_dilations_0 = const()[name = string("op_2928_dilations_0"), val = tensor([1])]; tensor var_2913 = transpose(perm = var_2912, x = var_2908)[name = string("transpose_33")]; tensor var_2928 = conv(dilations = var_2928_dilations_0, groups = var_2928_groups_0, pad = var_2928_pad_0, pad_type = var_2928_pad_type_0, strides = var_2928_strides_0, weight = squeeze_6_palettized, x = var_2913)[name = string("op_2928")]; tensor var_2932 = const()[name = string("op_2932"), val = tensor([0, 2, 1])]; int32 var_2938 = const()[name = string("op_2938"), val = int32(-1)]; fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_151 = transpose(perm = var_2932, x = var_2928)[name = string("transpose_32")]; tensor var_2944_cast_fp16 = mul(x = x_151, y = const_83_promoted_to_fp16)[name = string("op_2944_cast_fp16")]; bool input_153_interleave_0 = const()[name = string("input_153_interleave_0"), val = bool(false)]; tensor input_153_cast_fp16 = concat(axis = var_2938, interleave = input_153_interleave_0, values = (x_151, var_2944_cast_fp16))[name = string("input_153_cast_fp16")]; tensor normed_153_axes_0 = const()[name = string("normed_153_axes_0"), val = tensor([-1])]; fp16 var_2936_to_fp16 = const()[name = string("op_2936_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_153_cast_fp16 = layer_norm(axes = normed_153_axes_0, epsilon = var_2936_to_fp16, x = input_153_cast_fp16)[name = string("normed_153_cast_fp16")]; tensor var_2949_split_sizes_0 = const()[name = string("op_2949_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2949_axis_0 = const()[name = string("op_2949_axis_0"), val = int32(-1)]; tensor var_2949_cast_fp16_0, tensor var_2949_cast_fp16_1 = split(axis = var_2949_axis_0, split_sizes = var_2949_split_sizes_0, x = normed_153_cast_fp16)[name = string("op_2949_cast_fp16")]; tensor const_84_to_fp16 = const()[name = string("const_84_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(351395328)))]; tensor var_2952_cast_fp16 = mul(x = var_2949_cast_fp16_0, y = const_84_to_fp16)[name = string("op_2952_cast_fp16")]; tensor x_155_cast_fp16 = add(x = x_143_cast_fp16, y = var_2952_cast_fp16)[name = string("x_155_cast_fp16")]; int32 var_2959 = const()[name = string("op_2959"), val = int32(-1)]; fp16 const_85_promoted_to_fp16 = const()[name = string("const_85_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2965_cast_fp16 = mul(x = x_155_cast_fp16, y = const_85_promoted_to_fp16)[name = string("op_2965_cast_fp16")]; bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; tensor input_155_cast_fp16 = concat(axis = var_2959, interleave = input_155_interleave_0, values = (x_155_cast_fp16, var_2965_cast_fp16))[name = string("input_155_cast_fp16")]; tensor normed_157_axes_0 = const()[name = string("normed_157_axes_0"), val = tensor([-1])]; fp16 var_2957_to_fp16 = const()[name = string("op_2957_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_157_cast_fp16 = layer_norm(axes = normed_157_axes_0, epsilon = var_2957_to_fp16, x = input_155_cast_fp16)[name = string("normed_157_cast_fp16")]; tensor var_2970_split_sizes_0 = const()[name = string("op_2970_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2970_axis_0 = const()[name = string("op_2970_axis_0"), val = int32(-1)]; tensor var_2970_cast_fp16_0, tensor var_2970_cast_fp16_1 = split(axis = var_2970_axis_0, split_sizes = var_2970_split_sizes_0, x = normed_157_cast_fp16)[name = string("op_2970_cast_fp16")]; tensor const_86_to_fp16 = const()[name = string("const_86_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(351398464)))]; tensor var_2973_cast_fp16 = mul(x = var_2970_cast_fp16_0, y = const_86_to_fp16)[name = string("op_2973_cast_fp16")]; tensor var_2986 = const()[name = string("op_2986"), val = tensor([0, 2, 1])]; tensor input_157_axes_0 = const()[name = string("input_157_axes_0"), val = tensor([2])]; tensor var_2987 = transpose(perm = var_2986, x = var_2973_cast_fp16)[name = string("transpose_31")]; tensor input_157 = expand_dims(axes = input_157_axes_0, x = var_2987)[name = string("input_157")]; string var_3000_pad_type_0 = const()[name = string("op_3000_pad_type_0"), val = string("valid")]; tensor var_3000_strides_0 = const()[name = string("op_3000_strides_0"), val = tensor([1, 1])]; tensor var_3000_pad_0 = const()[name = string("op_3000_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3000_dilations_0 = const()[name = string("op_3000_dilations_0"), val = tensor([1, 1])]; int32 var_3000_groups_0 = const()[name = string("op_3000_groups_0"), val = int32(1)]; tensor var_3000 = conv(dilations = var_3000_dilations_0, groups = var_3000_groups_0, pad = var_3000_pad_0, pad_type = var_3000_pad_type_0, strides = var_3000_strides_0, weight = layers_6_mlp_gate_proj_weight_palettized, x = input_157)[name = string("op_3000")]; string var_3002_mode_0 = const()[name = string("op_3002_mode_0"), val = string("TANH_APPROXIMATION")]; tensor var_3002 = gelu(mode = var_3002_mode_0, x = var_3000)[name = string("op_3002")]; string var_3013_pad_type_0 = const()[name = string("op_3013_pad_type_0"), val = string("valid")]; tensor var_3013_strides_0 = const()[name = string("op_3013_strides_0"), val = tensor([1, 1])]; tensor var_3013_pad_0 = const()[name = string("op_3013_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3013_dilations_0 = const()[name = string("op_3013_dilations_0"), val = tensor([1, 1])]; int32 var_3013_groups_0 = const()[name = string("op_3013_groups_0"), val = int32(1)]; tensor var_3013 = conv(dilations = var_3013_dilations_0, groups = var_3013_groups_0, pad = var_3013_pad_0, pad_type = var_3013_pad_type_0, strides = var_3013_strides_0, weight = layers_6_mlp_up_proj_weight_palettized, x = input_157)[name = string("op_3013")]; tensor input_159 = mul(x = var_3002, y = var_3013)[name = string("input_159")]; string var_3025_pad_type_0 = const()[name = string("op_3025_pad_type_0"), val = string("valid")]; tensor var_3025_strides_0 = const()[name = string("op_3025_strides_0"), val = tensor([1, 1])]; tensor var_3025_pad_0 = const()[name = string("op_3025_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3025_dilations_0 = const()[name = string("op_3025_dilations_0"), val = tensor([1, 1])]; int32 var_3025_groups_0 = const()[name = string("op_3025_groups_0"), val = int32(1)]; tensor var_3025 = conv(dilations = var_3025_dilations_0, groups = var_3025_groups_0, pad = var_3025_pad_0, pad_type = var_3025_pad_type_0, strides = var_3025_strides_0, weight = layers_6_mlp_down_proj_weight_palettized, x = input_159)[name = string("op_3025")]; tensor var_3027_axes_0 = const()[name = string("op_3027_axes_0"), val = tensor([2])]; tensor var_3027 = squeeze(axes = var_3027_axes_0, x = var_3025)[name = string("op_3027")]; tensor var_3031 = const()[name = string("op_3031"), val = tensor([0, 2, 1])]; int32 var_3037 = const()[name = string("op_3037"), val = int32(-1)]; fp16 const_87_promoted_to_fp16 = const()[name = string("const_87_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_159 = transpose(perm = var_3031, x = var_3027)[name = string("transpose_30")]; tensor var_3043_cast_fp16 = mul(x = x_159, y = const_87_promoted_to_fp16)[name = string("op_3043_cast_fp16")]; bool input_161_interleave_0 = const()[name = string("input_161_interleave_0"), val = bool(false)]; tensor input_161_cast_fp16 = concat(axis = var_3037, interleave = input_161_interleave_0, values = (x_159, var_3043_cast_fp16))[name = string("input_161_cast_fp16")]; tensor normed_161_axes_0 = const()[name = string("normed_161_axes_0"), val = tensor([-1])]; fp16 var_3035_to_fp16 = const()[name = string("op_3035_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_161_cast_fp16 = layer_norm(axes = normed_161_axes_0, epsilon = var_3035_to_fp16, x = input_161_cast_fp16)[name = string("normed_161_cast_fp16")]; tensor var_3048_split_sizes_0 = const()[name = string("op_3048_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3048_axis_0 = const()[name = string("op_3048_axis_0"), val = int32(-1)]; tensor var_3048_cast_fp16_0, tensor var_3048_cast_fp16_1 = split(axis = var_3048_axis_0, split_sizes = var_3048_split_sizes_0, x = normed_161_cast_fp16)[name = string("op_3048_cast_fp16")]; tensor const_88_to_fp16 = const()[name = string("const_88_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(351401600)))]; tensor var_3051_cast_fp16 = mul(x = var_3048_cast_fp16_0, y = const_88_to_fp16)[name = string("op_3051_cast_fp16")]; tensor hidden_states_97_cast_fp16 = add(x = x_155_cast_fp16, y = var_3051_cast_fp16)[name = string("hidden_states_97_cast_fp16")]; tensor var_3062 = linear(bias = linear_0_bias_0, weight = layers_6_per_layer_input_gate_weight_palettized, x = hidden_states_97_cast_fp16)[name = string("linear_12")]; string gated_13_mode_0 = const()[name = string("gated_13_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_13 = gelu(mode = gated_13_mode_0, x = var_3062)[name = string("gated_13")]; tensor var_3079_begin_0 = const()[name = string("op_3079_begin_0"), val = tensor([0, 0, 7680])]; tensor var_3079_end_0 = const()[name = string("op_3079_end_0"), val = tensor([1, 1, 7936])]; tensor var_3079_end_mask_0 = const()[name = string("op_3079_end_mask_0"), val = tensor([true, true, false])]; tensor var_3079_cast_fp16 = slice_by_index(begin = var_3079_begin_0, end = var_3079_end_0, end_mask = var_3079_end_mask_0, x = per_layer_combined)[name = string("op_3079_cast_fp16")]; tensor input_165_cast_fp16 = mul(x = gated_13, y = var_3079_cast_fp16)[name = string("input_165_cast_fp16")]; tensor layers_6_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(351404736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(351601408))))[name = string("layers_6_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_13_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_6_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_165_cast_fp16)[name = string("linear_13_cast_fp16")]; int32 var_3088 = const()[name = string("op_3088"), val = int32(-1)]; fp16 const_89_promoted_to_fp16 = const()[name = string("const_89_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3094_cast_fp16 = mul(x = linear_13_cast_fp16, y = const_89_promoted_to_fp16)[name = string("op_3094_cast_fp16")]; bool input_167_interleave_0 = const()[name = string("input_167_interleave_0"), val = bool(false)]; tensor input_167_cast_fp16 = concat(axis = var_3088, interleave = input_167_interleave_0, values = (linear_13_cast_fp16, var_3094_cast_fp16))[name = string("input_167_cast_fp16")]; tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; fp16 var_3086_to_fp16 = const()[name = string("op_3086_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_3086_to_fp16, x = input_167_cast_fp16)[name = string("normed_165_cast_fp16")]; tensor var_3099_split_sizes_0 = const()[name = string("op_3099_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3099_axis_0 = const()[name = string("op_3099_axis_0"), val = int32(-1)]; tensor var_3099_cast_fp16_0, tensor var_3099_cast_fp16_1 = split(axis = var_3099_axis_0, split_sizes = var_3099_split_sizes_0, x = normed_165_cast_fp16)[name = string("op_3099_cast_fp16")]; tensor const_90_to_fp16 = const()[name = string("const_90_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(351603008)))]; tensor var_3102_cast_fp16 = mul(x = var_3099_cast_fp16_0, y = const_90_to_fp16)[name = string("op_3102_cast_fp16")]; tensor hidden_states_101_cast_fp16 = add(x = hidden_states_97_cast_fp16, y = var_3102_cast_fp16)[name = string("hidden_states_101_cast_fp16")]; tensor layers_6_layer_scalar_to_fp16 = const()[name = string("layers_6_layer_scalar_to_fp16"), val = tensor([0x1.bep-1])]; tensor x_167_cast_fp16 = mul(x = hidden_states_101_cast_fp16, y = layers_6_layer_scalar_to_fp16)[name = string("x_167_cast_fp16")]; int32 var_3110 = const()[name = string("op_3110"), val = int32(-1)]; fp16 const_91_promoted_to_fp16 = const()[name = string("const_91_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3116_cast_fp16 = mul(x = x_167_cast_fp16, y = const_91_promoted_to_fp16)[name = string("op_3116_cast_fp16")]; bool input_169_interleave_0 = const()[name = string("input_169_interleave_0"), val = bool(false)]; tensor input_169_cast_fp16 = concat(axis = var_3110, interleave = input_169_interleave_0, values = (x_167_cast_fp16, var_3116_cast_fp16))[name = string("input_169_cast_fp16")]; tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; fp16 var_3108_to_fp16 = const()[name = string("op_3108_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_3108_to_fp16, x = input_169_cast_fp16)[name = string("normed_169_cast_fp16")]; tensor var_3121_split_sizes_0 = const()[name = string("op_3121_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3121_axis_0 = const()[name = string("op_3121_axis_0"), val = int32(-1)]; tensor var_3121_cast_fp16_0, tensor var_3121_cast_fp16_1 = split(axis = var_3121_axis_0, split_sizes = var_3121_split_sizes_0, x = normed_169_cast_fp16)[name = string("op_3121_cast_fp16")]; tensor const_92_to_fp16 = const()[name = string("const_92_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(351606144)))]; tensor var_3124_cast_fp16 = mul(x = var_3121_cast_fp16_0, y = const_92_to_fp16)[name = string("op_3124_cast_fp16")]; tensor var_3132 = const()[name = string("op_3132"), val = tensor([0, 2, 1])]; tensor var_3135_axes_0 = const()[name = string("op_3135_axes_0"), val = tensor([2])]; tensor var_3133_cast_fp16 = transpose(perm = var_3132, x = var_3124_cast_fp16)[name = string("transpose_29")]; tensor var_3135_cast_fp16 = expand_dims(axes = var_3135_axes_0, x = var_3133_cast_fp16)[name = string("op_3135_cast_fp16")]; string var_3151_pad_type_0 = const()[name = string("op_3151_pad_type_0"), val = string("valid")]; tensor var_3151_strides_0 = const()[name = string("op_3151_strides_0"), val = tensor([1, 1])]; tensor var_3151_pad_0 = const()[name = string("op_3151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3151_dilations_0 = const()[name = string("op_3151_dilations_0"), val = tensor([1, 1])]; int32 var_3151_groups_0 = const()[name = string("op_3151_groups_0"), val = int32(1)]; tensor var_3151 = conv(dilations = var_3151_dilations_0, groups = var_3151_groups_0, pad = var_3151_pad_0, pad_type = var_3151_pad_type_0, strides = var_3151_strides_0, weight = layers_7_self_attn_q_proj_weight_palettized, x = var_3135_cast_fp16)[name = string("op_3151")]; tensor var_3156 = const()[name = string("op_3156"), val = tensor([1, 8, 256, 1])]; tensor var_3157 = reshape(shape = var_3156, x = var_3151)[name = string("op_3157")]; tensor var_3162 = const()[name = string("op_3162"), val = tensor([0, 1, 3, 2])]; tensor var_3172 = const()[name = string("op_3172"), val = tensor([1, 8, 256])]; tensor var_3163 = transpose(perm = var_3162, x = var_3157)[name = string("transpose_28")]; tensor x_171 = reshape(shape = var_3172, x = var_3163)[name = string("x_171")]; int32 var_3178 = const()[name = string("op_3178"), val = int32(-1)]; fp16 const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3184_cast_fp16 = mul(x = x_171, y = const_93_promoted_to_fp16)[name = string("op_3184_cast_fp16")]; bool input_173_interleave_0 = const()[name = string("input_173_interleave_0"), val = bool(false)]; tensor input_173_cast_fp16 = concat(axis = var_3178, interleave = input_173_interleave_0, values = (x_171, var_3184_cast_fp16))[name = string("input_173_cast_fp16")]; tensor normed_173_axes_0 = const()[name = string("normed_173_axes_0"), val = tensor([-1])]; fp16 var_3176_to_fp16 = const()[name = string("op_3176_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_173_cast_fp16 = layer_norm(axes = normed_173_axes_0, epsilon = var_3176_to_fp16, x = input_173_cast_fp16)[name = string("normed_173_cast_fp16")]; tensor var_3189_split_sizes_0 = const()[name = string("op_3189_split_sizes_0"), val = tensor([256, 256])]; int32 var_3189_axis_0 = const()[name = string("op_3189_axis_0"), val = int32(-1)]; tensor var_3189_cast_fp16_0, tensor var_3189_cast_fp16_1 = split(axis = var_3189_axis_0, split_sizes = var_3189_split_sizes_0, x = normed_173_cast_fp16)[name = string("op_3189_cast_fp16")]; tensor var_3192_cast_fp16 = mul(x = var_3189_cast_fp16_0, y = const_16_to_fp16)[name = string("op_3192_cast_fp16")]; tensor var_3198 = const()[name = string("op_3198"), val = tensor([1, 8, 1, 256])]; tensor q_45 = reshape(shape = var_3198, x = var_3192_cast_fp16)[name = string("q_45")]; tensor var_3200_cast_fp16 = mul(x = q_45, y = cos_s)[name = string("op_3200_cast_fp16")]; tensor var_3201_split_sizes_0 = const()[name = string("op_3201_split_sizes_0"), val = tensor([128, 128])]; int32 var_3201_axis_0 = const()[name = string("op_3201_axis_0"), val = int32(-1)]; tensor var_3201_0, tensor var_3201_1 = split(axis = var_3201_axis_0, split_sizes = var_3201_split_sizes_0, x = q_45)[name = string("op_3201")]; fp16 const_95_promoted = const()[name = string("const_95_promoted"), val = fp16(-0x1p+0)]; tensor var_3203 = mul(x = var_3201_1, y = const_95_promoted)[name = string("op_3203")]; int32 var_3205 = const()[name = string("op_3205"), val = int32(-1)]; bool var_3206_interleave_0 = const()[name = string("op_3206_interleave_0"), val = bool(false)]; tensor var_3206 = concat(axis = var_3205, interleave = var_3206_interleave_0, values = (var_3203, var_3201_0))[name = string("op_3206")]; tensor var_3207_cast_fp16 = mul(x = var_3206, y = sin_s)[name = string("op_3207_cast_fp16")]; tensor q_47_cast_fp16 = add(x = var_3200_cast_fp16, y = var_3207_cast_fp16)[name = string("q_47_cast_fp16")]; bool var_3231_transpose_x_0 = const()[name = string("op_3231_transpose_x_0"), val = bool(false)]; bool var_3231_transpose_y_0 = const()[name = string("op_3231_transpose_y_0"), val = bool(false)]; tensor var_3231_cast_fp16 = matmul(transpose_x = var_3231_transpose_x_0, transpose_y = var_3231_transpose_y_0, x = q_47_cast_fp16, y = transpose_45_cast_fp16)[name = string("op_3231_cast_fp16")]; tensor var_3238_cast_fp16 = add(x = var_3231_cast_fp16, y = causal_mask)[name = string("op_3238_cast_fp16")]; int32 var_3239 = const()[name = string("op_3239"), val = int32(-1)]; tensor var_3241_cast_fp16 = softmax(axis = var_3239, x = var_3238_cast_fp16)[name = string("op_3241_cast_fp16")]; bool var_3257_transpose_x_0 = const()[name = string("op_3257_transpose_x_0"), val = bool(false)]; bool var_3257_transpose_y_0 = const()[name = string("op_3257_transpose_y_0"), val = bool(false)]; tensor var_3257_cast_fp16 = matmul(transpose_x = var_3257_transpose_x_0, transpose_y = var_3257_transpose_y_0, x = var_3241_cast_fp16, y = Ve_3_cast_fp16)[name = string("op_3257_cast_fp16")]; tensor var_3267 = const()[name = string("op_3267"), val = tensor([0, 2, 1, 3])]; tensor var_3274 = const()[name = string("op_3274"), val = tensor([1, 1, -1])]; tensor var_3268 = transpose(perm = var_3267, x = var_3257_cast_fp16)[name = string("transpose_27")]; tensor var_3275 = reshape(shape = var_3274, x = var_3268)[name = string("op_3275")]; tensor var_3279 = const()[name = string("op_3279"), val = tensor([0, 2, 1])]; tensor squeeze_7_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(351609280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(353182208))))[name = string("squeeze_7_palettized")]; string var_3295_pad_type_0 = const()[name = string("op_3295_pad_type_0"), val = string("valid")]; int32 var_3295_groups_0 = const()[name = string("op_3295_groups_0"), val = int32(1)]; tensor var_3295_strides_0 = const()[name = string("op_3295_strides_0"), val = tensor([1])]; tensor var_3295_pad_0 = const()[name = string("op_3295_pad_0"), val = tensor([0, 0])]; tensor var_3295_dilations_0 = const()[name = string("op_3295_dilations_0"), val = tensor([1])]; tensor var_3280 = transpose(perm = var_3279, x = var_3275)[name = string("transpose_26")]; tensor var_3295 = conv(dilations = var_3295_dilations_0, groups = var_3295_groups_0, pad = var_3295_pad_0, pad_type = var_3295_pad_type_0, strides = var_3295_strides_0, weight = squeeze_7_palettized, x = var_3280)[name = string("op_3295")]; tensor var_3299 = const()[name = string("op_3299"), val = tensor([0, 2, 1])]; int32 var_3305 = const()[name = string("op_3305"), val = int32(-1)]; fp16 const_96_promoted_to_fp16 = const()[name = string("const_96_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_175 = transpose(perm = var_3299, x = var_3295)[name = string("transpose_25")]; tensor var_3311_cast_fp16 = mul(x = x_175, y = const_96_promoted_to_fp16)[name = string("op_3311_cast_fp16")]; bool input_177_interleave_0 = const()[name = string("input_177_interleave_0"), val = bool(false)]; tensor input_177_cast_fp16 = concat(axis = var_3305, interleave = input_177_interleave_0, values = (x_175, var_3311_cast_fp16))[name = string("input_177_cast_fp16")]; tensor normed_177_axes_0 = const()[name = string("normed_177_axes_0"), val = tensor([-1])]; fp16 var_3303_to_fp16 = const()[name = string("op_3303_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_177_cast_fp16 = layer_norm(axes = normed_177_axes_0, epsilon = var_3303_to_fp16, x = input_177_cast_fp16)[name = string("normed_177_cast_fp16")]; tensor var_3316_split_sizes_0 = const()[name = string("op_3316_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3316_axis_0 = const()[name = string("op_3316_axis_0"), val = int32(-1)]; tensor var_3316_cast_fp16_0, tensor var_3316_cast_fp16_1 = split(axis = var_3316_axis_0, split_sizes = var_3316_split_sizes_0, x = normed_177_cast_fp16)[name = string("op_3316_cast_fp16")]; tensor const_97_to_fp16 = const()[name = string("const_97_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(353183808)))]; tensor var_3319_cast_fp16 = mul(x = var_3316_cast_fp16_0, y = const_97_to_fp16)[name = string("op_3319_cast_fp16")]; tensor x_179_cast_fp16 = add(x = x_167_cast_fp16, y = var_3319_cast_fp16)[name = string("x_179_cast_fp16")]; int32 var_3326 = const()[name = string("op_3326"), val = int32(-1)]; fp16 const_98_promoted_to_fp16 = const()[name = string("const_98_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3332_cast_fp16 = mul(x = x_179_cast_fp16, y = const_98_promoted_to_fp16)[name = string("op_3332_cast_fp16")]; bool input_179_interleave_0 = const()[name = string("input_179_interleave_0"), val = bool(false)]; tensor input_179_cast_fp16 = concat(axis = var_3326, interleave = input_179_interleave_0, values = (x_179_cast_fp16, var_3332_cast_fp16))[name = string("input_179_cast_fp16")]; tensor normed_181_axes_0 = const()[name = string("normed_181_axes_0"), val = tensor([-1])]; fp16 var_3324_to_fp16 = const()[name = string("op_3324_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_181_cast_fp16 = layer_norm(axes = normed_181_axes_0, epsilon = var_3324_to_fp16, x = input_179_cast_fp16)[name = string("normed_181_cast_fp16")]; tensor var_3337_split_sizes_0 = const()[name = string("op_3337_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3337_axis_0 = const()[name = string("op_3337_axis_0"), val = int32(-1)]; tensor var_3337_cast_fp16_0, tensor var_3337_cast_fp16_1 = split(axis = var_3337_axis_0, split_sizes = var_3337_split_sizes_0, x = normed_181_cast_fp16)[name = string("op_3337_cast_fp16")]; tensor const_99_to_fp16 = const()[name = string("const_99_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(353186944)))]; tensor var_3340_cast_fp16 = mul(x = var_3337_cast_fp16_0, y = const_99_to_fp16)[name = string("op_3340_cast_fp16")]; tensor var_3353 = const()[name = string("op_3353"), val = tensor([0, 2, 1])]; tensor input_181_axes_0 = const()[name = string("input_181_axes_0"), val = tensor([2])]; tensor var_3354 = transpose(perm = var_3353, x = var_3340_cast_fp16)[name = string("transpose_24")]; tensor input_181 = expand_dims(axes = input_181_axes_0, x = var_3354)[name = string("input_181")]; string var_3367_pad_type_0 = const()[name = string("op_3367_pad_type_0"), val = string("valid")]; tensor var_3367_strides_0 = const()[name = string("op_3367_strides_0"), val = tensor([1, 1])]; tensor var_3367_pad_0 = const()[name = string("op_3367_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3367_dilations_0 = const()[name = string("op_3367_dilations_0"), val = tensor([1, 1])]; int32 var_3367_groups_0 = const()[name = string("op_3367_groups_0"), val = int32(1)]; tensor var_3367 = conv(dilations = var_3367_dilations_0, groups = var_3367_groups_0, pad = var_3367_pad_0, pad_type = var_3367_pad_type_0, strides = var_3367_strides_0, weight = layers_7_mlp_gate_proj_weight_palettized, x = input_181)[name = string("op_3367")]; string var_3369_mode_0 = const()[name = string("op_3369_mode_0"), val = string("TANH_APPROXIMATION")]; tensor var_3369 = gelu(mode = var_3369_mode_0, x = var_3367)[name = string("op_3369")]; string var_3380_pad_type_0 = const()[name = string("op_3380_pad_type_0"), val = string("valid")]; tensor var_3380_strides_0 = const()[name = string("op_3380_strides_0"), val = tensor([1, 1])]; tensor var_3380_pad_0 = const()[name = string("op_3380_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3380_dilations_0 = const()[name = string("op_3380_dilations_0"), val = tensor([1, 1])]; int32 var_3380_groups_0 = const()[name = string("op_3380_groups_0"), val = int32(1)]; tensor var_3380 = conv(dilations = var_3380_dilations_0, groups = var_3380_groups_0, pad = var_3380_pad_0, pad_type = var_3380_pad_type_0, strides = var_3380_strides_0, weight = layers_7_mlp_up_proj_weight_palettized, x = input_181)[name = string("op_3380")]; tensor input_183 = mul(x = var_3369, y = var_3380)[name = string("input_183")]; string var_3392_pad_type_0 = const()[name = string("op_3392_pad_type_0"), val = string("valid")]; tensor var_3392_strides_0 = const()[name = string("op_3392_strides_0"), val = tensor([1, 1])]; tensor var_3392_pad_0 = const()[name = string("op_3392_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3392_dilations_0 = const()[name = string("op_3392_dilations_0"), val = tensor([1, 1])]; int32 var_3392_groups_0 = const()[name = string("op_3392_groups_0"), val = int32(1)]; tensor var_3392 = conv(dilations = var_3392_dilations_0, groups = var_3392_groups_0, pad = var_3392_pad_0, pad_type = var_3392_pad_type_0, strides = var_3392_strides_0, weight = layers_7_mlp_down_proj_weight_palettized, x = input_183)[name = string("op_3392")]; tensor var_3394_axes_0 = const()[name = string("op_3394_axes_0"), val = tensor([2])]; tensor var_3394 = squeeze(axes = var_3394_axes_0, x = var_3392)[name = string("op_3394")]; tensor var_3398 = const()[name = string("op_3398"), val = tensor([0, 2, 1])]; int32 var_3404 = const()[name = string("op_3404"), val = int32(-1)]; fp16 const_100_promoted_to_fp16 = const()[name = string("const_100_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_183 = transpose(perm = var_3398, x = var_3394)[name = string("transpose_23")]; tensor var_3410_cast_fp16 = mul(x = x_183, y = const_100_promoted_to_fp16)[name = string("op_3410_cast_fp16")]; bool input_185_interleave_0 = const()[name = string("input_185_interleave_0"), val = bool(false)]; tensor input_185_cast_fp16 = concat(axis = var_3404, interleave = input_185_interleave_0, values = (x_183, var_3410_cast_fp16))[name = string("input_185_cast_fp16")]; tensor normed_185_axes_0 = const()[name = string("normed_185_axes_0"), val = tensor([-1])]; fp16 var_3402_to_fp16 = const()[name = string("op_3402_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_185_cast_fp16 = layer_norm(axes = normed_185_axes_0, epsilon = var_3402_to_fp16, x = input_185_cast_fp16)[name = string("normed_185_cast_fp16")]; tensor var_3415_split_sizes_0 = const()[name = string("op_3415_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3415_axis_0 = const()[name = string("op_3415_axis_0"), val = int32(-1)]; tensor var_3415_cast_fp16_0, tensor var_3415_cast_fp16_1 = split(axis = var_3415_axis_0, split_sizes = var_3415_split_sizes_0, x = normed_185_cast_fp16)[name = string("op_3415_cast_fp16")]; tensor const_101_to_fp16 = const()[name = string("const_101_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(353190080)))]; tensor var_3418_cast_fp16 = mul(x = var_3415_cast_fp16_0, y = const_101_to_fp16)[name = string("op_3418_cast_fp16")]; tensor hidden_states_111_cast_fp16 = add(x = x_179_cast_fp16, y = var_3418_cast_fp16)[name = string("hidden_states_111_cast_fp16")]; tensor var_3429 = linear(bias = linear_0_bias_0, weight = layers_7_per_layer_input_gate_weight_palettized, x = hidden_states_111_cast_fp16)[name = string("linear_14")]; string gated_15_mode_0 = const()[name = string("gated_15_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_15 = gelu(mode = gated_15_mode_0, x = var_3429)[name = string("gated_15")]; tensor var_3446_begin_0 = const()[name = string("op_3446_begin_0"), val = tensor([0, 0, 7936])]; tensor var_3446_end_0 = const()[name = string("op_3446_end_0"), val = tensor([1, 1, 8192])]; tensor var_3446_end_mask_0 = const()[name = string("op_3446_end_mask_0"), val = tensor([true, true, false])]; tensor var_3446_cast_fp16 = slice_by_index(begin = var_3446_begin_0, end = var_3446_end_0, end_mask = var_3446_end_mask_0, x = per_layer_combined)[name = string("op_3446_cast_fp16")]; tensor input_189_cast_fp16 = mul(x = gated_15, y = var_3446_cast_fp16)[name = string("input_189_cast_fp16")]; tensor layers_7_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(353193216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(353389888))))[name = string("layers_7_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_15_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_7_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_189_cast_fp16)[name = string("linear_15_cast_fp16")]; int32 var_3455 = const()[name = string("op_3455"), val = int32(-1)]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3461_cast_fp16 = mul(x = linear_15_cast_fp16, y = const_102_promoted_to_fp16)[name = string("op_3461_cast_fp16")]; bool input_191_interleave_0 = const()[name = string("input_191_interleave_0"), val = bool(false)]; tensor input_191_cast_fp16 = concat(axis = var_3455, interleave = input_191_interleave_0, values = (linear_15_cast_fp16, var_3461_cast_fp16))[name = string("input_191_cast_fp16")]; tensor normed_189_axes_0 = const()[name = string("normed_189_axes_0"), val = tensor([-1])]; fp16 var_3453_to_fp16 = const()[name = string("op_3453_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_189_cast_fp16 = layer_norm(axes = normed_189_axes_0, epsilon = var_3453_to_fp16, x = input_191_cast_fp16)[name = string("normed_189_cast_fp16")]; tensor var_3466_split_sizes_0 = const()[name = string("op_3466_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3466_axis_0 = const()[name = string("op_3466_axis_0"), val = int32(-1)]; tensor var_3466_cast_fp16_0, tensor var_3466_cast_fp16_1 = split(axis = var_3466_axis_0, split_sizes = var_3466_split_sizes_0, x = normed_189_cast_fp16)[name = string("op_3466_cast_fp16")]; tensor const_103_to_fp16 = const()[name = string("const_103_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(353391488)))]; tensor var_3469_cast_fp16 = mul(x = var_3466_cast_fp16_0, y = const_103_to_fp16)[name = string("op_3469_cast_fp16")]; tensor hidden_states_115_cast_fp16 = add(x = hidden_states_111_cast_fp16, y = var_3469_cast_fp16)[name = string("hidden_states_115_cast_fp16")]; tensor layers_7_layer_scalar_to_fp16 = const()[name = string("layers_7_layer_scalar_to_fp16"), val = tensor([0x1.a8p-1])]; tensor x_191_cast_fp16 = mul(x = hidden_states_115_cast_fp16, y = layers_7_layer_scalar_to_fp16)[name = string("x_191_cast_fp16")]; int32 var_3477 = const()[name = string("op_3477"), val = int32(-1)]; fp16 const_104_promoted_to_fp16 = const()[name = string("const_104_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3483_cast_fp16 = mul(x = x_191_cast_fp16, y = const_104_promoted_to_fp16)[name = string("op_3483_cast_fp16")]; bool input_193_interleave_0 = const()[name = string("input_193_interleave_0"), val = bool(false)]; tensor input_193_cast_fp16 = concat(axis = var_3477, interleave = input_193_interleave_0, values = (x_191_cast_fp16, var_3483_cast_fp16))[name = string("input_193_cast_fp16")]; tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; fp16 var_3475_to_fp16 = const()[name = string("op_3475_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_3475_to_fp16, x = input_193_cast_fp16)[name = string("normed_193_cast_fp16")]; tensor var_3488_split_sizes_0 = const()[name = string("op_3488_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3488_axis_0 = const()[name = string("op_3488_axis_0"), val = int32(-1)]; tensor var_3488_cast_fp16_0, tensor var_3488_cast_fp16_1 = split(axis = var_3488_axis_0, split_sizes = var_3488_split_sizes_0, x = normed_193_cast_fp16)[name = string("op_3488_cast_fp16")]; tensor const_105_to_fp16 = const()[name = string("const_105_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(353394624)))]; tensor var_3491_cast_fp16 = mul(x = var_3488_cast_fp16_0, y = const_105_to_fp16)[name = string("op_3491_cast_fp16")]; tensor var_3499 = const()[name = string("op_3499"), val = tensor([0, 2, 1])]; tensor var_3502_axes_0 = const()[name = string("op_3502_axes_0"), val = tensor([2])]; tensor var_3500_cast_fp16 = transpose(perm = var_3499, x = var_3491_cast_fp16)[name = string("transpose_22")]; tensor var_3502_cast_fp16 = expand_dims(axes = var_3502_axes_0, x = var_3500_cast_fp16)[name = string("op_3502_cast_fp16")]; string var_3518_pad_type_0 = const()[name = string("op_3518_pad_type_0"), val = string("valid")]; tensor var_3518_strides_0 = const()[name = string("op_3518_strides_0"), val = tensor([1, 1])]; tensor var_3518_pad_0 = const()[name = string("op_3518_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3518_dilations_0 = const()[name = string("op_3518_dilations_0"), val = tensor([1, 1])]; int32 var_3518_groups_0 = const()[name = string("op_3518_groups_0"), val = int32(1)]; tensor var_3518 = conv(dilations = var_3518_dilations_0, groups = var_3518_groups_0, pad = var_3518_pad_0, pad_type = var_3518_pad_type_0, strides = var_3518_strides_0, weight = layers_8_self_attn_q_proj_weight_palettized, x = var_3502_cast_fp16)[name = string("op_3518")]; tensor var_3523 = const()[name = string("op_3523"), val = tensor([1, 8, 256, 1])]; tensor var_3524 = reshape(shape = var_3523, x = var_3518)[name = string("op_3524")]; tensor var_3529 = const()[name = string("op_3529"), val = tensor([0, 1, 3, 2])]; tensor var_3539 = const()[name = string("op_3539"), val = tensor([1, 8, 256])]; tensor var_3530 = transpose(perm = var_3529, x = var_3524)[name = string("transpose_21")]; tensor x_195 = reshape(shape = var_3539, x = var_3530)[name = string("x_195")]; int32 var_3545 = const()[name = string("op_3545"), val = int32(-1)]; fp16 const_106_promoted_to_fp16 = const()[name = string("const_106_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3551_cast_fp16 = mul(x = x_195, y = const_106_promoted_to_fp16)[name = string("op_3551_cast_fp16")]; bool input_197_interleave_0 = const()[name = string("input_197_interleave_0"), val = bool(false)]; tensor input_197_cast_fp16 = concat(axis = var_3545, interleave = input_197_interleave_0, values = (x_195, var_3551_cast_fp16))[name = string("input_197_cast_fp16")]; tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; fp16 var_3543_to_fp16 = const()[name = string("op_3543_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_3543_to_fp16, x = input_197_cast_fp16)[name = string("normed_197_cast_fp16")]; tensor var_3556_split_sizes_0 = const()[name = string("op_3556_split_sizes_0"), val = tensor([256, 256])]; int32 var_3556_axis_0 = const()[name = string("op_3556_axis_0"), val = int32(-1)]; tensor var_3556_cast_fp16_0, tensor var_3556_cast_fp16_1 = split(axis = var_3556_axis_0, split_sizes = var_3556_split_sizes_0, x = normed_197_cast_fp16)[name = string("op_3556_cast_fp16")]; tensor var_3559_cast_fp16 = mul(x = var_3556_cast_fp16_0, y = const_16_to_fp16)[name = string("op_3559_cast_fp16")]; tensor var_3565 = const()[name = string("op_3565"), val = tensor([1, 8, 1, 256])]; tensor q_51 = reshape(shape = var_3565, x = var_3559_cast_fp16)[name = string("q_51")]; tensor var_3567_cast_fp16 = mul(x = q_51, y = cos_s)[name = string("op_3567_cast_fp16")]; tensor var_3568_split_sizes_0 = const()[name = string("op_3568_split_sizes_0"), val = tensor([128, 128])]; int32 var_3568_axis_0 = const()[name = string("op_3568_axis_0"), val = int32(-1)]; tensor var_3568_0, tensor var_3568_1 = split(axis = var_3568_axis_0, split_sizes = var_3568_split_sizes_0, x = q_51)[name = string("op_3568")]; fp16 const_108_promoted = const()[name = string("const_108_promoted"), val = fp16(-0x1p+0)]; tensor var_3570 = mul(x = var_3568_1, y = const_108_promoted)[name = string("op_3570")]; int32 var_3572 = const()[name = string("op_3572"), val = int32(-1)]; bool var_3573_interleave_0 = const()[name = string("op_3573_interleave_0"), val = bool(false)]; tensor var_3573 = concat(axis = var_3572, interleave = var_3573_interleave_0, values = (var_3570, var_3568_0))[name = string("op_3573")]; tensor var_3574_cast_fp16 = mul(x = var_3573, y = sin_s)[name = string("op_3574_cast_fp16")]; tensor q_53_cast_fp16 = add(x = var_3567_cast_fp16, y = var_3574_cast_fp16)[name = string("q_53_cast_fp16")]; bool var_3598_transpose_x_0 = const()[name = string("op_3598_transpose_x_0"), val = bool(false)]; bool var_3598_transpose_y_0 = const()[name = string("op_3598_transpose_y_0"), val = bool(false)]; tensor var_3598_cast_fp16 = matmul(transpose_x = var_3598_transpose_x_0, transpose_y = var_3598_transpose_y_0, x = q_53_cast_fp16, y = transpose_45_cast_fp16)[name = string("op_3598_cast_fp16")]; tensor var_3605_cast_fp16 = add(x = var_3598_cast_fp16, y = causal_mask)[name = string("op_3605_cast_fp16")]; int32 var_3606 = const()[name = string("op_3606"), val = int32(-1)]; tensor var_3608_cast_fp16 = softmax(axis = var_3606, x = var_3605_cast_fp16)[name = string("op_3608_cast_fp16")]; bool var_3624_transpose_x_0 = const()[name = string("op_3624_transpose_x_0"), val = bool(false)]; bool var_3624_transpose_y_0 = const()[name = string("op_3624_transpose_y_0"), val = bool(false)]; tensor var_3624_cast_fp16 = matmul(transpose_x = var_3624_transpose_x_0, transpose_y = var_3624_transpose_y_0, x = var_3608_cast_fp16, y = Ve_3_cast_fp16)[name = string("op_3624_cast_fp16")]; tensor var_3634 = const()[name = string("op_3634"), val = tensor([0, 2, 1, 3])]; tensor var_3641 = const()[name = string("op_3641"), val = tensor([1, 1, -1])]; tensor var_3635 = transpose(perm = var_3634, x = var_3624_cast_fp16)[name = string("transpose_20")]; tensor var_3642 = reshape(shape = var_3641, x = var_3635)[name = string("op_3642")]; tensor var_3646 = const()[name = string("op_3646"), val = tensor([0, 2, 1])]; tensor squeeze_8_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(353397760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354970688))))[name = string("squeeze_8_palettized")]; string var_3662_pad_type_0 = const()[name = string("op_3662_pad_type_0"), val = string("valid")]; int32 var_3662_groups_0 = const()[name = string("op_3662_groups_0"), val = int32(1)]; tensor var_3662_strides_0 = const()[name = string("op_3662_strides_0"), val = tensor([1])]; tensor var_3662_pad_0 = const()[name = string("op_3662_pad_0"), val = tensor([0, 0])]; tensor var_3662_dilations_0 = const()[name = string("op_3662_dilations_0"), val = tensor([1])]; tensor var_3647 = transpose(perm = var_3646, x = var_3642)[name = string("transpose_19")]; tensor var_3662 = conv(dilations = var_3662_dilations_0, groups = var_3662_groups_0, pad = var_3662_pad_0, pad_type = var_3662_pad_type_0, strides = var_3662_strides_0, weight = squeeze_8_palettized, x = var_3647)[name = string("op_3662")]; tensor var_3666 = const()[name = string("op_3666"), val = tensor([0, 2, 1])]; int32 var_3672 = const()[name = string("op_3672"), val = int32(-1)]; fp16 const_109_promoted_to_fp16 = const()[name = string("const_109_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_199 = transpose(perm = var_3666, x = var_3662)[name = string("transpose_18")]; tensor var_3678_cast_fp16 = mul(x = x_199, y = const_109_promoted_to_fp16)[name = string("op_3678_cast_fp16")]; bool input_201_interleave_0 = const()[name = string("input_201_interleave_0"), val = bool(false)]; tensor input_201_cast_fp16 = concat(axis = var_3672, interleave = input_201_interleave_0, values = (x_199, var_3678_cast_fp16))[name = string("input_201_cast_fp16")]; tensor normed_201_axes_0 = const()[name = string("normed_201_axes_0"), val = tensor([-1])]; fp16 var_3670_to_fp16 = const()[name = string("op_3670_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_201_cast_fp16 = layer_norm(axes = normed_201_axes_0, epsilon = var_3670_to_fp16, x = input_201_cast_fp16)[name = string("normed_201_cast_fp16")]; tensor var_3683_split_sizes_0 = const()[name = string("op_3683_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3683_axis_0 = const()[name = string("op_3683_axis_0"), val = int32(-1)]; tensor var_3683_cast_fp16_0, tensor var_3683_cast_fp16_1 = split(axis = var_3683_axis_0, split_sizes = var_3683_split_sizes_0, x = normed_201_cast_fp16)[name = string("op_3683_cast_fp16")]; tensor const_110_to_fp16 = const()[name = string("const_110_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354972288)))]; tensor var_3686_cast_fp16 = mul(x = var_3683_cast_fp16_0, y = const_110_to_fp16)[name = string("op_3686_cast_fp16")]; tensor x_203_cast_fp16 = add(x = x_191_cast_fp16, y = var_3686_cast_fp16)[name = string("x_203_cast_fp16")]; int32 var_3693 = const()[name = string("op_3693"), val = int32(-1)]; fp16 const_111_promoted_to_fp16 = const()[name = string("const_111_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3699_cast_fp16 = mul(x = x_203_cast_fp16, y = const_111_promoted_to_fp16)[name = string("op_3699_cast_fp16")]; bool input_203_interleave_0 = const()[name = string("input_203_interleave_0"), val = bool(false)]; tensor input_203_cast_fp16 = concat(axis = var_3693, interleave = input_203_interleave_0, values = (x_203_cast_fp16, var_3699_cast_fp16))[name = string("input_203_cast_fp16")]; tensor normed_205_axes_0 = const()[name = string("normed_205_axes_0"), val = tensor([-1])]; fp16 var_3691_to_fp16 = const()[name = string("op_3691_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_205_cast_fp16 = layer_norm(axes = normed_205_axes_0, epsilon = var_3691_to_fp16, x = input_203_cast_fp16)[name = string("normed_205_cast_fp16")]; tensor var_3704_split_sizes_0 = const()[name = string("op_3704_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3704_axis_0 = const()[name = string("op_3704_axis_0"), val = int32(-1)]; tensor var_3704_cast_fp16_0, tensor var_3704_cast_fp16_1 = split(axis = var_3704_axis_0, split_sizes = var_3704_split_sizes_0, x = normed_205_cast_fp16)[name = string("op_3704_cast_fp16")]; tensor const_112_to_fp16 = const()[name = string("const_112_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354975424)))]; tensor var_3707_cast_fp16 = mul(x = var_3704_cast_fp16_0, y = const_112_to_fp16)[name = string("op_3707_cast_fp16")]; tensor var_3720 = const()[name = string("op_3720"), val = tensor([0, 2, 1])]; tensor input_205_axes_0 = const()[name = string("input_205_axes_0"), val = tensor([2])]; tensor var_3721 = transpose(perm = var_3720, x = var_3707_cast_fp16)[name = string("transpose_17")]; tensor input_205 = expand_dims(axes = input_205_axes_0, x = var_3721)[name = string("input_205")]; string var_3734_pad_type_0 = const()[name = string("op_3734_pad_type_0"), val = string("valid")]; tensor var_3734_strides_0 = const()[name = string("op_3734_strides_0"), val = tensor([1, 1])]; tensor var_3734_pad_0 = const()[name = string("op_3734_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3734_dilations_0 = const()[name = string("op_3734_dilations_0"), val = tensor([1, 1])]; int32 var_3734_groups_0 = const()[name = string("op_3734_groups_0"), val = int32(1)]; tensor var_3734 = conv(dilations = var_3734_dilations_0, groups = var_3734_groups_0, pad = var_3734_pad_0, pad_type = var_3734_pad_type_0, strides = var_3734_strides_0, weight = layers_8_mlp_gate_proj_weight_palettized, x = input_205)[name = string("op_3734")]; string var_3736_mode_0 = const()[name = string("op_3736_mode_0"), val = string("TANH_APPROXIMATION")]; tensor var_3736 = gelu(mode = var_3736_mode_0, x = var_3734)[name = string("op_3736")]; string var_3747_pad_type_0 = const()[name = string("op_3747_pad_type_0"), val = string("valid")]; tensor var_3747_strides_0 = const()[name = string("op_3747_strides_0"), val = tensor([1, 1])]; tensor var_3747_pad_0 = const()[name = string("op_3747_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3747_dilations_0 = const()[name = string("op_3747_dilations_0"), val = tensor([1, 1])]; int32 var_3747_groups_0 = const()[name = string("op_3747_groups_0"), val = int32(1)]; tensor var_3747 = conv(dilations = var_3747_dilations_0, groups = var_3747_groups_0, pad = var_3747_pad_0, pad_type = var_3747_pad_type_0, strides = var_3747_strides_0, weight = layers_8_mlp_up_proj_weight_palettized, x = input_205)[name = string("op_3747")]; tensor input_207 = mul(x = var_3736, y = var_3747)[name = string("input_207")]; string var_3759_pad_type_0 = const()[name = string("op_3759_pad_type_0"), val = string("valid")]; tensor var_3759_strides_0 = const()[name = string("op_3759_strides_0"), val = tensor([1, 1])]; tensor var_3759_pad_0 = const()[name = string("op_3759_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3759_dilations_0 = const()[name = string("op_3759_dilations_0"), val = tensor([1, 1])]; int32 var_3759_groups_0 = const()[name = string("op_3759_groups_0"), val = int32(1)]; tensor var_3759 = conv(dilations = var_3759_dilations_0, groups = var_3759_groups_0, pad = var_3759_pad_0, pad_type = var_3759_pad_type_0, strides = var_3759_strides_0, weight = layers_8_mlp_down_proj_weight_palettized, x = input_207)[name = string("op_3759")]; tensor var_3761_axes_0 = const()[name = string("op_3761_axes_0"), val = tensor([2])]; tensor var_3761 = squeeze(axes = var_3761_axes_0, x = var_3759)[name = string("op_3761")]; tensor var_3765 = const()[name = string("op_3765"), val = tensor([0, 2, 1])]; int32 var_3771 = const()[name = string("op_3771"), val = int32(-1)]; fp16 const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_207 = transpose(perm = var_3765, x = var_3761)[name = string("transpose_16")]; tensor var_3777_cast_fp16 = mul(x = x_207, y = const_113_promoted_to_fp16)[name = string("op_3777_cast_fp16")]; bool input_209_interleave_0 = const()[name = string("input_209_interleave_0"), val = bool(false)]; tensor input_209_cast_fp16 = concat(axis = var_3771, interleave = input_209_interleave_0, values = (x_207, var_3777_cast_fp16))[name = string("input_209_cast_fp16")]; tensor normed_209_axes_0 = const()[name = string("normed_209_axes_0"), val = tensor([-1])]; fp16 var_3769_to_fp16 = const()[name = string("op_3769_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_209_cast_fp16 = layer_norm(axes = normed_209_axes_0, epsilon = var_3769_to_fp16, x = input_209_cast_fp16)[name = string("normed_209_cast_fp16")]; tensor var_3782_split_sizes_0 = const()[name = string("op_3782_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3782_axis_0 = const()[name = string("op_3782_axis_0"), val = int32(-1)]; tensor var_3782_cast_fp16_0, tensor var_3782_cast_fp16_1 = split(axis = var_3782_axis_0, split_sizes = var_3782_split_sizes_0, x = normed_209_cast_fp16)[name = string("op_3782_cast_fp16")]; tensor const_114_to_fp16 = const()[name = string("const_114_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354978560)))]; tensor var_3785_cast_fp16 = mul(x = var_3782_cast_fp16_0, y = const_114_to_fp16)[name = string("op_3785_cast_fp16")]; tensor hidden_states_125_cast_fp16 = add(x = x_203_cast_fp16, y = var_3785_cast_fp16)[name = string("hidden_states_125_cast_fp16")]; tensor var_3796 = linear(bias = linear_0_bias_0, weight = layers_8_per_layer_input_gate_weight_palettized, x = hidden_states_125_cast_fp16)[name = string("linear_16")]; string gated_17_mode_0 = const()[name = string("gated_17_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_17 = gelu(mode = gated_17_mode_0, x = var_3796)[name = string("gated_17")]; tensor var_3813_begin_0 = const()[name = string("op_3813_begin_0"), val = tensor([0, 0, 8192])]; tensor var_3813_end_0 = const()[name = string("op_3813_end_0"), val = tensor([1, 1, 8448])]; tensor var_3813_end_mask_0 = const()[name = string("op_3813_end_mask_0"), val = tensor([true, true, false])]; tensor var_3813_cast_fp16 = slice_by_index(begin = var_3813_begin_0, end = var_3813_end_0, end_mask = var_3813_end_mask_0, x = per_layer_combined)[name = string("op_3813_cast_fp16")]; tensor input_213_cast_fp16 = mul(x = gated_17, y = var_3813_cast_fp16)[name = string("input_213_cast_fp16")]; tensor layers_8_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354981696))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355178368))))[name = string("layers_8_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_17_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_8_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_213_cast_fp16)[name = string("linear_17_cast_fp16")]; int32 var_3822 = const()[name = string("op_3822"), val = int32(-1)]; fp16 const_115_promoted_to_fp16 = const()[name = string("const_115_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3828_cast_fp16 = mul(x = linear_17_cast_fp16, y = const_115_promoted_to_fp16)[name = string("op_3828_cast_fp16")]; bool input_215_interleave_0 = const()[name = string("input_215_interleave_0"), val = bool(false)]; tensor input_215_cast_fp16 = concat(axis = var_3822, interleave = input_215_interleave_0, values = (linear_17_cast_fp16, var_3828_cast_fp16))[name = string("input_215_cast_fp16")]; tensor normed_213_axes_0 = const()[name = string("normed_213_axes_0"), val = tensor([-1])]; fp16 var_3820_to_fp16 = const()[name = string("op_3820_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_213_cast_fp16 = layer_norm(axes = normed_213_axes_0, epsilon = var_3820_to_fp16, x = input_215_cast_fp16)[name = string("normed_213_cast_fp16")]; tensor var_3833_split_sizes_0 = const()[name = string("op_3833_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3833_axis_0 = const()[name = string("op_3833_axis_0"), val = int32(-1)]; tensor var_3833_cast_fp16_0, tensor var_3833_cast_fp16_1 = split(axis = var_3833_axis_0, split_sizes = var_3833_split_sizes_0, x = normed_213_cast_fp16)[name = string("op_3833_cast_fp16")]; tensor const_116_to_fp16 = const()[name = string("const_116_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355179968)))]; tensor var_3836_cast_fp16 = mul(x = var_3833_cast_fp16_0, y = const_116_to_fp16)[name = string("op_3836_cast_fp16")]; tensor hidden_states_129_cast_fp16 = add(x = hidden_states_125_cast_fp16, y = var_3836_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; tensor layers_8_layer_scalar_to_fp16 = const()[name = string("layers_8_layer_scalar_to_fp16"), val = tensor([0x1.bep-1])]; tensor x_215_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = layers_8_layer_scalar_to_fp16)[name = string("x_215_cast_fp16")]; int32 var_3844 = const()[name = string("op_3844"), val = int32(-1)]; fp16 const_117_promoted_to_fp16 = const()[name = string("const_117_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3850_cast_fp16 = mul(x = x_215_cast_fp16, y = const_117_promoted_to_fp16)[name = string("op_3850_cast_fp16")]; bool input_217_interleave_0 = const()[name = string("input_217_interleave_0"), val = bool(false)]; tensor input_217_cast_fp16 = concat(axis = var_3844, interleave = input_217_interleave_0, values = (x_215_cast_fp16, var_3850_cast_fp16))[name = string("input_217_cast_fp16")]; tensor normed_217_axes_0 = const()[name = string("normed_217_axes_0"), val = tensor([-1])]; fp16 var_3842_to_fp16 = const()[name = string("op_3842_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_217_cast_fp16 = layer_norm(axes = normed_217_axes_0, epsilon = var_3842_to_fp16, x = input_217_cast_fp16)[name = string("normed_217_cast_fp16")]; tensor var_3855_split_sizes_0 = const()[name = string("op_3855_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3855_axis_0 = const()[name = string("op_3855_axis_0"), val = int32(-1)]; tensor var_3855_cast_fp16_0, tensor var_3855_cast_fp16_1 = split(axis = var_3855_axis_0, split_sizes = var_3855_split_sizes_0, x = normed_217_cast_fp16)[name = string("op_3855_cast_fp16")]; tensor const_118_to_fp16 = const()[name = string("const_118_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355183104)))]; tensor var_3858_cast_fp16 = mul(x = var_3855_cast_fp16_0, y = const_118_to_fp16)[name = string("op_3858_cast_fp16")]; tensor var_3866 = const()[name = string("op_3866"), val = tensor([0, 2, 1])]; tensor var_3869_axes_0 = const()[name = string("op_3869_axes_0"), val = tensor([2])]; tensor var_3867_cast_fp16 = transpose(perm = var_3866, x = var_3858_cast_fp16)[name = string("transpose_15")]; tensor var_3869_cast_fp16 = expand_dims(axes = var_3869_axes_0, x = var_3867_cast_fp16)[name = string("op_3869_cast_fp16")]; string var_3885_pad_type_0 = const()[name = string("op_3885_pad_type_0"), val = string("valid")]; tensor var_3885_strides_0 = const()[name = string("op_3885_strides_0"), val = tensor([1, 1])]; tensor var_3885_pad_0 = const()[name = string("op_3885_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3885_dilations_0 = const()[name = string("op_3885_dilations_0"), val = tensor([1, 1])]; int32 var_3885_groups_0 = const()[name = string("op_3885_groups_0"), val = int32(1)]; tensor var_3885 = conv(dilations = var_3885_dilations_0, groups = var_3885_groups_0, pad = var_3885_pad_0, pad_type = var_3885_pad_type_0, strides = var_3885_strides_0, weight = layers_9_self_attn_q_proj_weight_palettized, x = var_3869_cast_fp16)[name = string("op_3885")]; tensor var_3890 = const()[name = string("op_3890"), val = tensor([1, 8, 256, 1])]; tensor var_3891 = reshape(shape = var_3890, x = var_3885)[name = string("op_3891")]; tensor var_3896 = const()[name = string("op_3896"), val = tensor([0, 1, 3, 2])]; tensor var_3906 = const()[name = string("op_3906"), val = tensor([1, 8, 256])]; tensor var_3897 = transpose(perm = var_3896, x = var_3891)[name = string("transpose_14")]; tensor x_219 = reshape(shape = var_3906, x = var_3897)[name = string("x_219")]; int32 var_3912 = const()[name = string("op_3912"), val = int32(-1)]; fp16 const_119_promoted_to_fp16 = const()[name = string("const_119_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3918_cast_fp16 = mul(x = x_219, y = const_119_promoted_to_fp16)[name = string("op_3918_cast_fp16")]; bool input_221_interleave_0 = const()[name = string("input_221_interleave_0"), val = bool(false)]; tensor input_221_cast_fp16 = concat(axis = var_3912, interleave = input_221_interleave_0, values = (x_219, var_3918_cast_fp16))[name = string("input_221_cast_fp16")]; tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; fp16 var_3910_to_fp16 = const()[name = string("op_3910_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_3910_to_fp16, x = input_221_cast_fp16)[name = string("normed_221_cast_fp16")]; tensor var_3923_split_sizes_0 = const()[name = string("op_3923_split_sizes_0"), val = tensor([256, 256])]; int32 var_3923_axis_0 = const()[name = string("op_3923_axis_0"), val = int32(-1)]; tensor var_3923_cast_fp16_0, tensor var_3923_cast_fp16_1 = split(axis = var_3923_axis_0, split_sizes = var_3923_split_sizes_0, x = normed_221_cast_fp16)[name = string("op_3923_cast_fp16")]; tensor var_3926_cast_fp16 = mul(x = var_3923_cast_fp16_0, y = const_16_to_fp16)[name = string("op_3926_cast_fp16")]; tensor var_3932 = const()[name = string("op_3932"), val = tensor([1, 8, 1, 256])]; tensor q_57 = reshape(shape = var_3932, x = var_3926_cast_fp16)[name = string("q_57")]; tensor var_3934_cast_fp16 = mul(x = q_57, y = cos_s)[name = string("op_3934_cast_fp16")]; tensor var_3935_split_sizes_0 = const()[name = string("op_3935_split_sizes_0"), val = tensor([128, 128])]; int32 var_3935_axis_0 = const()[name = string("op_3935_axis_0"), val = int32(-1)]; tensor var_3935_0, tensor var_3935_1 = split(axis = var_3935_axis_0, split_sizes = var_3935_split_sizes_0, x = q_57)[name = string("op_3935")]; fp16 const_121_promoted = const()[name = string("const_121_promoted"), val = fp16(-0x1p+0)]; tensor var_3937 = mul(x = var_3935_1, y = const_121_promoted)[name = string("op_3937")]; int32 var_3939 = const()[name = string("op_3939"), val = int32(-1)]; bool var_3940_interleave_0 = const()[name = string("op_3940_interleave_0"), val = bool(false)]; tensor var_3940 = concat(axis = var_3939, interleave = var_3940_interleave_0, values = (var_3937, var_3935_0))[name = string("op_3940")]; tensor var_3941_cast_fp16 = mul(x = var_3940, y = sin_s)[name = string("op_3941_cast_fp16")]; tensor q_59_cast_fp16 = add(x = var_3934_cast_fp16, y = var_3941_cast_fp16)[name = string("q_59_cast_fp16")]; bool var_3965_transpose_x_0 = const()[name = string("op_3965_transpose_x_0"), val = bool(false)]; bool var_3965_transpose_y_0 = const()[name = string("op_3965_transpose_y_0"), val = bool(false)]; tensor var_3965_cast_fp16 = matmul(transpose_x = var_3965_transpose_x_0, transpose_y = var_3965_transpose_y_0, x = q_59_cast_fp16, y = transpose_45_cast_fp16)[name = string("op_3965_cast_fp16")]; tensor var_3972_cast_fp16 = add(x = var_3965_cast_fp16, y = causal_mask)[name = string("op_3972_cast_fp16")]; int32 var_3973 = const()[name = string("op_3973"), val = int32(-1)]; tensor var_3975_cast_fp16 = softmax(axis = var_3973, x = var_3972_cast_fp16)[name = string("op_3975_cast_fp16")]; bool var_3991_transpose_x_0 = const()[name = string("op_3991_transpose_x_0"), val = bool(false)]; bool var_3991_transpose_y_0 = const()[name = string("op_3991_transpose_y_0"), val = bool(false)]; tensor var_3991_cast_fp16 = matmul(transpose_x = var_3991_transpose_x_0, transpose_y = var_3991_transpose_y_0, x = var_3975_cast_fp16, y = Ve_3_cast_fp16)[name = string("op_3991_cast_fp16")]; tensor var_4001 = const()[name = string("op_4001"), val = tensor([0, 2, 1, 3])]; tensor var_4008 = const()[name = string("op_4008"), val = tensor([1, 1, -1])]; tensor var_4002 = transpose(perm = var_4001, x = var_3991_cast_fp16)[name = string("transpose_13")]; tensor var_4009 = reshape(shape = var_4008, x = var_4002)[name = string("op_4009")]; tensor var_4013 = const()[name = string("op_4013"), val = tensor([0, 2, 1])]; tensor squeeze_9_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355186240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356759168))))[name = string("squeeze_9_palettized")]; string var_4029_pad_type_0 = const()[name = string("op_4029_pad_type_0"), val = string("valid")]; int32 var_4029_groups_0 = const()[name = string("op_4029_groups_0"), val = int32(1)]; tensor var_4029_strides_0 = const()[name = string("op_4029_strides_0"), val = tensor([1])]; tensor var_4029_pad_0 = const()[name = string("op_4029_pad_0"), val = tensor([0, 0])]; tensor var_4029_dilations_0 = const()[name = string("op_4029_dilations_0"), val = tensor([1])]; tensor var_4014 = transpose(perm = var_4013, x = var_4009)[name = string("transpose_12")]; tensor var_4029 = conv(dilations = var_4029_dilations_0, groups = var_4029_groups_0, pad = var_4029_pad_0, pad_type = var_4029_pad_type_0, strides = var_4029_strides_0, weight = squeeze_9_palettized, x = var_4014)[name = string("op_4029")]; tensor var_4033 = const()[name = string("op_4033"), val = tensor([0, 2, 1])]; int32 var_4039 = const()[name = string("op_4039"), val = int32(-1)]; fp16 const_122_promoted_to_fp16 = const()[name = string("const_122_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_223 = transpose(perm = var_4033, x = var_4029)[name = string("transpose_11")]; tensor var_4045_cast_fp16 = mul(x = x_223, y = const_122_promoted_to_fp16)[name = string("op_4045_cast_fp16")]; bool input_225_interleave_0 = const()[name = string("input_225_interleave_0"), val = bool(false)]; tensor input_225_cast_fp16 = concat(axis = var_4039, interleave = input_225_interleave_0, values = (x_223, var_4045_cast_fp16))[name = string("input_225_cast_fp16")]; tensor normed_225_axes_0 = const()[name = string("normed_225_axes_0"), val = tensor([-1])]; fp16 var_4037_to_fp16 = const()[name = string("op_4037_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_225_cast_fp16 = layer_norm(axes = normed_225_axes_0, epsilon = var_4037_to_fp16, x = input_225_cast_fp16)[name = string("normed_225_cast_fp16")]; tensor var_4050_split_sizes_0 = const()[name = string("op_4050_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4050_axis_0 = const()[name = string("op_4050_axis_0"), val = int32(-1)]; tensor var_4050_cast_fp16_0, tensor var_4050_cast_fp16_1 = split(axis = var_4050_axis_0, split_sizes = var_4050_split_sizes_0, x = normed_225_cast_fp16)[name = string("op_4050_cast_fp16")]; tensor const_123_to_fp16 = const()[name = string("const_123_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356760768)))]; tensor var_4053_cast_fp16 = mul(x = var_4050_cast_fp16_0, y = const_123_to_fp16)[name = string("op_4053_cast_fp16")]; tensor x_227_cast_fp16 = add(x = x_215_cast_fp16, y = var_4053_cast_fp16)[name = string("x_227_cast_fp16")]; int32 var_4060 = const()[name = string("op_4060"), val = int32(-1)]; fp16 const_124_promoted_to_fp16 = const()[name = string("const_124_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4066_cast_fp16 = mul(x = x_227_cast_fp16, y = const_124_promoted_to_fp16)[name = string("op_4066_cast_fp16")]; bool input_227_interleave_0 = const()[name = string("input_227_interleave_0"), val = bool(false)]; tensor input_227_cast_fp16 = concat(axis = var_4060, interleave = input_227_interleave_0, values = (x_227_cast_fp16, var_4066_cast_fp16))[name = string("input_227_cast_fp16")]; tensor normed_229_axes_0 = const()[name = string("normed_229_axes_0"), val = tensor([-1])]; fp16 var_4058_to_fp16 = const()[name = string("op_4058_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_229_cast_fp16 = layer_norm(axes = normed_229_axes_0, epsilon = var_4058_to_fp16, x = input_227_cast_fp16)[name = string("normed_229_cast_fp16")]; tensor var_4071_split_sizes_0 = const()[name = string("op_4071_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4071_axis_0 = const()[name = string("op_4071_axis_0"), val = int32(-1)]; tensor var_4071_cast_fp16_0, tensor var_4071_cast_fp16_1 = split(axis = var_4071_axis_0, split_sizes = var_4071_split_sizes_0, x = normed_229_cast_fp16)[name = string("op_4071_cast_fp16")]; tensor const_125_to_fp16 = const()[name = string("const_125_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356763904)))]; tensor var_4074_cast_fp16 = mul(x = var_4071_cast_fp16_0, y = const_125_to_fp16)[name = string("op_4074_cast_fp16")]; tensor var_4087 = const()[name = string("op_4087"), val = tensor([0, 2, 1])]; tensor input_229_axes_0 = const()[name = string("input_229_axes_0"), val = tensor([2])]; tensor var_4088 = transpose(perm = var_4087, x = var_4074_cast_fp16)[name = string("transpose_10")]; tensor input_229 = expand_dims(axes = input_229_axes_0, x = var_4088)[name = string("input_229")]; string var_4101_pad_type_0 = const()[name = string("op_4101_pad_type_0"), val = string("valid")]; tensor var_4101_strides_0 = const()[name = string("op_4101_strides_0"), val = tensor([1, 1])]; tensor var_4101_pad_0 = const()[name = string("op_4101_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4101_dilations_0 = const()[name = string("op_4101_dilations_0"), val = tensor([1, 1])]; int32 var_4101_groups_0 = const()[name = string("op_4101_groups_0"), val = int32(1)]; tensor var_4101 = conv(dilations = var_4101_dilations_0, groups = var_4101_groups_0, pad = var_4101_pad_0, pad_type = var_4101_pad_type_0, strides = var_4101_strides_0, weight = layers_9_mlp_gate_proj_weight_palettized, x = input_229)[name = string("op_4101")]; string var_4103_mode_0 = const()[name = string("op_4103_mode_0"), val = string("TANH_APPROXIMATION")]; tensor var_4103 = gelu(mode = var_4103_mode_0, x = var_4101)[name = string("op_4103")]; string var_4114_pad_type_0 = const()[name = string("op_4114_pad_type_0"), val = string("valid")]; tensor var_4114_strides_0 = const()[name = string("op_4114_strides_0"), val = tensor([1, 1])]; tensor var_4114_pad_0 = const()[name = string("op_4114_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4114_dilations_0 = const()[name = string("op_4114_dilations_0"), val = tensor([1, 1])]; int32 var_4114_groups_0 = const()[name = string("op_4114_groups_0"), val = int32(1)]; tensor var_4114 = conv(dilations = var_4114_dilations_0, groups = var_4114_groups_0, pad = var_4114_pad_0, pad_type = var_4114_pad_type_0, strides = var_4114_strides_0, weight = layers_9_mlp_up_proj_weight_palettized, x = input_229)[name = string("op_4114")]; tensor input_231 = mul(x = var_4103, y = var_4114)[name = string("input_231")]; string var_4126_pad_type_0 = const()[name = string("op_4126_pad_type_0"), val = string("valid")]; tensor var_4126_strides_0 = const()[name = string("op_4126_strides_0"), val = tensor([1, 1])]; tensor var_4126_pad_0 = const()[name = string("op_4126_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4126_dilations_0 = const()[name = string("op_4126_dilations_0"), val = tensor([1, 1])]; int32 var_4126_groups_0 = const()[name = string("op_4126_groups_0"), val = int32(1)]; tensor var_4126 = conv(dilations = var_4126_dilations_0, groups = var_4126_groups_0, pad = var_4126_pad_0, pad_type = var_4126_pad_type_0, strides = var_4126_strides_0, weight = layers_9_mlp_down_proj_weight_palettized, x = input_231)[name = string("op_4126")]; tensor var_4128_axes_0 = const()[name = string("op_4128_axes_0"), val = tensor([2])]; tensor var_4128 = squeeze(axes = var_4128_axes_0, x = var_4126)[name = string("op_4128")]; tensor var_4132 = const()[name = string("op_4132"), val = tensor([0, 2, 1])]; int32 var_4138 = const()[name = string("op_4138"), val = int32(-1)]; fp16 const_126_promoted_to_fp16 = const()[name = string("const_126_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_231 = transpose(perm = var_4132, x = var_4128)[name = string("transpose_9")]; tensor var_4144_cast_fp16 = mul(x = x_231, y = const_126_promoted_to_fp16)[name = string("op_4144_cast_fp16")]; bool input_233_interleave_0 = const()[name = string("input_233_interleave_0"), val = bool(false)]; tensor input_233_cast_fp16 = concat(axis = var_4138, interleave = input_233_interleave_0, values = (x_231, var_4144_cast_fp16))[name = string("input_233_cast_fp16")]; tensor normed_233_axes_0 = const()[name = string("normed_233_axes_0"), val = tensor([-1])]; fp16 var_4136_to_fp16 = const()[name = string("op_4136_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_233_cast_fp16 = layer_norm(axes = normed_233_axes_0, epsilon = var_4136_to_fp16, x = input_233_cast_fp16)[name = string("normed_233_cast_fp16")]; tensor var_4149_split_sizes_0 = const()[name = string("op_4149_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4149_axis_0 = const()[name = string("op_4149_axis_0"), val = int32(-1)]; tensor var_4149_cast_fp16_0, tensor var_4149_cast_fp16_1 = split(axis = var_4149_axis_0, split_sizes = var_4149_split_sizes_0, x = normed_233_cast_fp16)[name = string("op_4149_cast_fp16")]; tensor const_127_to_fp16 = const()[name = string("const_127_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356767040)))]; tensor var_4152_cast_fp16 = mul(x = var_4149_cast_fp16_0, y = const_127_to_fp16)[name = string("op_4152_cast_fp16")]; tensor hidden_states_139_cast_fp16 = add(x = x_227_cast_fp16, y = var_4152_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; tensor var_4163 = linear(bias = linear_0_bias_0, weight = layers_9_per_layer_input_gate_weight_palettized, x = hidden_states_139_cast_fp16)[name = string("linear_18")]; string gated_19_mode_0 = const()[name = string("gated_19_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_19 = gelu(mode = gated_19_mode_0, x = var_4163)[name = string("gated_19")]; tensor var_4180_begin_0 = const()[name = string("op_4180_begin_0"), val = tensor([0, 0, 8448])]; tensor var_4180_end_0 = const()[name = string("op_4180_end_0"), val = tensor([1, 1, 8704])]; tensor var_4180_end_mask_0 = const()[name = string("op_4180_end_mask_0"), val = tensor([true, true, false])]; tensor var_4180_cast_fp16 = slice_by_index(begin = var_4180_begin_0, end = var_4180_end_0, end_mask = var_4180_end_mask_0, x = per_layer_combined)[name = string("op_4180_cast_fp16")]; tensor input_237_cast_fp16 = mul(x = gated_19, y = var_4180_cast_fp16)[name = string("input_237_cast_fp16")]; tensor layers_9_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356770176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356966848))))[name = string("layers_9_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_19_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_9_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_237_cast_fp16)[name = string("linear_19_cast_fp16")]; int32 var_4189 = const()[name = string("op_4189"), val = int32(-1)]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4195_cast_fp16 = mul(x = linear_19_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_4195_cast_fp16")]; bool input_239_interleave_0 = const()[name = string("input_239_interleave_0"), val = bool(false)]; tensor input_239_cast_fp16 = concat(axis = var_4189, interleave = input_239_interleave_0, values = (linear_19_cast_fp16, var_4195_cast_fp16))[name = string("input_239_cast_fp16")]; tensor normed_237_axes_0 = const()[name = string("normed_237_axes_0"), val = tensor([-1])]; fp16 var_4187_to_fp16 = const()[name = string("op_4187_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_237_cast_fp16 = layer_norm(axes = normed_237_axes_0, epsilon = var_4187_to_fp16, x = input_239_cast_fp16)[name = string("normed_237_cast_fp16")]; tensor var_4200_split_sizes_0 = const()[name = string("op_4200_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4200_axis_0 = const()[name = string("op_4200_axis_0"), val = int32(-1)]; tensor var_4200_cast_fp16_0, tensor var_4200_cast_fp16_1 = split(axis = var_4200_axis_0, split_sizes = var_4200_split_sizes_0, x = normed_237_cast_fp16)[name = string("op_4200_cast_fp16")]; tensor const_129_to_fp16 = const()[name = string("const_129_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356968448)))]; tensor var_4203_cast_fp16 = mul(x = var_4200_cast_fp16_0, y = const_129_to_fp16)[name = string("op_4203_cast_fp16")]; tensor hidden_states_143_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = var_4203_cast_fp16)[name = string("hidden_states_143_cast_fp16")]; tensor layers_9_layer_scalar_to_fp16 = const()[name = string("layers_9_layer_scalar_to_fp16"), val = tensor([0x1.64p-1])]; tensor x_239_cast_fp16 = mul(x = hidden_states_143_cast_fp16, y = layers_9_layer_scalar_to_fp16)[name = string("x_239_cast_fp16")]; int32 var_4211 = const()[name = string("op_4211"), val = int32(-1)]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4217_cast_fp16 = mul(x = x_239_cast_fp16, y = const_130_promoted_to_fp16)[name = string("op_4217_cast_fp16")]; bool input_241_interleave_0 = const()[name = string("input_241_interleave_0"), val = bool(false)]; tensor input_241_cast_fp16 = concat(axis = var_4211, interleave = input_241_interleave_0, values = (x_239_cast_fp16, var_4217_cast_fp16))[name = string("input_241_cast_fp16")]; tensor normed_241_axes_0 = const()[name = string("normed_241_axes_0"), val = tensor([-1])]; fp16 var_4209_to_fp16 = const()[name = string("op_4209_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_241_cast_fp16 = layer_norm(axes = normed_241_axes_0, epsilon = var_4209_to_fp16, x = input_241_cast_fp16)[name = string("normed_241_cast_fp16")]; tensor var_4222_split_sizes_0 = const()[name = string("op_4222_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4222_axis_0 = const()[name = string("op_4222_axis_0"), val = int32(-1)]; tensor var_4222_cast_fp16_0, tensor var_4222_cast_fp16_1 = split(axis = var_4222_axis_0, split_sizes = var_4222_split_sizes_0, x = normed_241_cast_fp16)[name = string("op_4222_cast_fp16")]; tensor const_131_to_fp16 = const()[name = string("const_131_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356971584)))]; tensor var_4225_cast_fp16 = mul(x = var_4222_cast_fp16_0, y = const_131_to_fp16)[name = string("op_4225_cast_fp16")]; tensor var_4233 = const()[name = string("op_4233"), val = tensor([0, 2, 1])]; tensor var_4236_axes_0 = const()[name = string("op_4236_axes_0"), val = tensor([2])]; tensor var_4234_cast_fp16 = transpose(perm = var_4233, x = var_4225_cast_fp16)[name = string("transpose_8")]; tensor var_4236_cast_fp16 = expand_dims(axes = var_4236_axes_0, x = var_4234_cast_fp16)[name = string("op_4236_cast_fp16")]; string var_4252_pad_type_0 = const()[name = string("op_4252_pad_type_0"), val = string("valid")]; tensor var_4252_strides_0 = const()[name = string("op_4252_strides_0"), val = tensor([1, 1])]; tensor var_4252_pad_0 = const()[name = string("op_4252_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4252_dilations_0 = const()[name = string("op_4252_dilations_0"), val = tensor([1, 1])]; int32 var_4252_groups_0 = const()[name = string("op_4252_groups_0"), val = int32(1)]; tensor var_4252 = conv(dilations = var_4252_dilations_0, groups = var_4252_groups_0, pad = var_4252_pad_0, pad_type = var_4252_pad_type_0, strides = var_4252_strides_0, weight = layers_10_self_attn_q_proj_weight_palettized, x = var_4236_cast_fp16)[name = string("op_4252")]; tensor var_4257 = const()[name = string("op_4257"), val = tensor([1, 8, 512, 1])]; tensor var_4258 = reshape(shape = var_4257, x = var_4252)[name = string("op_4258")]; tensor var_4263 = const()[name = string("op_4263"), val = tensor([0, 1, 3, 2])]; tensor var_4273 = const()[name = string("op_4273"), val = tensor([1, 8, 512])]; tensor var_4264 = transpose(perm = var_4263, x = var_4258)[name = string("transpose_7")]; tensor x_243 = reshape(shape = var_4273, x = var_4264)[name = string("x_243")]; int32 var_4279 = const()[name = string("op_4279"), val = int32(-1)]; fp16 const_132_promoted_to_fp16 = const()[name = string("const_132_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4285_cast_fp16 = mul(x = x_243, y = const_132_promoted_to_fp16)[name = string("op_4285_cast_fp16")]; bool input_245_interleave_0 = const()[name = string("input_245_interleave_0"), val = bool(false)]; tensor input_245_cast_fp16 = concat(axis = var_4279, interleave = input_245_interleave_0, values = (x_243, var_4285_cast_fp16))[name = string("input_245_cast_fp16")]; tensor normed_245_axes_0 = const()[name = string("normed_245_axes_0"), val = tensor([-1])]; fp16 var_4277_to_fp16 = const()[name = string("op_4277_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_245_cast_fp16 = layer_norm(axes = normed_245_axes_0, epsilon = var_4277_to_fp16, x = input_245_cast_fp16)[name = string("normed_245_cast_fp16")]; tensor var_4290_split_sizes_0 = const()[name = string("op_4290_split_sizes_0"), val = tensor([512, 512])]; int32 var_4290_axis_0 = const()[name = string("op_4290_axis_0"), val = int32(-1)]; tensor var_4290_cast_fp16_0, tensor var_4290_cast_fp16_1 = split(axis = var_4290_axis_0, split_sizes = var_4290_split_sizes_0, x = normed_245_cast_fp16)[name = string("op_4290_cast_fp16")]; tensor var_4293_cast_fp16 = mul(x = var_4290_cast_fp16_0, y = const_3_to_fp16)[name = string("op_4293_cast_fp16")]; tensor var_4299 = const()[name = string("op_4299"), val = tensor([1, 8, 1, 512])]; tensor q_63 = reshape(shape = var_4299, x = var_4293_cast_fp16)[name = string("q_63")]; tensor var_4301_cast_fp16 = mul(x = q_63, y = cos_f)[name = string("op_4301_cast_fp16")]; tensor var_4302_split_sizes_0 = const()[name = string("op_4302_split_sizes_0"), val = tensor([256, 256])]; int32 var_4302_axis_0 = const()[name = string("op_4302_axis_0"), val = int32(-1)]; tensor var_4302_0, tensor var_4302_1 = split(axis = var_4302_axis_0, split_sizes = var_4302_split_sizes_0, x = q_63)[name = string("op_4302")]; fp16 const_134_promoted = const()[name = string("const_134_promoted"), val = fp16(-0x1p+0)]; tensor var_4304 = mul(x = var_4302_1, y = const_134_promoted)[name = string("op_4304")]; int32 var_4306 = const()[name = string("op_4306"), val = int32(-1)]; bool var_4307_interleave_0 = const()[name = string("op_4307_interleave_0"), val = bool(false)]; tensor var_4307 = concat(axis = var_4306, interleave = var_4307_interleave_0, values = (var_4304, var_4302_0))[name = string("op_4307")]; tensor var_4308_cast_fp16 = mul(x = var_4307, y = sin_f)[name = string("op_4308_cast_fp16")]; tensor q_cast_fp16 = add(x = var_4301_cast_fp16, y = var_4308_cast_fp16)[name = string("q_cast_fp16")]; bool var_4332_transpose_x_0 = const()[name = string("op_4332_transpose_x_0"), val = bool(false)]; bool var_4332_transpose_y_0 = const()[name = string("op_4332_transpose_y_0"), val = bool(false)]; tensor var_4332_cast_fp16 = matmul(transpose_x = var_4332_transpose_x_0, transpose_y = var_4332_transpose_y_0, x = q_cast_fp16, y = transpose_44_cast_fp16)[name = string("op_4332_cast_fp16")]; tensor var_4339_cast_fp16 = add(x = var_4332_cast_fp16, y = causal_mask)[name = string("op_4339_cast_fp16")]; int32 var_4340 = const()[name = string("op_4340"), val = int32(-1)]; tensor var_4342_cast_fp16 = softmax(axis = var_4340, x = var_4339_cast_fp16)[name = string("op_4342_cast_fp16")]; bool var_4358_transpose_x_0 = const()[name = string("op_4358_transpose_x_0"), val = bool(false)]; bool var_4358_transpose_y_0 = const()[name = string("op_4358_transpose_y_0"), val = bool(false)]; tensor var_4358_cast_fp16 = matmul(transpose_x = var_4358_transpose_x_0, transpose_y = var_4358_transpose_y_0, x = var_4342_cast_fp16, y = Ve_1_cast_fp16)[name = string("op_4358_cast_fp16")]; tensor var_4368 = const()[name = string("op_4368"), val = tensor([0, 2, 1, 3])]; tensor var_4375 = const()[name = string("op_4375"), val = tensor([1, 1, -1])]; tensor var_4369 = transpose(perm = var_4368, x = var_4358_cast_fp16)[name = string("transpose_6")]; tensor var_4376 = reshape(shape = var_4375, x = var_4369)[name = string("op_4376")]; tensor var_4380 = const()[name = string("op_4380"), val = tensor([0, 2, 1])]; tensor squeeze_10_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(356974720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360120512))))[name = string("squeeze_10_palettized")]; string var_4396_pad_type_0 = const()[name = string("op_4396_pad_type_0"), val = string("valid")]; int32 var_4396_groups_0 = const()[name = string("op_4396_groups_0"), val = int32(1)]; tensor var_4396_strides_0 = const()[name = string("op_4396_strides_0"), val = tensor([1])]; tensor var_4396_pad_0 = const()[name = string("op_4396_pad_0"), val = tensor([0, 0])]; tensor var_4396_dilations_0 = const()[name = string("op_4396_dilations_0"), val = tensor([1])]; tensor var_4381 = transpose(perm = var_4380, x = var_4376)[name = string("transpose_5")]; tensor var_4396 = conv(dilations = var_4396_dilations_0, groups = var_4396_groups_0, pad = var_4396_pad_0, pad_type = var_4396_pad_type_0, strides = var_4396_strides_0, weight = squeeze_10_palettized, x = var_4381)[name = string("op_4396")]; tensor var_4400 = const()[name = string("op_4400"), val = tensor([0, 2, 1])]; int32 var_4406 = const()[name = string("op_4406"), val = int32(-1)]; fp16 const_135_promoted_to_fp16 = const()[name = string("const_135_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_247 = transpose(perm = var_4400, x = var_4396)[name = string("transpose_4")]; tensor var_4412_cast_fp16 = mul(x = x_247, y = const_135_promoted_to_fp16)[name = string("op_4412_cast_fp16")]; bool input_249_interleave_0 = const()[name = string("input_249_interleave_0"), val = bool(false)]; tensor input_249_cast_fp16 = concat(axis = var_4406, interleave = input_249_interleave_0, values = (x_247, var_4412_cast_fp16))[name = string("input_249_cast_fp16")]; tensor normed_249_axes_0 = const()[name = string("normed_249_axes_0"), val = tensor([-1])]; fp16 var_4404_to_fp16 = const()[name = string("op_4404_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_249_cast_fp16 = layer_norm(axes = normed_249_axes_0, epsilon = var_4404_to_fp16, x = input_249_cast_fp16)[name = string("normed_249_cast_fp16")]; tensor var_4417_split_sizes_0 = const()[name = string("op_4417_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4417_axis_0 = const()[name = string("op_4417_axis_0"), val = int32(-1)]; tensor var_4417_cast_fp16_0, tensor var_4417_cast_fp16_1 = split(axis = var_4417_axis_0, split_sizes = var_4417_split_sizes_0, x = normed_249_cast_fp16)[name = string("op_4417_cast_fp16")]; tensor const_136_to_fp16 = const()[name = string("const_136_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360122112)))]; tensor var_4420_cast_fp16 = mul(x = var_4417_cast_fp16_0, y = const_136_to_fp16)[name = string("op_4420_cast_fp16")]; tensor x_251_cast_fp16 = add(x = x_239_cast_fp16, y = var_4420_cast_fp16)[name = string("x_251_cast_fp16")]; int32 var_4427 = const()[name = string("op_4427"), val = int32(-1)]; fp16 const_137_promoted_to_fp16 = const()[name = string("const_137_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4433_cast_fp16 = mul(x = x_251_cast_fp16, y = const_137_promoted_to_fp16)[name = string("op_4433_cast_fp16")]; bool input_251_interleave_0 = const()[name = string("input_251_interleave_0"), val = bool(false)]; tensor input_251_cast_fp16 = concat(axis = var_4427, interleave = input_251_interleave_0, values = (x_251_cast_fp16, var_4433_cast_fp16))[name = string("input_251_cast_fp16")]; tensor normed_253_axes_0 = const()[name = string("normed_253_axes_0"), val = tensor([-1])]; fp16 var_4425_to_fp16 = const()[name = string("op_4425_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_253_cast_fp16 = layer_norm(axes = normed_253_axes_0, epsilon = var_4425_to_fp16, x = input_251_cast_fp16)[name = string("normed_253_cast_fp16")]; tensor var_4438_split_sizes_0 = const()[name = string("op_4438_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4438_axis_0 = const()[name = string("op_4438_axis_0"), val = int32(-1)]; tensor var_4438_cast_fp16_0, tensor var_4438_cast_fp16_1 = split(axis = var_4438_axis_0, split_sizes = var_4438_split_sizes_0, x = normed_253_cast_fp16)[name = string("op_4438_cast_fp16")]; tensor const_138_to_fp16 = const()[name = string("const_138_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360125248)))]; tensor var_4441_cast_fp16 = mul(x = var_4438_cast_fp16_0, y = const_138_to_fp16)[name = string("op_4441_cast_fp16")]; tensor var_4454 = const()[name = string("op_4454"), val = tensor([0, 2, 1])]; tensor input_253_axes_0 = const()[name = string("input_253_axes_0"), val = tensor([2])]; tensor var_4455 = transpose(perm = var_4454, x = var_4441_cast_fp16)[name = string("transpose_3")]; tensor input_253 = expand_dims(axes = input_253_axes_0, x = var_4455)[name = string("input_253")]; string var_4468_pad_type_0 = const()[name = string("op_4468_pad_type_0"), val = string("valid")]; tensor var_4468_strides_0 = const()[name = string("op_4468_strides_0"), val = tensor([1, 1])]; tensor var_4468_pad_0 = const()[name = string("op_4468_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4468_dilations_0 = const()[name = string("op_4468_dilations_0"), val = tensor([1, 1])]; int32 var_4468_groups_0 = const()[name = string("op_4468_groups_0"), val = int32(1)]; tensor var_4468 = conv(dilations = var_4468_dilations_0, groups = var_4468_groups_0, pad = var_4468_pad_0, pad_type = var_4468_pad_type_0, strides = var_4468_strides_0, weight = layers_10_mlp_gate_proj_weight_palettized, x = input_253)[name = string("op_4468")]; string var_4470_mode_0 = const()[name = string("op_4470_mode_0"), val = string("TANH_APPROXIMATION")]; tensor var_4470 = gelu(mode = var_4470_mode_0, x = var_4468)[name = string("op_4470")]; string var_4481_pad_type_0 = const()[name = string("op_4481_pad_type_0"), val = string("valid")]; tensor var_4481_strides_0 = const()[name = string("op_4481_strides_0"), val = tensor([1, 1])]; tensor var_4481_pad_0 = const()[name = string("op_4481_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4481_dilations_0 = const()[name = string("op_4481_dilations_0"), val = tensor([1, 1])]; int32 var_4481_groups_0 = const()[name = string("op_4481_groups_0"), val = int32(1)]; tensor var_4481 = conv(dilations = var_4481_dilations_0, groups = var_4481_groups_0, pad = var_4481_pad_0, pad_type = var_4481_pad_type_0, strides = var_4481_strides_0, weight = layers_10_mlp_up_proj_weight_palettized, x = input_253)[name = string("op_4481")]; tensor input_255 = mul(x = var_4470, y = var_4481)[name = string("input_255")]; string var_4493_pad_type_0 = const()[name = string("op_4493_pad_type_0"), val = string("valid")]; tensor var_4493_strides_0 = const()[name = string("op_4493_strides_0"), val = tensor([1, 1])]; tensor var_4493_pad_0 = const()[name = string("op_4493_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4493_dilations_0 = const()[name = string("op_4493_dilations_0"), val = tensor([1, 1])]; int32 var_4493_groups_0 = const()[name = string("op_4493_groups_0"), val = int32(1)]; tensor var_4493 = conv(dilations = var_4493_dilations_0, groups = var_4493_groups_0, pad = var_4493_pad_0, pad_type = var_4493_pad_type_0, strides = var_4493_strides_0, weight = layers_10_mlp_down_proj_weight_palettized, x = input_255)[name = string("op_4493")]; tensor var_4495_axes_0 = const()[name = string("op_4495_axes_0"), val = tensor([2])]; tensor var_4495 = squeeze(axes = var_4495_axes_0, x = var_4493)[name = string("op_4495")]; tensor var_4499 = const()[name = string("op_4499"), val = tensor([0, 2, 1])]; int32 var_4505 = const()[name = string("op_4505"), val = int32(-1)]; fp16 const_139_promoted_to_fp16 = const()[name = string("const_139_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor x_255 = transpose(perm = var_4499, x = var_4495)[name = string("transpose_2")]; tensor var_4511_cast_fp16 = mul(x = x_255, y = const_139_promoted_to_fp16)[name = string("op_4511_cast_fp16")]; bool input_257_interleave_0 = const()[name = string("input_257_interleave_0"), val = bool(false)]; tensor input_257_cast_fp16 = concat(axis = var_4505, interleave = input_257_interleave_0, values = (x_255, var_4511_cast_fp16))[name = string("input_257_cast_fp16")]; tensor normed_257_axes_0 = const()[name = string("normed_257_axes_0"), val = tensor([-1])]; fp16 var_4503_to_fp16 = const()[name = string("op_4503_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_257_cast_fp16 = layer_norm(axes = normed_257_axes_0, epsilon = var_4503_to_fp16, x = input_257_cast_fp16)[name = string("normed_257_cast_fp16")]; tensor var_4516_split_sizes_0 = const()[name = string("op_4516_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4516_axis_0 = const()[name = string("op_4516_axis_0"), val = int32(-1)]; tensor var_4516_cast_fp16_0, tensor var_4516_cast_fp16_1 = split(axis = var_4516_axis_0, split_sizes = var_4516_split_sizes_0, x = normed_257_cast_fp16)[name = string("op_4516_cast_fp16")]; tensor const_140_to_fp16 = const()[name = string("const_140_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360128384)))]; tensor var_4519_cast_fp16 = mul(x = var_4516_cast_fp16_0, y = const_140_to_fp16)[name = string("op_4519_cast_fp16")]; tensor hidden_states_153_cast_fp16 = add(x = x_251_cast_fp16, y = var_4519_cast_fp16)[name = string("hidden_states_153_cast_fp16")]; tensor var_4530 = linear(bias = linear_0_bias_0, weight = layers_10_per_layer_input_gate_weight_palettized, x = hidden_states_153_cast_fp16)[name = string("linear_20")]; string gated_mode_0 = const()[name = string("gated_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated = gelu(mode = gated_mode_0, x = var_4530)[name = string("gated")]; tensor var_4547_begin_0 = const()[name = string("op_4547_begin_0"), val = tensor([0, 0, 8704])]; tensor var_4547_end_0 = const()[name = string("op_4547_end_0"), val = tensor([1, 1, 1])]; tensor var_4547_end_mask_0 = const()[name = string("op_4547_end_mask_0"), val = tensor([true, true, true])]; tensor var_4547_cast_fp16 = slice_by_index(begin = var_4547_begin_0, end = var_4547_end_0, end_mask = var_4547_end_mask_0, x = per_layer_combined)[name = string("op_4547_cast_fp16")]; tensor input_261_cast_fp16 = mul(x = gated, y = var_4547_cast_fp16)[name = string("input_261_cast_fp16")]; tensor layers_10_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360131520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360328192))))[name = string("layers_10_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_21_cast_fp16 = linear(bias = linear_1_bias_0_to_fp16, weight = layers_10_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_261_cast_fp16)[name = string("linear_21_cast_fp16")]; int32 var_4556 = const()[name = string("op_4556"), val = int32(-1)]; fp16 const_141_promoted_to_fp16 = const()[name = string("const_141_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4562_cast_fp16 = mul(x = linear_21_cast_fp16, y = const_141_promoted_to_fp16)[name = string("op_4562_cast_fp16")]; bool input_263_interleave_0 = const()[name = string("input_263_interleave_0"), val = bool(false)]; tensor input_263_cast_fp16 = concat(axis = var_4556, interleave = input_263_interleave_0, values = (linear_21_cast_fp16, var_4562_cast_fp16))[name = string("input_263_cast_fp16")]; tensor normed_261_axes_0 = const()[name = string("normed_261_axes_0"), val = tensor([-1])]; fp16 var_4554_to_fp16 = const()[name = string("op_4554_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_261_cast_fp16 = layer_norm(axes = normed_261_axes_0, epsilon = var_4554_to_fp16, x = input_263_cast_fp16)[name = string("normed_261_cast_fp16")]; tensor var_4567_split_sizes_0 = const()[name = string("op_4567_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4567_axis_0 = const()[name = string("op_4567_axis_0"), val = int32(-1)]; tensor var_4567_cast_fp16_0, tensor var_4567_cast_fp16_1 = split(axis = var_4567_axis_0, split_sizes = var_4567_split_sizes_0, x = normed_261_cast_fp16)[name = string("op_4567_cast_fp16")]; tensor const_142_to_fp16 = const()[name = string("const_142_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360329792)))]; tensor var_4570_cast_fp16 = mul(x = var_4567_cast_fp16_0, y = const_142_to_fp16)[name = string("op_4570_cast_fp16")]; tensor hidden_states_157_cast_fp16 = add(x = hidden_states_153_cast_fp16, y = var_4570_cast_fp16)[name = string("hidden_states_157_cast_fp16")]; tensor layers_10_layer_scalar_to_fp16 = const()[name = string("layers_10_layer_scalar_to_fp16"), val = tensor([0x1.56p-3])]; tensor x_263_cast_fp16 = mul(x = hidden_states_157_cast_fp16, y = layers_10_layer_scalar_to_fp16)[name = string("x_263_cast_fp16")]; int32 var_4578 = const()[name = string("op_4578"), val = int32(-1)]; fp16 const_143_promoted_to_fp16 = const()[name = string("const_143_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4584_cast_fp16 = mul(x = x_263_cast_fp16, y = const_143_promoted_to_fp16)[name = string("op_4584_cast_fp16")]; bool input_265_interleave_0 = const()[name = string("input_265_interleave_0"), val = bool(false)]; tensor input_265_cast_fp16 = concat(axis = var_4578, interleave = input_265_interleave_0, values = (x_263_cast_fp16, var_4584_cast_fp16))[name = string("input_265_cast_fp16")]; tensor normed_265_axes_0 = const()[name = string("normed_265_axes_0"), val = tensor([-1])]; fp16 var_4576_to_fp16 = const()[name = string("op_4576_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_265_cast_fp16 = layer_norm(axes = normed_265_axes_0, epsilon = var_4576_to_fp16, x = input_265_cast_fp16)[name = string("normed_265_cast_fp16")]; tensor var_4589_split_sizes_0 = const()[name = string("op_4589_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4589_axis_0 = const()[name = string("op_4589_axis_0"), val = int32(-1)]; tensor var_4589_cast_fp16_0, tensor var_4589_cast_fp16_1 = split(axis = var_4589_axis_0, split_sizes = var_4589_split_sizes_0, x = normed_265_cast_fp16)[name = string("op_4589_cast_fp16")]; tensor const_144_to_fp16 = const()[name = string("const_144_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360332928)))]; tensor var_4592_cast_fp16 = mul(x = var_4589_cast_fp16_0, y = const_144_to_fp16)[name = string("op_4592_cast_fp16")]; tensor var_4602 = const()[name = string("op_4602"), val = tensor([0, 2, 1])]; tensor squeeze_11_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360336064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(561662720))))[name = string("squeeze_11_palettized")]; string var_4618_pad_type_0 = const()[name = string("op_4618_pad_type_0"), val = string("valid")]; int32 var_4618_groups_0 = const()[name = string("op_4618_groups_0"), val = int32(1)]; tensor var_4618_strides_0 = const()[name = string("op_4618_strides_0"), val = tensor([1])]; tensor var_4618_pad_0 = const()[name = string("op_4618_pad_0"), val = tensor([0, 0])]; tensor var_4618_dilations_0 = const()[name = string("op_4618_dilations_0"), val = tensor([1])]; tensor var_4603 = transpose(perm = var_4602, x = var_4592_cast_fp16)[name = string("transpose_1")]; tensor var_4618 = conv(dilations = var_4618_dilations_0, groups = var_4618_groups_0, pad = var_4618_pad_0, pad_type = var_4618_pad_type_0, strides = var_4618_strides_0, weight = squeeze_11_palettized, x = var_4603)[name = string("op_4618")]; tensor var_4622 = const()[name = string("op_4622"), val = tensor([0, 2, 1])]; fp16 _inversed_4625_y_0_to_fp16 = const()[name = string("_inversed_4625_y_0_to_fp16"), val = fp16(0x1.11p-5)]; tensor logits_1 = transpose(perm = var_4622, x = var_4618)[name = string("transpose_0")]; tensor _inversed_4625_cast_fp16 = mul(x = logits_1, y = _inversed_4625_y_0_to_fp16)[name = string("_inversed_4625_cast_fp16")]; tensor var_4626_cast_fp16 = tanh(x = _inversed_4625_cast_fp16)[name = string("op_4626_cast_fp16")]; fp16 var_4627_to_fp16 = const()[name = string("op_4627_to_fp16"), val = fp16(0x1.ep+4)]; tensor logits_3_cast_fp16 = mul(x = var_4626_cast_fp16, y = var_4627_to_fp16)[name = string("logits_3_cast_fp16")]; tensor logits_axes_0 = const()[name = string("logits_axes_0"), val = tensor([0])]; tensor logits_cast_fp16 = squeeze(axes = logits_axes_0, x = logits_3_cast_fp16)[name = string("logits_cast_fp16")]; int32 var_4632 = const()[name = string("op_4632"), val = int32(-1)]; int32 token_id_axis_0 = const()[name = string("token_id_axis_0"), val = int32(-1)]; bool token_id_keep_dims_0 = const()[name = string("token_id_keep_dims_0"), val = bool(false)]; string token_id_output_dtype_0 = const()[name = string("token_id_output_dtype_0"), val = string("int32")]; tensor token_id = reduce_argmax(axis = token_id_axis_0, keep_dims = token_id_keep_dims_0, output_dtype = token_id_output_dtype_0, x = logits_cast_fp16)[name = string("token_id_cast_fp16")]; tensor var_4634_axes_0 = const()[name = string("op_4634_axes_0"), val = tensor([-1])]; tensor var_4634 = expand_dims(axes = var_4634_axes_0, x = token_id)[name = string("op_4634")]; bool var_4635_validate_indices_0 = const()[name = string("op_4635_validate_indices_0"), val = bool(false)]; tensor var_4635_cast_fp16 = gather_along_axis(axis = var_4632, indices = var_4634, validate_indices = var_4635_validate_indices_0, x = logits_cast_fp16)[name = string("op_4635_cast_fp16")]; tensor var_4636_axes_0 = const()[name = string("op_4636_axes_0"), val = tensor([-1])]; tensor token_logit = squeeze(axes = var_4636_axes_0, x = var_4635_cast_fp16)[name = string("op_4636_cast_fp16")]; } -> (token_id, token_logit); }