diff --git "a/qwen25_FFN_PF_lut6_chunk_03of04.mlmodelc/model.mil" "b/qwen25_FFN_PF_lut6_chunk_03of04.mlmodelc/model.mil" new file mode 100644--- /dev/null +++ "b/qwen25_FFN_PF_lut6_chunk_03of04.mlmodelc/model.mil" @@ -0,0 +1,4047 @@ +program(1.3) +[buildInfo = dict({{"coremlc-component-MIL", "3510.2.1"}, {"coremlc-version", "3500.32.1"}})] +{ + func infer(tensor causal_mask, tensor current_pos, tensor hidden_states, state> model_model_kv_cache_0, tensor position_ids) { + tensor model_model_layers_10_self_attn_q_proj_bias = const()[name = string("model_model_layers_10_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64)))]; + tensor model_model_layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1772736))))[name = string("model_model_layers_10_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_10_self_attn_k_proj_bias = const()[name = string("model_model_layers_10_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1821952)))]; + tensor model_model_layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1822528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2117504))))[name = string("model_model_layers_10_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_10_self_attn_v_proj_bias = const()[name = string("model_model_layers_10_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2125760)))]; + tensor model_model_layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2126336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2421312))))[name = string("model_model_layers_10_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_10_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2429568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12751552))))[name = string("model_model_layers_10_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_10_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13038336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23360320))))[name = string("model_model_layers_10_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_10_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23647104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33969088))))[name = string("model_model_layers_10_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_11_self_attn_q_proj_bias = const()[name = string("model_model_layers_11_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34018304)))]; + tensor model_model_layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34021440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35790976))))[name = string("model_model_layers_11_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_11_self_attn_k_proj_bias = const()[name = string("model_model_layers_11_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35840192)))]; + tensor model_model_layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35840768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36135744))))[name = string("model_model_layers_11_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_11_self_attn_v_proj_bias = const()[name = string("model_model_layers_11_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36144000)))]; + tensor model_model_layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36144576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36439552))))[name = string("model_model_layers_11_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_11_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36447808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46769792))))[name = string("model_model_layers_11_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_11_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47056576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57378560))))[name = string("model_model_layers_11_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_11_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57665344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67987328))))[name = string("model_model_layers_11_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_12_self_attn_q_proj_bias = const()[name = string("model_model_layers_12_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68036544)))]; + tensor model_model_layers_12_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68039680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69809216))))[name = string("model_model_layers_12_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_12_self_attn_k_proj_bias = const()[name = string("model_model_layers_12_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69858432)))]; + tensor model_model_layers_12_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69859008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70153984))))[name = string("model_model_layers_12_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_12_self_attn_v_proj_bias = const()[name = string("model_model_layers_12_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70162240)))]; + tensor model_model_layers_12_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70162816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70457792))))[name = string("model_model_layers_12_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_12_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70466048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80788032))))[name = string("model_model_layers_12_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_12_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81074816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91396800))))[name = string("model_model_layers_12_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_12_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91683584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102005568))))[name = string("model_model_layers_12_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_13_self_attn_q_proj_bias = const()[name = string("model_model_layers_13_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102054784)))]; + tensor model_model_layers_13_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102057920))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103827456))))[name = string("model_model_layers_13_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_13_self_attn_k_proj_bias = const()[name = string("model_model_layers_13_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103876672)))]; + tensor model_model_layers_13_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103877248))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104172224))))[name = string("model_model_layers_13_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_13_self_attn_v_proj_bias = const()[name = string("model_model_layers_13_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104180480)))]; + tensor model_model_layers_13_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104181056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104476032))))[name = string("model_model_layers_13_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_13_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104484288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(114806272))))[name = string("model_model_layers_13_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_13_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115093056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125415040))))[name = string("model_model_layers_13_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_13_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125701824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(136023808))))[name = string("model_model_layers_13_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_14_self_attn_q_proj_bias = const()[name = string("model_model_layers_14_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(136073024)))]; + tensor model_model_layers_14_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(136076160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(137845696))))[name = string("model_model_layers_14_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_14_self_attn_k_proj_bias = const()[name = string("model_model_layers_14_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(137894912)))]; + tensor model_model_layers_14_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(137895488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138190464))))[name = string("model_model_layers_14_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_14_self_attn_v_proj_bias = const()[name = string("model_model_layers_14_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138198720)))]; + tensor model_model_layers_14_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138199296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138494272))))[name = string("model_model_layers_14_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_14_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138502528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148824512))))[name = string("model_model_layers_14_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_14_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149111296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159433280))))[name = string("model_model_layers_14_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_14_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159720064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(170042048))))[name = string("model_model_layers_14_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_15_self_attn_q_proj_bias = const()[name = string("model_model_layers_15_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(170091264)))]; + tensor model_model_layers_15_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(170094400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(171863936))))[name = string("model_model_layers_15_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_15_self_attn_k_proj_bias = const()[name = string("model_model_layers_15_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(171913152)))]; + tensor model_model_layers_15_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(171913728))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172208704))))[name = string("model_model_layers_15_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_15_self_attn_v_proj_bias = const()[name = string("model_model_layers_15_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172216960)))]; + tensor model_model_layers_15_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172217536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172512512))))[name = string("model_model_layers_15_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_15_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172520768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(182842752))))[name = string("model_model_layers_15_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_15_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(183129536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193451520))))[name = string("model_model_layers_15_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_15_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193738304))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204060288))))[name = string("model_model_layers_15_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_16_self_attn_q_proj_bias = const()[name = string("model_model_layers_16_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204109504)))]; + tensor model_model_layers_16_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204112640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205882176))))[name = string("model_model_layers_16_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_16_self_attn_k_proj_bias = const()[name = string("model_model_layers_16_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205931392)))]; + tensor model_model_layers_16_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205931968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206226944))))[name = string("model_model_layers_16_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_16_self_attn_v_proj_bias = const()[name = string("model_model_layers_16_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206235200)))]; + tensor model_model_layers_16_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206235776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206530752))))[name = string("model_model_layers_16_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_16_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206539008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216860992))))[name = string("model_model_layers_16_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_16_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(217147776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(227469760))))[name = string("model_model_layers_16_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_16_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(227756544))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(238078528))))[name = string("model_model_layers_16_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_17_self_attn_q_proj_bias = const()[name = string("model_model_layers_17_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(238127744)))]; + tensor model_model_layers_17_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(238130880))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239900416))))[name = string("model_model_layers_17_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_17_self_attn_k_proj_bias = const()[name = string("model_model_layers_17_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239949632)))]; + tensor model_model_layers_17_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239950208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240245184))))[name = string("model_model_layers_17_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_17_self_attn_v_proj_bias = const()[name = string("model_model_layers_17_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240253440)))]; + tensor model_model_layers_17_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240254016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240548992))))[name = string("model_model_layers_17_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_17_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240557248))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(250879232))))[name = string("model_model_layers_17_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_17_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(251166016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261488000))))[name = string("model_model_layers_17_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_17_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261774784))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272096768))))[name = string("model_model_layers_17_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_18_self_attn_q_proj_bias = const()[name = string("model_model_layers_18_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272145984)))]; + tensor model_model_layers_18_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272149120))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273918656))))[name = string("model_model_layers_18_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_18_self_attn_k_proj_bias = const()[name = string("model_model_layers_18_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273967872)))]; + tensor model_model_layers_18_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273968448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274263424))))[name = string("model_model_layers_18_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_18_self_attn_v_proj_bias = const()[name = string("model_model_layers_18_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274271680)))]; + tensor model_model_layers_18_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274272256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274567232))))[name = string("model_model_layers_18_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_18_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274575488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284897472))))[name = string("model_model_layers_18_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_18_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(285184256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295506240))))[name = string("model_model_layers_18_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_18_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295793024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306115008))))[name = string("model_model_layers_18_mlp_down_proj_weight_palettized")]; + int32 var_385_batch_dims_0 = const()[name = string("op_385_batch_dims_0"), val = int32(0)]; + bool var_385_validate_indices_0 = const()[name = string("op_385_validate_indices_0"), val = bool(false)]; + tensor var_377_to_fp16 = const()[name = string("op_377_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306164224)))]; + string current_pos_to_int16_dtype_0 = const()[name = string("current_pos_to_int16_dtype_0"), val = string("int16")]; + string cast_78_dtype_0 = const()[name = string("cast_78_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; + tensor current_pos_to_int16 = cast(dtype = current_pos_to_int16_dtype_0, x = current_pos)[name = string("cast_5")]; + tensor cast_78 = cast(dtype = cast_78_dtype_0, x = current_pos_to_int16)[name = string("cast_4")]; + tensor greater_equal_0 = greater_equal(x = cast_78, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; + int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(4096)]; + tensor add_0 = add(x = cast_78, y = slice_by_index_0)[name = string("add_0")]; + tensor select_0 = select(a = cast_78, b = add_0, cond = greater_equal_0)[name = string("select_0")]; + string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; + string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; + tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; + tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; + tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; + int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(4096)]; + tensor add_0_1 = add(x = cast_0, y = slice_by_index_0_1)[name = string("add_0_1")]; + tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; + int32 op_385_cast_fp16_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_385_cast_fp16_cast_uint16_cast_uint16_axis_0"), val = int32(1)]; + tensor op_385_cast_fp16_cast_uint16_cast_uint16 = gather(axis = op_385_cast_fp16_cast_uint16_cast_uint16_axis_0, batch_dims = var_385_batch_dims_0, indices = select_0_1, validate_indices = var_385_validate_indices_0, x = var_377_to_fp16)[name = string("op_385_cast_fp16_cast_uint16_cast_uint16")]; + tensor var_390 = const()[name = string("op_390"), val = tensor([1, 1, 1, -1])]; + tensor sin_1_cast_fp16 = reshape(shape = var_390, x = op_385_cast_fp16_cast_uint16_cast_uint16)[name = string("sin_1_cast_fp16")]; + int32 var_400_axis_0 = const()[name = string("op_400_axis_0"), val = int32(1)]; + int32 var_400_batch_dims_0 = const()[name = string("op_400_batch_dims_0"), val = int32(0)]; + bool var_400_validate_indices_0 = const()[name = string("op_400_validate_indices_0"), val = bool(false)]; + tensor var_392_to_fp16 = const()[name = string("op_392_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307212864)))]; + string current_pos_to_uint16_dtype_0 = const()[name = string("current_pos_to_uint16_dtype_0"), val = string("uint16")]; + tensor current_pos_to_uint16 = cast(dtype = current_pos_to_uint16_dtype_0, x = current_pos)[name = string("cast_1")]; + tensor var_400_cast_fp16_cast_uint16 = gather(axis = var_400_axis_0, batch_dims = var_400_batch_dims_0, indices = current_pos_to_uint16, validate_indices = var_400_validate_indices_0, x = var_392_to_fp16)[name = string("op_400_cast_fp16_cast_uint16")]; + tensor var_405 = const()[name = string("op_405"), val = tensor([1, 1, 1, -1])]; + tensor cos_1_cast_fp16 = reshape(shape = var_405, x = var_400_cast_fp16_cast_uint16)[name = string("cos_1_cast_fp16")]; + int32 var_426 = const()[name = string("op_426"), val = int32(-1)]; + fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_428_cast_fp16 = mul(x = hidden_states, y = const_0_promoted_to_fp16)[name = string("op_428_cast_fp16")]; + bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; + tensor input_1_cast_fp16 = concat(axis = var_426, interleave = input_1_interleave_0, values = (hidden_states, var_428_cast_fp16))[name = string("input_1_cast_fp16")]; + tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; + fp16 var_423_to_fp16 = const()[name = string("op_423_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_423_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; + tensor normed_3_begin_0 = const()[name = string("normed_3_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_3_end_0 = const()[name = string("normed_3_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_3_end_mask_0 = const()[name = string("normed_3_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_3_cast_fp16 = slice_by_index(begin = normed_3_begin_0, end = normed_3_end_0, end_mask = normed_3_end_mask_0, x = normed_1_cast_fp16)[name = string("normed_3_cast_fp16")]; + tensor const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308261504)))]; + tensor hidden_states_3_cast_fp16 = mul(x = normed_3_cast_fp16, y = const_3_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; + tensor var_445 = const()[name = string("op_445"), val = tensor([0, 2, 1])]; + tensor var_448_axes_0 = const()[name = string("op_448_axes_0"), val = tensor([2])]; + tensor var_446_cast_fp16 = transpose(perm = var_445, x = hidden_states_3_cast_fp16)[name = string("transpose_53")]; + tensor var_448_cast_fp16 = expand_dims(axes = var_448_axes_0, x = var_446_cast_fp16)[name = string("op_448_cast_fp16")]; + string var_464_pad_type_0 = const()[name = string("op_464_pad_type_0"), val = string("valid")]; + tensor var_464_strides_0 = const()[name = string("op_464_strides_0"), val = tensor([1, 1])]; + tensor var_464_pad_0 = const()[name = string("op_464_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_464_dilations_0 = const()[name = string("op_464_dilations_0"), val = tensor([1, 1])]; + int32 var_464_groups_0 = const()[name = string("op_464_groups_0"), val = int32(1)]; + tensor var_464 = conv(bias = model_model_layers_10_self_attn_q_proj_bias, dilations = var_464_dilations_0, groups = var_464_groups_0, pad = var_464_pad_0, pad_type = var_464_pad_type_0, strides = var_464_strides_0, weight = model_model_layers_10_self_attn_q_proj_weight_palettized, x = var_448_cast_fp16)[name = string("op_464")]; + tensor var_469 = const()[name = string("op_469"), val = tensor([1, 12, 1, 128])]; + tensor var_470 = reshape(shape = var_469, x = var_464)[name = string("op_470")]; + string var_486_pad_type_0 = const()[name = string("op_486_pad_type_0"), val = string("valid")]; + tensor var_486_strides_0 = const()[name = string("op_486_strides_0"), val = tensor([1, 1])]; + tensor var_486_pad_0 = const()[name = string("op_486_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_486_dilations_0 = const()[name = string("op_486_dilations_0"), val = tensor([1, 1])]; + int32 var_486_groups_0 = const()[name = string("op_486_groups_0"), val = int32(1)]; + tensor var_486 = conv(bias = model_model_layers_10_self_attn_k_proj_bias, dilations = var_486_dilations_0, groups = var_486_groups_0, pad = var_486_pad_0, pad_type = var_486_pad_type_0, strides = var_486_strides_0, weight = model_model_layers_10_self_attn_k_proj_weight_palettized, x = var_448_cast_fp16)[name = string("op_486")]; + tensor var_491 = const()[name = string("op_491"), val = tensor([1, 2, 1, 128])]; + tensor var_492 = reshape(shape = var_491, x = var_486)[name = string("op_492")]; + string var_508_pad_type_0 = const()[name = string("op_508_pad_type_0"), val = string("valid")]; + tensor var_508_strides_0 = const()[name = string("op_508_strides_0"), val = tensor([1, 1])]; + tensor var_508_pad_0 = const()[name = string("op_508_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_508_dilations_0 = const()[name = string("op_508_dilations_0"), val = tensor([1, 1])]; + int32 var_508_groups_0 = const()[name = string("op_508_groups_0"), val = int32(1)]; + tensor var_508 = conv(bias = model_model_layers_10_self_attn_v_proj_bias, dilations = var_508_dilations_0, groups = var_508_groups_0, pad = var_508_pad_0, pad_type = var_508_pad_type_0, strides = var_508_strides_0, weight = model_model_layers_10_self_attn_v_proj_weight_palettized, x = var_448_cast_fp16)[name = string("op_508")]; + tensor var_513 = const()[name = string("op_513"), val = tensor([1, 2, 1, 128])]; + tensor var_514 = reshape(shape = var_513, x = var_508)[name = string("op_514")]; + tensor var_520 = mul(x = var_470, y = cos_1_cast_fp16)[name = string("op_520")]; + tensor x1_1_begin_0 = const()[name = string("x1_1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_1_end_0 = const()[name = string("x1_1_end_0"), val = tensor([1, 12, 1, 64])]; + tensor x1_1_end_mask_0 = const()[name = string("x1_1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_1 = slice_by_index(begin = x1_1_begin_0, end = x1_1_end_0, end_mask = x1_1_end_mask_0, x = var_470)[name = string("x1_1")]; + tensor x2_1_begin_0 = const()[name = string("x2_1_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_1_end_0 = const()[name = string("x2_1_end_0"), val = tensor([1, 12, 1, 128])]; + tensor x2_1_end_mask_0 = const()[name = string("x2_1_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_1 = slice_by_index(begin = x2_1_begin_0, end = x2_1_end_0, end_mask = x2_1_end_mask_0, x = var_470)[name = string("x2_1")]; + fp16 const_6_promoted = const()[name = string("const_6_promoted"), val = fp16(-0x1p+0)]; + tensor var_541 = mul(x = x2_1, y = const_6_promoted)[name = string("op_541")]; + int32 var_543 = const()[name = string("op_543"), val = int32(-1)]; + bool var_544_interleave_0 = const()[name = string("op_544_interleave_0"), val = bool(false)]; + tensor var_544 = concat(axis = var_543, interleave = var_544_interleave_0, values = (var_541, x1_1))[name = string("op_544")]; + tensor var_545 = mul(x = var_544, y = sin_1_cast_fp16)[name = string("op_545")]; + tensor query_states_1 = add(x = var_520, y = var_545)[name = string("query_states_1")]; + tensor var_548 = mul(x = var_492, y = cos_1_cast_fp16)[name = string("op_548")]; + tensor x1_3_begin_0 = const()[name = string("x1_3_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_3_end_0 = const()[name = string("x1_3_end_0"), val = tensor([1, 2, 1, 64])]; + tensor x1_3_end_mask_0 = const()[name = string("x1_3_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_3 = slice_by_index(begin = x1_3_begin_0, end = x1_3_end_0, end_mask = x1_3_end_mask_0, x = var_492)[name = string("x1_3")]; + tensor x2_3_begin_0 = const()[name = string("x2_3_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_3_end_0 = const()[name = string("x2_3_end_0"), val = tensor([1, 2, 1, 128])]; + tensor x2_3_end_mask_0 = const()[name = string("x2_3_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_3 = slice_by_index(begin = x2_3_begin_0, end = x2_3_end_0, end_mask = x2_3_end_mask_0, x = var_492)[name = string("x2_3")]; + fp16 const_9_promoted = const()[name = string("const_9_promoted"), val = fp16(-0x1p+0)]; + tensor var_569 = mul(x = x2_3, y = const_9_promoted)[name = string("op_569")]; + int32 var_571 = const()[name = string("op_571"), val = int32(-1)]; + bool var_572_interleave_0 = const()[name = string("op_572_interleave_0"), val = bool(false)]; + tensor var_572 = concat(axis = var_571, interleave = var_572_interleave_0, values = (var_569, x1_3))[name = string("op_572")]; + tensor var_573 = mul(x = var_572, y = sin_1_cast_fp16)[name = string("op_573")]; + tensor key_states_1 = add(x = var_548, y = var_573)[name = string("key_states_1")]; + int32 var_577 = const()[name = string("op_577"), val = int32(1)]; + tensor var_578 = add(x = current_pos, y = var_577)[name = string("op_578")]; + tensor read_state_0 = read_state(input = model_model_kv_cache_0)[name = string("read_state_0")]; + tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([10])]; + tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; + tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; + tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([11])]; + int32 concat_2_axis_0 = const()[name = string("concat_2_axis_0"), val = int32(0)]; + bool concat_2_interleave_0 = const()[name = string("concat_2_interleave_0"), val = bool(false)]; + tensor concat_2 = concat(axis = concat_2_axis_0, interleave = concat_2_interleave_0, values = (expand_dims_0, expand_dims_1, current_pos, expand_dims_3))[name = string("concat_2")]; + tensor concat_3_values1_0 = const()[name = string("concat_3_values1_0"), val = tensor([0])]; + tensor concat_3_values3_0 = const()[name = string("concat_3_values3_0"), val = tensor([0])]; + int32 concat_3_axis_0 = const()[name = string("concat_3_axis_0"), val = int32(0)]; + bool concat_3_interleave_0 = const()[name = string("concat_3_interleave_0"), val = bool(false)]; + tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (expand_dims_4, concat_3_values1_0, var_578, concat_3_values3_0))[name = string("concat_3")]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_2, begin_mask = model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0, end = concat_3, end_mask = model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_1_stride_0, update = key_states_1, x = read_state_0)[name = string("model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_0_write_state")]; + tensor coreml_update_state_18 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_0")]; + tensor expand_dims_6 = const()[name = string("expand_dims_6"), val = tensor([38])]; + tensor expand_dims_7 = const()[name = string("expand_dims_7"), val = tensor([0])]; + tensor expand_dims_9 = const()[name = string("expand_dims_9"), val = tensor([0])]; + tensor expand_dims_10 = const()[name = string("expand_dims_10"), val = tensor([39])]; + int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; + bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; + tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_6, expand_dims_7, current_pos, expand_dims_9))[name = string("concat_6")]; + tensor concat_7_values1_0 = const()[name = string("concat_7_values1_0"), val = tensor([0])]; + tensor concat_7_values3_0 = const()[name = string("concat_7_values3_0"), val = tensor([0])]; + int32 concat_7_axis_0 = const()[name = string("concat_7_axis_0"), val = int32(0)]; + bool concat_7_interleave_0 = const()[name = string("concat_7_interleave_0"), val = bool(false)]; + tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (expand_dims_10, concat_7_values1_0, var_578, concat_7_values3_0))[name = string("concat_7")]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_6, begin_mask = model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0, end = concat_7, end_mask = model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_2_stride_0, update = var_514, x = coreml_update_state_18)[name = string("model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_1_write_state")]; + tensor coreml_update_state_19 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_1")]; + tensor var_628_begin_0 = const()[name = string("op_628_begin_0"), val = tensor([10, 0, 0, 0])]; + tensor var_628_end_0 = const()[name = string("op_628_end_0"), val = tensor([11, 2, 2048, 128])]; + tensor var_628_end_mask_0 = const()[name = string("op_628_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_628_cast_fp16 = slice_by_index(begin = var_628_begin_0, end = var_628_end_0, end_mask = var_628_end_mask_0, x = coreml_update_state_19)[name = string("op_628_cast_fp16")]; + tensor K_layer_cache_1_axes_0 = const()[name = string("K_layer_cache_1_axes_0"), val = tensor([0])]; + tensor K_layer_cache_1_cast_fp16 = squeeze(axes = K_layer_cache_1_axes_0, x = var_628_cast_fp16)[name = string("K_layer_cache_1_cast_fp16")]; + tensor var_635_begin_0 = const()[name = string("op_635_begin_0"), val = tensor([38, 0, 0, 0])]; + tensor var_635_end_0 = const()[name = string("op_635_end_0"), val = tensor([39, 2, 2048, 128])]; + tensor var_635_end_mask_0 = const()[name = string("op_635_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_635_cast_fp16 = slice_by_index(begin = var_635_begin_0, end = var_635_end_0, end_mask = var_635_end_mask_0, x = coreml_update_state_19)[name = string("op_635_cast_fp16")]; + tensor V_layer_cache_1_axes_0 = const()[name = string("V_layer_cache_1_axes_0"), val = tensor([0])]; + tensor V_layer_cache_1_cast_fp16 = squeeze(axes = V_layer_cache_1_axes_0, x = var_635_cast_fp16)[name = string("V_layer_cache_1_cast_fp16")]; + tensor x_3_axes_0 = const()[name = string("x_3_axes_0"), val = tensor([1])]; + tensor x_3_cast_fp16 = expand_dims(axes = x_3_axes_0, x = K_layer_cache_1_cast_fp16)[name = string("x_3_cast_fp16")]; + tensor var_672 = const()[name = string("op_672"), val = tensor([1, 6, 1, 1])]; + tensor x_5_cast_fp16 = tile(reps = var_672, x = x_3_cast_fp16)[name = string("x_5_cast_fp16")]; + tensor var_684 = const()[name = string("op_684"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_3_cast_fp16 = reshape(shape = var_684, x = x_5_cast_fp16)[name = string("key_states_3_cast_fp16")]; + tensor x_9_axes_0 = const()[name = string("x_9_axes_0"), val = tensor([1])]; + tensor x_9_cast_fp16 = expand_dims(axes = x_9_axes_0, x = V_layer_cache_1_cast_fp16)[name = string("x_9_cast_fp16")]; + tensor var_692 = const()[name = string("op_692"), val = tensor([1, 6, 1, 1])]; + tensor x_11_cast_fp16 = tile(reps = var_692, x = x_9_cast_fp16)[name = string("x_11_cast_fp16")]; + tensor var_704 = const()[name = string("op_704"), val = tensor([1, -1, 2048, 128])]; + tensor value_states_3_cast_fp16 = reshape(shape = var_704, x = x_11_cast_fp16)[name = string("value_states_3_cast_fp16")]; + bool var_727_transpose_x_1 = const()[name = string("op_727_transpose_x_1"), val = bool(false)]; + bool var_727_transpose_y_1 = const()[name = string("op_727_transpose_y_1"), val = bool(true)]; + tensor var_727_cast_fp16 = matmul(transpose_x = var_727_transpose_x_1, transpose_y = var_727_transpose_y_1, x = query_states_1, y = key_states_3_cast_fp16)[name = string("op_727_cast_fp16")]; + fp16 var_728_to_fp16 = const()[name = string("op_728_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_1_cast_fp16 = mul(x = var_727_cast_fp16, y = var_728_to_fp16)[name = string("attn_logits_1_cast_fp16")]; + tensor attn_logits_3_cast_fp16 = add(x = attn_logits_1_cast_fp16, y = causal_mask)[name = string("attn_logits_3_cast_fp16")]; + int32 var_755 = const()[name = string("op_755"), val = int32(-1)]; + tensor var_757_cast_fp16 = softmax(axis = var_755, x = attn_logits_3_cast_fp16)[name = string("op_757_cast_fp16")]; + bool attn_output_1_transpose_x_0 = const()[name = string("attn_output_1_transpose_x_0"), val = bool(false)]; + bool attn_output_1_transpose_y_0 = const()[name = string("attn_output_1_transpose_y_0"), val = bool(false)]; + tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = var_757_cast_fp16, y = value_states_3_cast_fp16)[name = string("attn_output_1_cast_fp16")]; + tensor var_781_perm_0 = const()[name = string("op_781_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_785 = const()[name = string("op_785"), val = tensor([1, 1, 1536])]; + tensor var_781 = transpose(perm = var_781_perm_0, x = attn_output_1_cast_fp16)[name = string("transpose_52")]; + tensor attn_output_7 = reshape(shape = var_785, x = var_781)[name = string("attn_output_7")]; + tensor var_790 = const()[name = string("op_790"), val = tensor([0, 2, 1])]; + tensor squeeze_0_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308264640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310034176))))[name = string("squeeze_0_palettized")]; + string var_806_pad_type_0 = const()[name = string("op_806_pad_type_0"), val = string("valid")]; + int32 var_806_groups_0 = const()[name = string("op_806_groups_0"), val = int32(1)]; + tensor var_806_strides_0 = const()[name = string("op_806_strides_0"), val = tensor([1])]; + tensor var_806_pad_0 = const()[name = string("op_806_pad_0"), val = tensor([0, 0])]; + tensor var_806_dilations_0 = const()[name = string("op_806_dilations_0"), val = tensor([1])]; + tensor var_791 = transpose(perm = var_790, x = attn_output_7)[name = string("transpose_51")]; + tensor var_806 = conv(dilations = var_806_dilations_0, groups = var_806_groups_0, pad = var_806_pad_0, pad_type = var_806_pad_type_0, strides = var_806_strides_0, weight = squeeze_0_palettized, x = var_791)[name = string("op_806")]; + tensor var_810 = const()[name = string("op_810"), val = tensor([0, 2, 1])]; + tensor attn_output_11 = transpose(perm = var_810, x = var_806)[name = string("transpose_50")]; + tensor hidden_states_5_cast_fp16 = add(x = hidden_states, y = attn_output_11)[name = string("hidden_states_5_cast_fp16")]; + int32 var_823 = const()[name = string("op_823"), val = int32(-1)]; + fp16 const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_825_cast_fp16 = mul(x = hidden_states_5_cast_fp16, y = const_18_promoted_to_fp16)[name = string("op_825_cast_fp16")]; + bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; + tensor input_7_cast_fp16 = concat(axis = var_823, interleave = input_7_interleave_0, values = (hidden_states_5_cast_fp16, var_825_cast_fp16))[name = string("input_7_cast_fp16")]; + tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; + fp16 var_820_to_fp16 = const()[name = string("op_820_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_820_to_fp16, x = input_7_cast_fp16)[name = string("normed_5_cast_fp16")]; + tensor normed_7_begin_0 = const()[name = string("normed_7_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_7_end_0 = const()[name = string("normed_7_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_7_end_mask_0 = const()[name = string("normed_7_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_7_cast_fp16 = slice_by_index(begin = normed_7_begin_0, end = normed_7_end_0, end_mask = normed_7_end_mask_0, x = normed_5_cast_fp16)[name = string("normed_7_cast_fp16")]; + tensor const_21_promoted_to_fp16 = const()[name = string("const_21_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310083392)))]; + tensor x_13_cast_fp16 = mul(x = normed_7_cast_fp16, y = const_21_promoted_to_fp16)[name = string("x_13_cast_fp16")]; + tensor var_850 = const()[name = string("op_850"), val = tensor([0, 2, 1])]; + tensor input_9_axes_0 = const()[name = string("input_9_axes_0"), val = tensor([2])]; + tensor var_851 = transpose(perm = var_850, x = x_13_cast_fp16)[name = string("transpose_49")]; + tensor input_9 = expand_dims(axes = input_9_axes_0, x = var_851)[name = string("input_9")]; + string input_11_pad_type_0 = const()[name = string("input_11_pad_type_0"), val = string("valid")]; + tensor input_11_strides_0 = const()[name = string("input_11_strides_0"), val = tensor([1, 1])]; + tensor input_11_pad_0 = const()[name = string("input_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_11_dilations_0 = const()[name = string("input_11_dilations_0"), val = tensor([1, 1])]; + int32 input_11_groups_0 = const()[name = string("input_11_groups_0"), val = int32(1)]; + tensor input_11 = conv(dilations = input_11_dilations_0, groups = input_11_groups_0, pad = input_11_pad_0, pad_type = input_11_pad_type_0, strides = input_11_strides_0, weight = model_model_layers_10_mlp_gate_proj_weight_palettized, x = input_9)[name = string("input_11")]; + string b_1_pad_type_0 = const()[name = string("b_1_pad_type_0"), val = string("valid")]; + tensor b_1_strides_0 = const()[name = string("b_1_strides_0"), val = tensor([1, 1])]; + tensor b_1_pad_0 = const()[name = string("b_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_1_dilations_0 = const()[name = string("b_1_dilations_0"), val = tensor([1, 1])]; + int32 b_1_groups_0 = const()[name = string("b_1_groups_0"), val = int32(1)]; + tensor b_1 = conv(dilations = b_1_dilations_0, groups = b_1_groups_0, pad = b_1_pad_0, pad_type = b_1_pad_type_0, strides = b_1_strides_0, weight = model_model_layers_10_mlp_up_proj_weight_palettized, x = input_9)[name = string("b_1")]; + tensor c_1 = silu(x = input_11)[name = string("c_1")]; + tensor input_13 = mul(x = c_1, y = b_1)[name = string("input_13")]; + string e_1_pad_type_0 = const()[name = string("e_1_pad_type_0"), val = string("valid")]; + tensor e_1_strides_0 = const()[name = string("e_1_strides_0"), val = tensor([1, 1])]; + tensor e_1_pad_0 = const()[name = string("e_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_1_dilations_0 = const()[name = string("e_1_dilations_0"), val = tensor([1, 1])]; + int32 e_1_groups_0 = const()[name = string("e_1_groups_0"), val = int32(1)]; + tensor e_1 = conv(dilations = e_1_dilations_0, groups = e_1_groups_0, pad = e_1_pad_0, pad_type = e_1_pad_type_0, strides = e_1_strides_0, weight = model_model_layers_10_mlp_down_proj_weight_palettized, x = input_13)[name = string("e_1")]; + tensor var_873_axes_0 = const()[name = string("op_873_axes_0"), val = tensor([2])]; + tensor var_873 = squeeze(axes = var_873_axes_0, x = e_1)[name = string("op_873")]; + tensor var_874 = const()[name = string("op_874"), val = tensor([0, 2, 1])]; + tensor var_875 = transpose(perm = var_874, x = var_873)[name = string("transpose_48")]; + tensor hidden_states_7_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = var_875)[name = string("hidden_states_7_cast_fp16")]; + int32 var_887 = const()[name = string("op_887"), val = int32(-1)]; + fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_889_cast_fp16 = mul(x = hidden_states_7_cast_fp16, y = const_22_promoted_to_fp16)[name = string("op_889_cast_fp16")]; + bool input_15_interleave_0 = const()[name = string("input_15_interleave_0"), val = bool(false)]; + tensor input_15_cast_fp16 = concat(axis = var_887, interleave = input_15_interleave_0, values = (hidden_states_7_cast_fp16, var_889_cast_fp16))[name = string("input_15_cast_fp16")]; + tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; + fp16 var_884_to_fp16 = const()[name = string("op_884_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_884_to_fp16, x = input_15_cast_fp16)[name = string("normed_9_cast_fp16")]; + tensor normed_11_begin_0 = const()[name = string("normed_11_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_11_end_0 = const()[name = string("normed_11_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_11_end_mask_0 = const()[name = string("normed_11_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_11_cast_fp16 = slice_by_index(begin = normed_11_begin_0, end = normed_11_end_0, end_mask = normed_11_end_mask_0, x = normed_9_cast_fp16)[name = string("normed_11_cast_fp16")]; + tensor const_25_promoted_to_fp16 = const()[name = string("const_25_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310086528)))]; + tensor hidden_states_9_cast_fp16 = mul(x = normed_11_cast_fp16, y = const_25_promoted_to_fp16)[name = string("hidden_states_9_cast_fp16")]; + tensor var_906 = const()[name = string("op_906"), val = tensor([0, 2, 1])]; + tensor var_909_axes_0 = const()[name = string("op_909_axes_0"), val = tensor([2])]; + tensor var_907_cast_fp16 = transpose(perm = var_906, x = hidden_states_9_cast_fp16)[name = string("transpose_47")]; + tensor var_909_cast_fp16 = expand_dims(axes = var_909_axes_0, x = var_907_cast_fp16)[name = string("op_909_cast_fp16")]; + string var_925_pad_type_0 = const()[name = string("op_925_pad_type_0"), val = string("valid")]; + tensor var_925_strides_0 = const()[name = string("op_925_strides_0"), val = tensor([1, 1])]; + tensor var_925_pad_0 = const()[name = string("op_925_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_925_dilations_0 = const()[name = string("op_925_dilations_0"), val = tensor([1, 1])]; + int32 var_925_groups_0 = const()[name = string("op_925_groups_0"), val = int32(1)]; + tensor var_925 = conv(bias = model_model_layers_11_self_attn_q_proj_bias, dilations = var_925_dilations_0, groups = var_925_groups_0, pad = var_925_pad_0, pad_type = var_925_pad_type_0, strides = var_925_strides_0, weight = model_model_layers_11_self_attn_q_proj_weight_palettized, x = var_909_cast_fp16)[name = string("op_925")]; + tensor var_930 = const()[name = string("op_930"), val = tensor([1, 12, 1, 128])]; + tensor var_931 = reshape(shape = var_930, x = var_925)[name = string("op_931")]; + string var_947_pad_type_0 = const()[name = string("op_947_pad_type_0"), val = string("valid")]; + tensor var_947_strides_0 = const()[name = string("op_947_strides_0"), val = tensor([1, 1])]; + tensor var_947_pad_0 = const()[name = string("op_947_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_947_dilations_0 = const()[name = string("op_947_dilations_0"), val = tensor([1, 1])]; + int32 var_947_groups_0 = const()[name = string("op_947_groups_0"), val = int32(1)]; + tensor var_947 = conv(bias = model_model_layers_11_self_attn_k_proj_bias, dilations = var_947_dilations_0, groups = var_947_groups_0, pad = var_947_pad_0, pad_type = var_947_pad_type_0, strides = var_947_strides_0, weight = model_model_layers_11_self_attn_k_proj_weight_palettized, x = var_909_cast_fp16)[name = string("op_947")]; + tensor var_952 = const()[name = string("op_952"), val = tensor([1, 2, 1, 128])]; + tensor var_953 = reshape(shape = var_952, x = var_947)[name = string("op_953")]; + string var_969_pad_type_0 = const()[name = string("op_969_pad_type_0"), val = string("valid")]; + tensor var_969_strides_0 = const()[name = string("op_969_strides_0"), val = tensor([1, 1])]; + tensor var_969_pad_0 = const()[name = string("op_969_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_969_dilations_0 = const()[name = string("op_969_dilations_0"), val = tensor([1, 1])]; + int32 var_969_groups_0 = const()[name = string("op_969_groups_0"), val = int32(1)]; + tensor var_969 = conv(bias = model_model_layers_11_self_attn_v_proj_bias, dilations = var_969_dilations_0, groups = var_969_groups_0, pad = var_969_pad_0, pad_type = var_969_pad_type_0, strides = var_969_strides_0, weight = model_model_layers_11_self_attn_v_proj_weight_palettized, x = var_909_cast_fp16)[name = string("op_969")]; + tensor var_974 = const()[name = string("op_974"), val = tensor([1, 2, 1, 128])]; + tensor var_975 = reshape(shape = var_974, x = var_969)[name = string("op_975")]; + tensor var_981 = mul(x = var_931, y = cos_1_cast_fp16)[name = string("op_981")]; + tensor x1_5_begin_0 = const()[name = string("x1_5_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_5_end_0 = const()[name = string("x1_5_end_0"), val = tensor([1, 12, 1, 64])]; + tensor x1_5_end_mask_0 = const()[name = string("x1_5_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_5 = slice_by_index(begin = x1_5_begin_0, end = x1_5_end_0, end_mask = x1_5_end_mask_0, x = var_931)[name = string("x1_5")]; + tensor x2_5_begin_0 = const()[name = string("x2_5_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_5_end_0 = const()[name = string("x2_5_end_0"), val = tensor([1, 12, 1, 128])]; + tensor x2_5_end_mask_0 = const()[name = string("x2_5_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_5 = slice_by_index(begin = x2_5_begin_0, end = x2_5_end_0, end_mask = x2_5_end_mask_0, x = var_931)[name = string("x2_5")]; + fp16 const_28_promoted = const()[name = string("const_28_promoted"), val = fp16(-0x1p+0)]; + tensor var_1002 = mul(x = x2_5, y = const_28_promoted)[name = string("op_1002")]; + int32 var_1004 = const()[name = string("op_1004"), val = int32(-1)]; + bool var_1005_interleave_0 = const()[name = string("op_1005_interleave_0"), val = bool(false)]; + tensor var_1005 = concat(axis = var_1004, interleave = var_1005_interleave_0, values = (var_1002, x1_5))[name = string("op_1005")]; + tensor var_1006 = mul(x = var_1005, y = sin_1_cast_fp16)[name = string("op_1006")]; + tensor query_states_3 = add(x = var_981, y = var_1006)[name = string("query_states_3")]; + tensor var_1009 = mul(x = var_953, y = cos_1_cast_fp16)[name = string("op_1009")]; + tensor x1_7_begin_0 = const()[name = string("x1_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_7_end_0 = const()[name = string("x1_7_end_0"), val = tensor([1, 2, 1, 64])]; + tensor x1_7_end_mask_0 = const()[name = string("x1_7_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_7 = slice_by_index(begin = x1_7_begin_0, end = x1_7_end_0, end_mask = x1_7_end_mask_0, x = var_953)[name = string("x1_7")]; + tensor x2_7_begin_0 = const()[name = string("x2_7_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_7_end_0 = const()[name = string("x2_7_end_0"), val = tensor([1, 2, 1, 128])]; + tensor x2_7_end_mask_0 = const()[name = string("x2_7_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_7 = slice_by_index(begin = x2_7_begin_0, end = x2_7_end_0, end_mask = x2_7_end_mask_0, x = var_953)[name = string("x2_7")]; + fp16 const_31_promoted = const()[name = string("const_31_promoted"), val = fp16(-0x1p+0)]; + tensor var_1030 = mul(x = x2_7, y = const_31_promoted)[name = string("op_1030")]; + int32 var_1032 = const()[name = string("op_1032"), val = int32(-1)]; + bool var_1033_interleave_0 = const()[name = string("op_1033_interleave_0"), val = bool(false)]; + tensor var_1033 = concat(axis = var_1032, interleave = var_1033_interleave_0, values = (var_1030, x1_7))[name = string("op_1033")]; + tensor var_1034 = mul(x = var_1033, y = sin_1_cast_fp16)[name = string("op_1034")]; + tensor key_states_5 = add(x = var_1009, y = var_1034)[name = string("key_states_5")]; + tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([11])]; + tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; + tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; + tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([12])]; + int32 concat_10_axis_0 = const()[name = string("concat_10_axis_0"), val = int32(0)]; + bool concat_10_interleave_0 = const()[name = string("concat_10_interleave_0"), val = bool(false)]; + tensor concat_10 = concat(axis = concat_10_axis_0, interleave = concat_10_interleave_0, values = (expand_dims_12, expand_dims_13, current_pos, expand_dims_15))[name = string("concat_10")]; + tensor concat_11_values1_0 = const()[name = string("concat_11_values1_0"), val = tensor([0])]; + tensor concat_11_values3_0 = const()[name = string("concat_11_values3_0"), val = tensor([0])]; + int32 concat_11_axis_0 = const()[name = string("concat_11_axis_0"), val = int32(0)]; + bool concat_11_interleave_0 = const()[name = string("concat_11_interleave_0"), val = bool(false)]; + tensor concat_11 = concat(axis = concat_11_axis_0, interleave = concat_11_interleave_0, values = (expand_dims_16, concat_11_values1_0, var_578, concat_11_values3_0))[name = string("concat_11")]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_10, begin_mask = model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0, end = concat_11, end_mask = model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_3_stride_0, update = key_states_5, x = coreml_update_state_19)[name = string("model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_2_write_state")]; + tensor coreml_update_state_20 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_2")]; + tensor expand_dims_18 = const()[name = string("expand_dims_18"), val = tensor([39])]; + tensor expand_dims_19 = const()[name = string("expand_dims_19"), val = tensor([0])]; + tensor expand_dims_21 = const()[name = string("expand_dims_21"), val = tensor([0])]; + tensor expand_dims_22 = const()[name = string("expand_dims_22"), val = tensor([40])]; + int32 concat_14_axis_0 = const()[name = string("concat_14_axis_0"), val = int32(0)]; + bool concat_14_interleave_0 = const()[name = string("concat_14_interleave_0"), val = bool(false)]; + tensor concat_14 = concat(axis = concat_14_axis_0, interleave = concat_14_interleave_0, values = (expand_dims_18, expand_dims_19, current_pos, expand_dims_21))[name = string("concat_14")]; + tensor concat_15_values1_0 = const()[name = string("concat_15_values1_0"), val = tensor([0])]; + tensor concat_15_values3_0 = const()[name = string("concat_15_values3_0"), val = tensor([0])]; + int32 concat_15_axis_0 = const()[name = string("concat_15_axis_0"), val = int32(0)]; + bool concat_15_interleave_0 = const()[name = string("concat_15_interleave_0"), val = bool(false)]; + tensor concat_15 = concat(axis = concat_15_axis_0, interleave = concat_15_interleave_0, values = (expand_dims_22, concat_15_values1_0, var_578, concat_15_values3_0))[name = string("concat_15")]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_14, begin_mask = model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0, end = concat_15, end_mask = model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_4_stride_0, update = var_975, x = coreml_update_state_20)[name = string("model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_3_write_state")]; + tensor coreml_update_state_21 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_3")]; + tensor var_1089_begin_0 = const()[name = string("op_1089_begin_0"), val = tensor([11, 0, 0, 0])]; + tensor var_1089_end_0 = const()[name = string("op_1089_end_0"), val = tensor([12, 2, 2048, 128])]; + tensor var_1089_end_mask_0 = const()[name = string("op_1089_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1089_cast_fp16 = slice_by_index(begin = var_1089_begin_0, end = var_1089_end_0, end_mask = var_1089_end_mask_0, x = coreml_update_state_21)[name = string("op_1089_cast_fp16")]; + tensor K_layer_cache_3_axes_0 = const()[name = string("K_layer_cache_3_axes_0"), val = tensor([0])]; + tensor K_layer_cache_3_cast_fp16 = squeeze(axes = K_layer_cache_3_axes_0, x = var_1089_cast_fp16)[name = string("K_layer_cache_3_cast_fp16")]; + tensor var_1096_begin_0 = const()[name = string("op_1096_begin_0"), val = tensor([39, 0, 0, 0])]; + tensor var_1096_end_0 = const()[name = string("op_1096_end_0"), val = tensor([40, 2, 2048, 128])]; + tensor var_1096_end_mask_0 = const()[name = string("op_1096_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1096_cast_fp16 = slice_by_index(begin = var_1096_begin_0, end = var_1096_end_0, end_mask = var_1096_end_mask_0, x = coreml_update_state_21)[name = string("op_1096_cast_fp16")]; + tensor V_layer_cache_3_axes_0 = const()[name = string("V_layer_cache_3_axes_0"), val = tensor([0])]; + tensor V_layer_cache_3_cast_fp16 = squeeze(axes = V_layer_cache_3_axes_0, x = var_1096_cast_fp16)[name = string("V_layer_cache_3_cast_fp16")]; + tensor x_19_axes_0 = const()[name = string("x_19_axes_0"), val = tensor([1])]; + tensor x_19_cast_fp16 = expand_dims(axes = x_19_axes_0, x = K_layer_cache_3_cast_fp16)[name = string("x_19_cast_fp16")]; + tensor var_1133 = const()[name = string("op_1133"), val = tensor([1, 6, 1, 1])]; + tensor x_21_cast_fp16 = tile(reps = var_1133, x = x_19_cast_fp16)[name = string("x_21_cast_fp16")]; + tensor var_1145 = const()[name = string("op_1145"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_7_cast_fp16 = reshape(shape = var_1145, x = x_21_cast_fp16)[name = string("key_states_7_cast_fp16")]; + tensor x_25_axes_0 = const()[name = string("x_25_axes_0"), val = tensor([1])]; + tensor x_25_cast_fp16 = expand_dims(axes = x_25_axes_0, x = V_layer_cache_3_cast_fp16)[name = string("x_25_cast_fp16")]; + tensor var_1153 = const()[name = string("op_1153"), val = tensor([1, 6, 1, 1])]; + tensor x_27_cast_fp16 = tile(reps = var_1153, x = x_25_cast_fp16)[name = string("x_27_cast_fp16")]; + tensor var_1165 = const()[name = string("op_1165"), val = tensor([1, -1, 2048, 128])]; + tensor value_states_7_cast_fp16 = reshape(shape = var_1165, x = x_27_cast_fp16)[name = string("value_states_7_cast_fp16")]; + bool var_1188_transpose_x_1 = const()[name = string("op_1188_transpose_x_1"), val = bool(false)]; + bool var_1188_transpose_y_1 = const()[name = string("op_1188_transpose_y_1"), val = bool(true)]; + tensor var_1188_cast_fp16 = matmul(transpose_x = var_1188_transpose_x_1, transpose_y = var_1188_transpose_y_1, x = query_states_3, y = key_states_7_cast_fp16)[name = string("op_1188_cast_fp16")]; + fp16 var_1189_to_fp16 = const()[name = string("op_1189_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_5_cast_fp16 = mul(x = var_1188_cast_fp16, y = var_1189_to_fp16)[name = string("attn_logits_5_cast_fp16")]; + tensor attn_logits_7_cast_fp16 = add(x = attn_logits_5_cast_fp16, y = causal_mask)[name = string("attn_logits_7_cast_fp16")]; + int32 var_1216 = const()[name = string("op_1216"), val = int32(-1)]; + tensor var_1218_cast_fp16 = softmax(axis = var_1216, x = attn_logits_7_cast_fp16)[name = string("op_1218_cast_fp16")]; + bool attn_output_13_transpose_x_0 = const()[name = string("attn_output_13_transpose_x_0"), val = bool(false)]; + bool attn_output_13_transpose_y_0 = const()[name = string("attn_output_13_transpose_y_0"), val = bool(false)]; + tensor attn_output_13_cast_fp16 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = var_1218_cast_fp16, y = value_states_7_cast_fp16)[name = string("attn_output_13_cast_fp16")]; + tensor var_1242_perm_0 = const()[name = string("op_1242_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1246 = const()[name = string("op_1246"), val = tensor([1, 1, 1536])]; + tensor var_1242 = transpose(perm = var_1242_perm_0, x = attn_output_13_cast_fp16)[name = string("transpose_46")]; + tensor attn_output_19 = reshape(shape = var_1246, x = var_1242)[name = string("attn_output_19")]; + tensor var_1251 = const()[name = string("op_1251"), val = tensor([0, 2, 1])]; + tensor squeeze_1_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310089664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311859200))))[name = string("squeeze_1_palettized")]; + string var_1267_pad_type_0 = const()[name = string("op_1267_pad_type_0"), val = string("valid")]; + int32 var_1267_groups_0 = const()[name = string("op_1267_groups_0"), val = int32(1)]; + tensor var_1267_strides_0 = const()[name = string("op_1267_strides_0"), val = tensor([1])]; + tensor var_1267_pad_0 = const()[name = string("op_1267_pad_0"), val = tensor([0, 0])]; + tensor var_1267_dilations_0 = const()[name = string("op_1267_dilations_0"), val = tensor([1])]; + tensor var_1252 = transpose(perm = var_1251, x = attn_output_19)[name = string("transpose_45")]; + tensor var_1267 = conv(dilations = var_1267_dilations_0, groups = var_1267_groups_0, pad = var_1267_pad_0, pad_type = var_1267_pad_type_0, strides = var_1267_strides_0, weight = squeeze_1_palettized, x = var_1252)[name = string("op_1267")]; + tensor var_1271 = const()[name = string("op_1271"), val = tensor([0, 2, 1])]; + tensor attn_output_23 = transpose(perm = var_1271, x = var_1267)[name = string("transpose_44")]; + tensor hidden_states_11_cast_fp16 = add(x = hidden_states_7_cast_fp16, y = attn_output_23)[name = string("hidden_states_11_cast_fp16")]; + int32 var_1284 = const()[name = string("op_1284"), val = int32(-1)]; + fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1286_cast_fp16 = mul(x = hidden_states_11_cast_fp16, y = const_40_promoted_to_fp16)[name = string("op_1286_cast_fp16")]; + bool input_21_interleave_0 = const()[name = string("input_21_interleave_0"), val = bool(false)]; + tensor input_21_cast_fp16 = concat(axis = var_1284, interleave = input_21_interleave_0, values = (hidden_states_11_cast_fp16, var_1286_cast_fp16))[name = string("input_21_cast_fp16")]; + tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; + fp16 var_1281_to_fp16 = const()[name = string("op_1281_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_1281_to_fp16, x = input_21_cast_fp16)[name = string("normed_13_cast_fp16")]; + tensor normed_15_begin_0 = const()[name = string("normed_15_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_15_end_0 = const()[name = string("normed_15_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_15_end_mask_0 = const()[name = string("normed_15_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_15_cast_fp16 = slice_by_index(begin = normed_15_begin_0, end = normed_15_end_0, end_mask = normed_15_end_mask_0, x = normed_13_cast_fp16)[name = string("normed_15_cast_fp16")]; + tensor const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311908416)))]; + tensor x_29_cast_fp16 = mul(x = normed_15_cast_fp16, y = const_43_promoted_to_fp16)[name = string("x_29_cast_fp16")]; + tensor var_1311 = const()[name = string("op_1311"), val = tensor([0, 2, 1])]; + tensor input_23_axes_0 = const()[name = string("input_23_axes_0"), val = tensor([2])]; + tensor var_1312 = transpose(perm = var_1311, x = x_29_cast_fp16)[name = string("transpose_43")]; + tensor input_23 = expand_dims(axes = input_23_axes_0, x = var_1312)[name = string("input_23")]; + string input_25_pad_type_0 = const()[name = string("input_25_pad_type_0"), val = string("valid")]; + tensor input_25_strides_0 = const()[name = string("input_25_strides_0"), val = tensor([1, 1])]; + tensor input_25_pad_0 = const()[name = string("input_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_25_dilations_0 = const()[name = string("input_25_dilations_0"), val = tensor([1, 1])]; + int32 input_25_groups_0 = const()[name = string("input_25_groups_0"), val = int32(1)]; + tensor input_25 = conv(dilations = input_25_dilations_0, groups = input_25_groups_0, pad = input_25_pad_0, pad_type = input_25_pad_type_0, strides = input_25_strides_0, weight = model_model_layers_11_mlp_gate_proj_weight_palettized, x = input_23)[name = string("input_25")]; + string b_3_pad_type_0 = const()[name = string("b_3_pad_type_0"), val = string("valid")]; + tensor b_3_strides_0 = const()[name = string("b_3_strides_0"), val = tensor([1, 1])]; + tensor b_3_pad_0 = const()[name = string("b_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_3_dilations_0 = const()[name = string("b_3_dilations_0"), val = tensor([1, 1])]; + int32 b_3_groups_0 = const()[name = string("b_3_groups_0"), val = int32(1)]; + tensor b_3 = conv(dilations = b_3_dilations_0, groups = b_3_groups_0, pad = b_3_pad_0, pad_type = b_3_pad_type_0, strides = b_3_strides_0, weight = model_model_layers_11_mlp_up_proj_weight_palettized, x = input_23)[name = string("b_3")]; + tensor c_3 = silu(x = input_25)[name = string("c_3")]; + tensor input_27 = mul(x = c_3, y = b_3)[name = string("input_27")]; + string e_3_pad_type_0 = const()[name = string("e_3_pad_type_0"), val = string("valid")]; + tensor e_3_strides_0 = const()[name = string("e_3_strides_0"), val = tensor([1, 1])]; + tensor e_3_pad_0 = const()[name = string("e_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_3_dilations_0 = const()[name = string("e_3_dilations_0"), val = tensor([1, 1])]; + int32 e_3_groups_0 = const()[name = string("e_3_groups_0"), val = int32(1)]; + tensor e_3 = conv(dilations = e_3_dilations_0, groups = e_3_groups_0, pad = e_3_pad_0, pad_type = e_3_pad_type_0, strides = e_3_strides_0, weight = model_model_layers_11_mlp_down_proj_weight_palettized, x = input_27)[name = string("e_3")]; + tensor var_1334_axes_0 = const()[name = string("op_1334_axes_0"), val = tensor([2])]; + tensor var_1334 = squeeze(axes = var_1334_axes_0, x = e_3)[name = string("op_1334")]; + tensor var_1335 = const()[name = string("op_1335"), val = tensor([0, 2, 1])]; + tensor var_1336 = transpose(perm = var_1335, x = var_1334)[name = string("transpose_42")]; + tensor hidden_states_13_cast_fp16 = add(x = hidden_states_11_cast_fp16, y = var_1336)[name = string("hidden_states_13_cast_fp16")]; + int32 var_1348 = const()[name = string("op_1348"), val = int32(-1)]; + fp16 const_44_promoted_to_fp16 = const()[name = string("const_44_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1350_cast_fp16 = mul(x = hidden_states_13_cast_fp16, y = const_44_promoted_to_fp16)[name = string("op_1350_cast_fp16")]; + bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; + tensor input_29_cast_fp16 = concat(axis = var_1348, interleave = input_29_interleave_0, values = (hidden_states_13_cast_fp16, var_1350_cast_fp16))[name = string("input_29_cast_fp16")]; + tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; + fp16 var_1345_to_fp16 = const()[name = string("op_1345_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_1345_to_fp16, x = input_29_cast_fp16)[name = string("normed_17_cast_fp16")]; + tensor normed_19_begin_0 = const()[name = string("normed_19_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_19_end_0 = const()[name = string("normed_19_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_19_end_mask_0 = const()[name = string("normed_19_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_19_cast_fp16 = slice_by_index(begin = normed_19_begin_0, end = normed_19_end_0, end_mask = normed_19_end_mask_0, x = normed_17_cast_fp16)[name = string("normed_19_cast_fp16")]; + tensor const_47_promoted_to_fp16 = const()[name = string("const_47_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311911552)))]; + tensor hidden_states_15_cast_fp16 = mul(x = normed_19_cast_fp16, y = const_47_promoted_to_fp16)[name = string("hidden_states_15_cast_fp16")]; + tensor var_1367 = const()[name = string("op_1367"), val = tensor([0, 2, 1])]; + tensor var_1370_axes_0 = const()[name = string("op_1370_axes_0"), val = tensor([2])]; + tensor var_1368_cast_fp16 = transpose(perm = var_1367, x = hidden_states_15_cast_fp16)[name = string("transpose_41")]; + tensor var_1370_cast_fp16 = expand_dims(axes = var_1370_axes_0, x = var_1368_cast_fp16)[name = string("op_1370_cast_fp16")]; + string var_1386_pad_type_0 = const()[name = string("op_1386_pad_type_0"), val = string("valid")]; + tensor var_1386_strides_0 = const()[name = string("op_1386_strides_0"), val = tensor([1, 1])]; + tensor var_1386_pad_0 = const()[name = string("op_1386_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1386_dilations_0 = const()[name = string("op_1386_dilations_0"), val = tensor([1, 1])]; + int32 var_1386_groups_0 = const()[name = string("op_1386_groups_0"), val = int32(1)]; + tensor var_1386 = conv(bias = model_model_layers_12_self_attn_q_proj_bias, dilations = var_1386_dilations_0, groups = var_1386_groups_0, pad = var_1386_pad_0, pad_type = var_1386_pad_type_0, strides = var_1386_strides_0, weight = model_model_layers_12_self_attn_q_proj_weight_palettized, x = var_1370_cast_fp16)[name = string("op_1386")]; + tensor var_1391 = const()[name = string("op_1391"), val = tensor([1, 12, 1, 128])]; + tensor var_1392 = reshape(shape = var_1391, x = var_1386)[name = string("op_1392")]; + string var_1408_pad_type_0 = const()[name = string("op_1408_pad_type_0"), val = string("valid")]; + tensor var_1408_strides_0 = const()[name = string("op_1408_strides_0"), val = tensor([1, 1])]; + tensor var_1408_pad_0 = const()[name = string("op_1408_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1408_dilations_0 = const()[name = string("op_1408_dilations_0"), val = tensor([1, 1])]; + int32 var_1408_groups_0 = const()[name = string("op_1408_groups_0"), val = int32(1)]; + tensor var_1408 = conv(bias = model_model_layers_12_self_attn_k_proj_bias, dilations = var_1408_dilations_0, groups = var_1408_groups_0, pad = var_1408_pad_0, pad_type = var_1408_pad_type_0, strides = var_1408_strides_0, weight = model_model_layers_12_self_attn_k_proj_weight_palettized, x = var_1370_cast_fp16)[name = string("op_1408")]; + tensor var_1413 = const()[name = string("op_1413"), val = tensor([1, 2, 1, 128])]; + tensor var_1414 = reshape(shape = var_1413, x = var_1408)[name = string("op_1414")]; + string var_1430_pad_type_0 = const()[name = string("op_1430_pad_type_0"), val = string("valid")]; + tensor var_1430_strides_0 = const()[name = string("op_1430_strides_0"), val = tensor([1, 1])]; + tensor var_1430_pad_0 = const()[name = string("op_1430_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1430_dilations_0 = const()[name = string("op_1430_dilations_0"), val = tensor([1, 1])]; + int32 var_1430_groups_0 = const()[name = string("op_1430_groups_0"), val = int32(1)]; + tensor var_1430 = conv(bias = model_model_layers_12_self_attn_v_proj_bias, dilations = var_1430_dilations_0, groups = var_1430_groups_0, pad = var_1430_pad_0, pad_type = var_1430_pad_type_0, strides = var_1430_strides_0, weight = model_model_layers_12_self_attn_v_proj_weight_palettized, x = var_1370_cast_fp16)[name = string("op_1430")]; + tensor var_1435 = const()[name = string("op_1435"), val = tensor([1, 2, 1, 128])]; + tensor var_1436 = reshape(shape = var_1435, x = var_1430)[name = string("op_1436")]; + tensor var_1442 = mul(x = var_1392, y = cos_1_cast_fp16)[name = string("op_1442")]; + tensor x1_9_begin_0 = const()[name = string("x1_9_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_9_end_0 = const()[name = string("x1_9_end_0"), val = tensor([1, 12, 1, 64])]; + tensor x1_9_end_mask_0 = const()[name = string("x1_9_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_9 = slice_by_index(begin = x1_9_begin_0, end = x1_9_end_0, end_mask = x1_9_end_mask_0, x = var_1392)[name = string("x1_9")]; + tensor x2_9_begin_0 = const()[name = string("x2_9_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_9_end_0 = const()[name = string("x2_9_end_0"), val = tensor([1, 12, 1, 128])]; + tensor x2_9_end_mask_0 = const()[name = string("x2_9_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_9 = slice_by_index(begin = x2_9_begin_0, end = x2_9_end_0, end_mask = x2_9_end_mask_0, x = var_1392)[name = string("x2_9")]; + fp16 const_50_promoted = const()[name = string("const_50_promoted"), val = fp16(-0x1p+0)]; + tensor var_1463 = mul(x = x2_9, y = const_50_promoted)[name = string("op_1463")]; + int32 var_1465 = const()[name = string("op_1465"), val = int32(-1)]; + bool var_1466_interleave_0 = const()[name = string("op_1466_interleave_0"), val = bool(false)]; + tensor var_1466 = concat(axis = var_1465, interleave = var_1466_interleave_0, values = (var_1463, x1_9))[name = string("op_1466")]; + tensor var_1467 = mul(x = var_1466, y = sin_1_cast_fp16)[name = string("op_1467")]; + tensor query_states_5 = add(x = var_1442, y = var_1467)[name = string("query_states_5")]; + tensor var_1470 = mul(x = var_1414, y = cos_1_cast_fp16)[name = string("op_1470")]; + tensor x1_11_begin_0 = const()[name = string("x1_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_11_end_0 = const()[name = string("x1_11_end_0"), val = tensor([1, 2, 1, 64])]; + tensor x1_11_end_mask_0 = const()[name = string("x1_11_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_11 = slice_by_index(begin = x1_11_begin_0, end = x1_11_end_0, end_mask = x1_11_end_mask_0, x = var_1414)[name = string("x1_11")]; + tensor x2_11_begin_0 = const()[name = string("x2_11_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_11_end_0 = const()[name = string("x2_11_end_0"), val = tensor([1, 2, 1, 128])]; + tensor x2_11_end_mask_0 = const()[name = string("x2_11_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_11 = slice_by_index(begin = x2_11_begin_0, end = x2_11_end_0, end_mask = x2_11_end_mask_0, x = var_1414)[name = string("x2_11")]; + fp16 const_53_promoted = const()[name = string("const_53_promoted"), val = fp16(-0x1p+0)]; + tensor var_1491 = mul(x = x2_11, y = const_53_promoted)[name = string("op_1491")]; + int32 var_1493 = const()[name = string("op_1493"), val = int32(-1)]; + bool var_1494_interleave_0 = const()[name = string("op_1494_interleave_0"), val = bool(false)]; + tensor var_1494 = concat(axis = var_1493, interleave = var_1494_interleave_0, values = (var_1491, x1_11))[name = string("op_1494")]; + tensor var_1495 = mul(x = var_1494, y = sin_1_cast_fp16)[name = string("op_1495")]; + tensor key_states_9 = add(x = var_1470, y = var_1495)[name = string("key_states_9")]; + tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([12])]; + tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; + tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; + tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([13])]; + int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; + bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; + tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_24, expand_dims_25, current_pos, expand_dims_27))[name = string("concat_18")]; + tensor concat_19_values1_0 = const()[name = string("concat_19_values1_0"), val = tensor([0])]; + tensor concat_19_values3_0 = const()[name = string("concat_19_values3_0"), val = tensor([0])]; + int32 concat_19_axis_0 = const()[name = string("concat_19_axis_0"), val = int32(0)]; + bool concat_19_interleave_0 = const()[name = string("concat_19_interleave_0"), val = bool(false)]; + tensor concat_19 = concat(axis = concat_19_axis_0, interleave = concat_19_interleave_0, values = (expand_dims_28, concat_19_values1_0, var_578, concat_19_values3_0))[name = string("concat_19")]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_18, begin_mask = model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0, end = concat_19, end_mask = model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_5_stride_0, update = key_states_9, x = coreml_update_state_21)[name = string("model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_4_write_state")]; + tensor coreml_update_state_22 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_4")]; + tensor expand_dims_30 = const()[name = string("expand_dims_30"), val = tensor([40])]; + tensor expand_dims_31 = const()[name = string("expand_dims_31"), val = tensor([0])]; + tensor expand_dims_33 = const()[name = string("expand_dims_33"), val = tensor([0])]; + tensor expand_dims_34 = const()[name = string("expand_dims_34"), val = tensor([41])]; + int32 concat_22_axis_0 = const()[name = string("concat_22_axis_0"), val = int32(0)]; + bool concat_22_interleave_0 = const()[name = string("concat_22_interleave_0"), val = bool(false)]; + tensor concat_22 = concat(axis = concat_22_axis_0, interleave = concat_22_interleave_0, values = (expand_dims_30, expand_dims_31, current_pos, expand_dims_33))[name = string("concat_22")]; + tensor concat_23_values1_0 = const()[name = string("concat_23_values1_0"), val = tensor([0])]; + tensor concat_23_values3_0 = const()[name = string("concat_23_values3_0"), val = tensor([0])]; + int32 concat_23_axis_0 = const()[name = string("concat_23_axis_0"), val = int32(0)]; + bool concat_23_interleave_0 = const()[name = string("concat_23_interleave_0"), val = bool(false)]; + tensor concat_23 = concat(axis = concat_23_axis_0, interleave = concat_23_interleave_0, values = (expand_dims_34, concat_23_values1_0, var_578, concat_23_values3_0))[name = string("concat_23")]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_22, begin_mask = model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0, end = concat_23, end_mask = model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_6_stride_0, update = var_1436, x = coreml_update_state_22)[name = string("model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_5_write_state")]; + tensor coreml_update_state_23 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_5")]; + tensor var_1550_begin_0 = const()[name = string("op_1550_begin_0"), val = tensor([12, 0, 0, 0])]; + tensor var_1550_end_0 = const()[name = string("op_1550_end_0"), val = tensor([13, 2, 2048, 128])]; + tensor var_1550_end_mask_0 = const()[name = string("op_1550_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1550_cast_fp16 = slice_by_index(begin = var_1550_begin_0, end = var_1550_end_0, end_mask = var_1550_end_mask_0, x = coreml_update_state_23)[name = string("op_1550_cast_fp16")]; + tensor K_layer_cache_5_axes_0 = const()[name = string("K_layer_cache_5_axes_0"), val = tensor([0])]; + tensor K_layer_cache_5_cast_fp16 = squeeze(axes = K_layer_cache_5_axes_0, x = var_1550_cast_fp16)[name = string("K_layer_cache_5_cast_fp16")]; + tensor var_1557_begin_0 = const()[name = string("op_1557_begin_0"), val = tensor([40, 0, 0, 0])]; + tensor var_1557_end_0 = const()[name = string("op_1557_end_0"), val = tensor([41, 2, 2048, 128])]; + tensor var_1557_end_mask_0 = const()[name = string("op_1557_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1557_cast_fp16 = slice_by_index(begin = var_1557_begin_0, end = var_1557_end_0, end_mask = var_1557_end_mask_0, x = coreml_update_state_23)[name = string("op_1557_cast_fp16")]; + tensor V_layer_cache_5_axes_0 = const()[name = string("V_layer_cache_5_axes_0"), val = tensor([0])]; + tensor V_layer_cache_5_cast_fp16 = squeeze(axes = V_layer_cache_5_axes_0, x = var_1557_cast_fp16)[name = string("V_layer_cache_5_cast_fp16")]; + tensor x_35_axes_0 = const()[name = string("x_35_axes_0"), val = tensor([1])]; + tensor x_35_cast_fp16 = expand_dims(axes = x_35_axes_0, x = K_layer_cache_5_cast_fp16)[name = string("x_35_cast_fp16")]; + tensor var_1594 = const()[name = string("op_1594"), val = tensor([1, 6, 1, 1])]; + tensor x_37_cast_fp16 = tile(reps = var_1594, x = x_35_cast_fp16)[name = string("x_37_cast_fp16")]; + tensor var_1606 = const()[name = string("op_1606"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_11_cast_fp16 = reshape(shape = var_1606, x = x_37_cast_fp16)[name = string("key_states_11_cast_fp16")]; + tensor x_41_axes_0 = const()[name = string("x_41_axes_0"), val = tensor([1])]; + tensor x_41_cast_fp16 = expand_dims(axes = x_41_axes_0, x = V_layer_cache_5_cast_fp16)[name = string("x_41_cast_fp16")]; + tensor var_1614 = const()[name = string("op_1614"), val = tensor([1, 6, 1, 1])]; + tensor x_43_cast_fp16 = tile(reps = var_1614, x = x_41_cast_fp16)[name = string("x_43_cast_fp16")]; + tensor var_1626 = const()[name = string("op_1626"), val = tensor([1, -1, 2048, 128])]; + tensor value_states_11_cast_fp16 = reshape(shape = var_1626, x = x_43_cast_fp16)[name = string("value_states_11_cast_fp16")]; + bool var_1649_transpose_x_1 = const()[name = string("op_1649_transpose_x_1"), val = bool(false)]; + bool var_1649_transpose_y_1 = const()[name = string("op_1649_transpose_y_1"), val = bool(true)]; + tensor var_1649_cast_fp16 = matmul(transpose_x = var_1649_transpose_x_1, transpose_y = var_1649_transpose_y_1, x = query_states_5, y = key_states_11_cast_fp16)[name = string("op_1649_cast_fp16")]; + fp16 var_1650_to_fp16 = const()[name = string("op_1650_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_9_cast_fp16 = mul(x = var_1649_cast_fp16, y = var_1650_to_fp16)[name = string("attn_logits_9_cast_fp16")]; + tensor attn_logits_11_cast_fp16 = add(x = attn_logits_9_cast_fp16, y = causal_mask)[name = string("attn_logits_11_cast_fp16")]; + int32 var_1677 = const()[name = string("op_1677"), val = int32(-1)]; + tensor var_1679_cast_fp16 = softmax(axis = var_1677, x = attn_logits_11_cast_fp16)[name = string("op_1679_cast_fp16")]; + bool attn_output_25_transpose_x_0 = const()[name = string("attn_output_25_transpose_x_0"), val = bool(false)]; + bool attn_output_25_transpose_y_0 = const()[name = string("attn_output_25_transpose_y_0"), val = bool(false)]; + tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_0, transpose_y = attn_output_25_transpose_y_0, x = var_1679_cast_fp16, y = value_states_11_cast_fp16)[name = string("attn_output_25_cast_fp16")]; + tensor var_1703_perm_0 = const()[name = string("op_1703_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1707 = const()[name = string("op_1707"), val = tensor([1, 1, 1536])]; + tensor var_1703 = transpose(perm = var_1703_perm_0, x = attn_output_25_cast_fp16)[name = string("transpose_40")]; + tensor attn_output_31 = reshape(shape = var_1707, x = var_1703)[name = string("attn_output_31")]; + tensor var_1712 = const()[name = string("op_1712"), val = tensor([0, 2, 1])]; + tensor squeeze_2_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311914688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313684224))))[name = string("squeeze_2_palettized")]; + string var_1728_pad_type_0 = const()[name = string("op_1728_pad_type_0"), val = string("valid")]; + int32 var_1728_groups_0 = const()[name = string("op_1728_groups_0"), val = int32(1)]; + tensor var_1728_strides_0 = const()[name = string("op_1728_strides_0"), val = tensor([1])]; + tensor var_1728_pad_0 = const()[name = string("op_1728_pad_0"), val = tensor([0, 0])]; + tensor var_1728_dilations_0 = const()[name = string("op_1728_dilations_0"), val = tensor([1])]; + tensor var_1713 = transpose(perm = var_1712, x = attn_output_31)[name = string("transpose_39")]; + tensor var_1728 = conv(dilations = var_1728_dilations_0, groups = var_1728_groups_0, pad = var_1728_pad_0, pad_type = var_1728_pad_type_0, strides = var_1728_strides_0, weight = squeeze_2_palettized, x = var_1713)[name = string("op_1728")]; + tensor var_1732 = const()[name = string("op_1732"), val = tensor([0, 2, 1])]; + tensor attn_output_35 = transpose(perm = var_1732, x = var_1728)[name = string("transpose_38")]; + tensor hidden_states_17_cast_fp16 = add(x = hidden_states_13_cast_fp16, y = attn_output_35)[name = string("hidden_states_17_cast_fp16")]; + int32 var_1745 = const()[name = string("op_1745"), val = int32(-1)]; + fp16 const_62_promoted_to_fp16 = const()[name = string("const_62_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1747_cast_fp16 = mul(x = hidden_states_17_cast_fp16, y = const_62_promoted_to_fp16)[name = string("op_1747_cast_fp16")]; + bool input_35_interleave_0 = const()[name = string("input_35_interleave_0"), val = bool(false)]; + tensor input_35_cast_fp16 = concat(axis = var_1745, interleave = input_35_interleave_0, values = (hidden_states_17_cast_fp16, var_1747_cast_fp16))[name = string("input_35_cast_fp16")]; + tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; + fp16 var_1742_to_fp16 = const()[name = string("op_1742_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_1742_to_fp16, x = input_35_cast_fp16)[name = string("normed_21_cast_fp16")]; + tensor normed_23_begin_0 = const()[name = string("normed_23_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_23_end_0 = const()[name = string("normed_23_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_23_end_mask_0 = const()[name = string("normed_23_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_23_cast_fp16 = slice_by_index(begin = normed_23_begin_0, end = normed_23_end_0, end_mask = normed_23_end_mask_0, x = normed_21_cast_fp16)[name = string("normed_23_cast_fp16")]; + tensor const_65_promoted_to_fp16 = const()[name = string("const_65_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313733440)))]; + tensor x_45_cast_fp16 = mul(x = normed_23_cast_fp16, y = const_65_promoted_to_fp16)[name = string("x_45_cast_fp16")]; + tensor var_1772 = const()[name = string("op_1772"), val = tensor([0, 2, 1])]; + tensor input_37_axes_0 = const()[name = string("input_37_axes_0"), val = tensor([2])]; + tensor var_1773 = transpose(perm = var_1772, x = x_45_cast_fp16)[name = string("transpose_37")]; + tensor input_37 = expand_dims(axes = input_37_axes_0, x = var_1773)[name = string("input_37")]; + string input_39_pad_type_0 = const()[name = string("input_39_pad_type_0"), val = string("valid")]; + tensor input_39_strides_0 = const()[name = string("input_39_strides_0"), val = tensor([1, 1])]; + tensor input_39_pad_0 = const()[name = string("input_39_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_39_dilations_0 = const()[name = string("input_39_dilations_0"), val = tensor([1, 1])]; + int32 input_39_groups_0 = const()[name = string("input_39_groups_0"), val = int32(1)]; + tensor input_39 = conv(dilations = input_39_dilations_0, groups = input_39_groups_0, pad = input_39_pad_0, pad_type = input_39_pad_type_0, strides = input_39_strides_0, weight = model_model_layers_12_mlp_gate_proj_weight_palettized, x = input_37)[name = string("input_39")]; + string b_5_pad_type_0 = const()[name = string("b_5_pad_type_0"), val = string("valid")]; + tensor b_5_strides_0 = const()[name = string("b_5_strides_0"), val = tensor([1, 1])]; + tensor b_5_pad_0 = const()[name = string("b_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_5_dilations_0 = const()[name = string("b_5_dilations_0"), val = tensor([1, 1])]; + int32 b_5_groups_0 = const()[name = string("b_5_groups_0"), val = int32(1)]; + tensor b_5 = conv(dilations = b_5_dilations_0, groups = b_5_groups_0, pad = b_5_pad_0, pad_type = b_5_pad_type_0, strides = b_5_strides_0, weight = model_model_layers_12_mlp_up_proj_weight_palettized, x = input_37)[name = string("b_5")]; + tensor c_5 = silu(x = input_39)[name = string("c_5")]; + tensor input_41 = mul(x = c_5, y = b_5)[name = string("input_41")]; + string e_5_pad_type_0 = const()[name = string("e_5_pad_type_0"), val = string("valid")]; + tensor e_5_strides_0 = const()[name = string("e_5_strides_0"), val = tensor([1, 1])]; + tensor e_5_pad_0 = const()[name = string("e_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_5_dilations_0 = const()[name = string("e_5_dilations_0"), val = tensor([1, 1])]; + int32 e_5_groups_0 = const()[name = string("e_5_groups_0"), val = int32(1)]; + tensor e_5 = conv(dilations = e_5_dilations_0, groups = e_5_groups_0, pad = e_5_pad_0, pad_type = e_5_pad_type_0, strides = e_5_strides_0, weight = model_model_layers_12_mlp_down_proj_weight_palettized, x = input_41)[name = string("e_5")]; + tensor var_1795_axes_0 = const()[name = string("op_1795_axes_0"), val = tensor([2])]; + tensor var_1795 = squeeze(axes = var_1795_axes_0, x = e_5)[name = string("op_1795")]; + tensor var_1796 = const()[name = string("op_1796"), val = tensor([0, 2, 1])]; + tensor var_1797 = transpose(perm = var_1796, x = var_1795)[name = string("transpose_36")]; + tensor hidden_states_19_cast_fp16 = add(x = hidden_states_17_cast_fp16, y = var_1797)[name = string("hidden_states_19_cast_fp16")]; + int32 var_1809 = const()[name = string("op_1809"), val = int32(-1)]; + fp16 const_66_promoted_to_fp16 = const()[name = string("const_66_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1811_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_66_promoted_to_fp16)[name = string("op_1811_cast_fp16")]; + bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; + tensor input_43_cast_fp16 = concat(axis = var_1809, interleave = input_43_interleave_0, values = (hidden_states_19_cast_fp16, var_1811_cast_fp16))[name = string("input_43_cast_fp16")]; + tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; + fp16 var_1806_to_fp16 = const()[name = string("op_1806_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_1806_to_fp16, x = input_43_cast_fp16)[name = string("normed_25_cast_fp16")]; + tensor normed_27_begin_0 = const()[name = string("normed_27_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_27_end_0 = const()[name = string("normed_27_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_27_end_mask_0 = const()[name = string("normed_27_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_27_cast_fp16 = slice_by_index(begin = normed_27_begin_0, end = normed_27_end_0, end_mask = normed_27_end_mask_0, x = normed_25_cast_fp16)[name = string("normed_27_cast_fp16")]; + tensor const_69_promoted_to_fp16 = const()[name = string("const_69_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313736576)))]; + tensor hidden_states_21_cast_fp16 = mul(x = normed_27_cast_fp16, y = const_69_promoted_to_fp16)[name = string("hidden_states_21_cast_fp16")]; + tensor var_1828 = const()[name = string("op_1828"), val = tensor([0, 2, 1])]; + tensor var_1831_axes_0 = const()[name = string("op_1831_axes_0"), val = tensor([2])]; + tensor var_1829_cast_fp16 = transpose(perm = var_1828, x = hidden_states_21_cast_fp16)[name = string("transpose_35")]; + tensor var_1831_cast_fp16 = expand_dims(axes = var_1831_axes_0, x = var_1829_cast_fp16)[name = string("op_1831_cast_fp16")]; + string var_1847_pad_type_0 = const()[name = string("op_1847_pad_type_0"), val = string("valid")]; + tensor var_1847_strides_0 = const()[name = string("op_1847_strides_0"), val = tensor([1, 1])]; + tensor var_1847_pad_0 = const()[name = string("op_1847_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1847_dilations_0 = const()[name = string("op_1847_dilations_0"), val = tensor([1, 1])]; + int32 var_1847_groups_0 = const()[name = string("op_1847_groups_0"), val = int32(1)]; + tensor var_1847 = conv(bias = model_model_layers_13_self_attn_q_proj_bias, dilations = var_1847_dilations_0, groups = var_1847_groups_0, pad = var_1847_pad_0, pad_type = var_1847_pad_type_0, strides = var_1847_strides_0, weight = model_model_layers_13_self_attn_q_proj_weight_palettized, x = var_1831_cast_fp16)[name = string("op_1847")]; + tensor var_1852 = const()[name = string("op_1852"), val = tensor([1, 12, 1, 128])]; + tensor var_1853 = reshape(shape = var_1852, x = var_1847)[name = string("op_1853")]; + string var_1869_pad_type_0 = const()[name = string("op_1869_pad_type_0"), val = string("valid")]; + tensor var_1869_strides_0 = const()[name = string("op_1869_strides_0"), val = tensor([1, 1])]; + tensor var_1869_pad_0 = const()[name = string("op_1869_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1869_dilations_0 = const()[name = string("op_1869_dilations_0"), val = tensor([1, 1])]; + int32 var_1869_groups_0 = const()[name = string("op_1869_groups_0"), val = int32(1)]; + tensor var_1869 = conv(bias = model_model_layers_13_self_attn_k_proj_bias, dilations = var_1869_dilations_0, groups = var_1869_groups_0, pad = var_1869_pad_0, pad_type = var_1869_pad_type_0, strides = var_1869_strides_0, weight = model_model_layers_13_self_attn_k_proj_weight_palettized, x = var_1831_cast_fp16)[name = string("op_1869")]; + tensor var_1874 = const()[name = string("op_1874"), val = tensor([1, 2, 1, 128])]; + tensor var_1875 = reshape(shape = var_1874, x = var_1869)[name = string("op_1875")]; + string var_1891_pad_type_0 = const()[name = string("op_1891_pad_type_0"), val = string("valid")]; + tensor var_1891_strides_0 = const()[name = string("op_1891_strides_0"), val = tensor([1, 1])]; + tensor var_1891_pad_0 = const()[name = string("op_1891_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1891_dilations_0 = const()[name = string("op_1891_dilations_0"), val = tensor([1, 1])]; + int32 var_1891_groups_0 = const()[name = string("op_1891_groups_0"), val = int32(1)]; + tensor var_1891 = conv(bias = model_model_layers_13_self_attn_v_proj_bias, dilations = var_1891_dilations_0, groups = var_1891_groups_0, pad = var_1891_pad_0, pad_type = var_1891_pad_type_0, strides = var_1891_strides_0, weight = model_model_layers_13_self_attn_v_proj_weight_palettized, x = var_1831_cast_fp16)[name = string("op_1891")]; + tensor var_1896 = const()[name = string("op_1896"), val = tensor([1, 2, 1, 128])]; + tensor var_1897 = reshape(shape = var_1896, x = var_1891)[name = string("op_1897")]; + tensor var_1903 = mul(x = var_1853, y = cos_1_cast_fp16)[name = string("op_1903")]; + tensor x1_13_begin_0 = const()[name = string("x1_13_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_13_end_0 = const()[name = string("x1_13_end_0"), val = tensor([1, 12, 1, 64])]; + tensor x1_13_end_mask_0 = const()[name = string("x1_13_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_13 = slice_by_index(begin = x1_13_begin_0, end = x1_13_end_0, end_mask = x1_13_end_mask_0, x = var_1853)[name = string("x1_13")]; + tensor x2_13_begin_0 = const()[name = string("x2_13_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_13_end_0 = const()[name = string("x2_13_end_0"), val = tensor([1, 12, 1, 128])]; + tensor x2_13_end_mask_0 = const()[name = string("x2_13_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_13 = slice_by_index(begin = x2_13_begin_0, end = x2_13_end_0, end_mask = x2_13_end_mask_0, x = var_1853)[name = string("x2_13")]; + fp16 const_72_promoted = const()[name = string("const_72_promoted"), val = fp16(-0x1p+0)]; + tensor var_1924 = mul(x = x2_13, y = const_72_promoted)[name = string("op_1924")]; + int32 var_1926 = const()[name = string("op_1926"), val = int32(-1)]; + bool var_1927_interleave_0 = const()[name = string("op_1927_interleave_0"), val = bool(false)]; + tensor var_1927 = concat(axis = var_1926, interleave = var_1927_interleave_0, values = (var_1924, x1_13))[name = string("op_1927")]; + tensor var_1928 = mul(x = var_1927, y = sin_1_cast_fp16)[name = string("op_1928")]; + tensor query_states_7 = add(x = var_1903, y = var_1928)[name = string("query_states_7")]; + tensor var_1931 = mul(x = var_1875, y = cos_1_cast_fp16)[name = string("op_1931")]; + tensor x1_15_begin_0 = const()[name = string("x1_15_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_15_end_0 = const()[name = string("x1_15_end_0"), val = tensor([1, 2, 1, 64])]; + tensor x1_15_end_mask_0 = const()[name = string("x1_15_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_15 = slice_by_index(begin = x1_15_begin_0, end = x1_15_end_0, end_mask = x1_15_end_mask_0, x = var_1875)[name = string("x1_15")]; + tensor x2_15_begin_0 = const()[name = string("x2_15_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_15_end_0 = const()[name = string("x2_15_end_0"), val = tensor([1, 2, 1, 128])]; + tensor x2_15_end_mask_0 = const()[name = string("x2_15_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_15 = slice_by_index(begin = x2_15_begin_0, end = x2_15_end_0, end_mask = x2_15_end_mask_0, x = var_1875)[name = string("x2_15")]; + fp16 const_75_promoted = const()[name = string("const_75_promoted"), val = fp16(-0x1p+0)]; + tensor var_1952 = mul(x = x2_15, y = const_75_promoted)[name = string("op_1952")]; + int32 var_1954 = const()[name = string("op_1954"), val = int32(-1)]; + bool var_1955_interleave_0 = const()[name = string("op_1955_interleave_0"), val = bool(false)]; + tensor var_1955 = concat(axis = var_1954, interleave = var_1955_interleave_0, values = (var_1952, x1_15))[name = string("op_1955")]; + tensor var_1956 = mul(x = var_1955, y = sin_1_cast_fp16)[name = string("op_1956")]; + tensor key_states_13 = add(x = var_1931, y = var_1956)[name = string("key_states_13")]; + tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([13])]; + tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; + tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; + tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([14])]; + int32 concat_26_axis_0 = const()[name = string("concat_26_axis_0"), val = int32(0)]; + bool concat_26_interleave_0 = const()[name = string("concat_26_interleave_0"), val = bool(false)]; + tensor concat_26 = concat(axis = concat_26_axis_0, interleave = concat_26_interleave_0, values = (expand_dims_36, expand_dims_37, current_pos, expand_dims_39))[name = string("concat_26")]; + tensor concat_27_values1_0 = const()[name = string("concat_27_values1_0"), val = tensor([0])]; + tensor concat_27_values3_0 = const()[name = string("concat_27_values3_0"), val = tensor([0])]; + int32 concat_27_axis_0 = const()[name = string("concat_27_axis_0"), val = int32(0)]; + bool concat_27_interleave_0 = const()[name = string("concat_27_interleave_0"), val = bool(false)]; + tensor concat_27 = concat(axis = concat_27_axis_0, interleave = concat_27_interleave_0, values = (expand_dims_40, concat_27_values1_0, var_578, concat_27_values3_0))[name = string("concat_27")]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_26, begin_mask = model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0, end = concat_27, end_mask = model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_7_stride_0, update = key_states_13, x = coreml_update_state_23)[name = string("model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_6_write_state")]; + tensor coreml_update_state_24 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_6")]; + tensor expand_dims_42 = const()[name = string("expand_dims_42"), val = tensor([41])]; + tensor expand_dims_43 = const()[name = string("expand_dims_43"), val = tensor([0])]; + tensor expand_dims_45 = const()[name = string("expand_dims_45"), val = tensor([0])]; + tensor expand_dims_46 = const()[name = string("expand_dims_46"), val = tensor([42])]; + int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; + bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; + tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_42, expand_dims_43, current_pos, expand_dims_45))[name = string("concat_30")]; + tensor concat_31_values1_0 = const()[name = string("concat_31_values1_0"), val = tensor([0])]; + tensor concat_31_values3_0 = const()[name = string("concat_31_values3_0"), val = tensor([0])]; + int32 concat_31_axis_0 = const()[name = string("concat_31_axis_0"), val = int32(0)]; + bool concat_31_interleave_0 = const()[name = string("concat_31_interleave_0"), val = bool(false)]; + tensor concat_31 = concat(axis = concat_31_axis_0, interleave = concat_31_interleave_0, values = (expand_dims_46, concat_31_values1_0, var_578, concat_31_values3_0))[name = string("concat_31")]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_30, begin_mask = model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0, end = concat_31, end_mask = model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_8_stride_0, update = var_1897, x = coreml_update_state_24)[name = string("model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_7_write_state")]; + tensor coreml_update_state_25 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_7")]; + tensor var_2011_begin_0 = const()[name = string("op_2011_begin_0"), val = tensor([13, 0, 0, 0])]; + tensor var_2011_end_0 = const()[name = string("op_2011_end_0"), val = tensor([14, 2, 2048, 128])]; + tensor var_2011_end_mask_0 = const()[name = string("op_2011_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2011_cast_fp16 = slice_by_index(begin = var_2011_begin_0, end = var_2011_end_0, end_mask = var_2011_end_mask_0, x = coreml_update_state_25)[name = string("op_2011_cast_fp16")]; + tensor K_layer_cache_7_axes_0 = const()[name = string("K_layer_cache_7_axes_0"), val = tensor([0])]; + tensor K_layer_cache_7_cast_fp16 = squeeze(axes = K_layer_cache_7_axes_0, x = var_2011_cast_fp16)[name = string("K_layer_cache_7_cast_fp16")]; + tensor var_2018_begin_0 = const()[name = string("op_2018_begin_0"), val = tensor([41, 0, 0, 0])]; + tensor var_2018_end_0 = const()[name = string("op_2018_end_0"), val = tensor([42, 2, 2048, 128])]; + tensor var_2018_end_mask_0 = const()[name = string("op_2018_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2018_cast_fp16 = slice_by_index(begin = var_2018_begin_0, end = var_2018_end_0, end_mask = var_2018_end_mask_0, x = coreml_update_state_25)[name = string("op_2018_cast_fp16")]; + tensor V_layer_cache_7_axes_0 = const()[name = string("V_layer_cache_7_axes_0"), val = tensor([0])]; + tensor V_layer_cache_7_cast_fp16 = squeeze(axes = V_layer_cache_7_axes_0, x = var_2018_cast_fp16)[name = string("V_layer_cache_7_cast_fp16")]; + tensor x_51_axes_0 = const()[name = string("x_51_axes_0"), val = tensor([1])]; + tensor x_51_cast_fp16 = expand_dims(axes = x_51_axes_0, x = K_layer_cache_7_cast_fp16)[name = string("x_51_cast_fp16")]; + tensor var_2055 = const()[name = string("op_2055"), val = tensor([1, 6, 1, 1])]; + tensor x_53_cast_fp16 = tile(reps = var_2055, x = x_51_cast_fp16)[name = string("x_53_cast_fp16")]; + tensor var_2067 = const()[name = string("op_2067"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_15_cast_fp16 = reshape(shape = var_2067, x = x_53_cast_fp16)[name = string("key_states_15_cast_fp16")]; + tensor x_57_axes_0 = const()[name = string("x_57_axes_0"), val = tensor([1])]; + tensor x_57_cast_fp16 = expand_dims(axes = x_57_axes_0, x = V_layer_cache_7_cast_fp16)[name = string("x_57_cast_fp16")]; + tensor var_2075 = const()[name = string("op_2075"), val = tensor([1, 6, 1, 1])]; + tensor x_59_cast_fp16 = tile(reps = var_2075, x = x_57_cast_fp16)[name = string("x_59_cast_fp16")]; + tensor var_2087 = const()[name = string("op_2087"), val = tensor([1, -1, 2048, 128])]; + tensor value_states_15_cast_fp16 = reshape(shape = var_2087, x = x_59_cast_fp16)[name = string("value_states_15_cast_fp16")]; + bool var_2110_transpose_x_1 = const()[name = string("op_2110_transpose_x_1"), val = bool(false)]; + bool var_2110_transpose_y_1 = const()[name = string("op_2110_transpose_y_1"), val = bool(true)]; + tensor var_2110_cast_fp16 = matmul(transpose_x = var_2110_transpose_x_1, transpose_y = var_2110_transpose_y_1, x = query_states_7, y = key_states_15_cast_fp16)[name = string("op_2110_cast_fp16")]; + fp16 var_2111_to_fp16 = const()[name = string("op_2111_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_13_cast_fp16 = mul(x = var_2110_cast_fp16, y = var_2111_to_fp16)[name = string("attn_logits_13_cast_fp16")]; + tensor attn_logits_15_cast_fp16 = add(x = attn_logits_13_cast_fp16, y = causal_mask)[name = string("attn_logits_15_cast_fp16")]; + int32 var_2138 = const()[name = string("op_2138"), val = int32(-1)]; + tensor var_2140_cast_fp16 = softmax(axis = var_2138, x = attn_logits_15_cast_fp16)[name = string("op_2140_cast_fp16")]; + bool attn_output_37_transpose_x_0 = const()[name = string("attn_output_37_transpose_x_0"), val = bool(false)]; + bool attn_output_37_transpose_y_0 = const()[name = string("attn_output_37_transpose_y_0"), val = bool(false)]; + tensor attn_output_37_cast_fp16 = matmul(transpose_x = attn_output_37_transpose_x_0, transpose_y = attn_output_37_transpose_y_0, x = var_2140_cast_fp16, y = value_states_15_cast_fp16)[name = string("attn_output_37_cast_fp16")]; + tensor var_2164_perm_0 = const()[name = string("op_2164_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2168 = const()[name = string("op_2168"), val = tensor([1, 1, 1536])]; + tensor var_2164 = transpose(perm = var_2164_perm_0, x = attn_output_37_cast_fp16)[name = string("transpose_34")]; + tensor attn_output_43 = reshape(shape = var_2168, x = var_2164)[name = string("attn_output_43")]; + tensor var_2173 = const()[name = string("op_2173"), val = tensor([0, 2, 1])]; + tensor squeeze_3_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313739712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315509248))))[name = string("squeeze_3_palettized")]; + string var_2189_pad_type_0 = const()[name = string("op_2189_pad_type_0"), val = string("valid")]; + int32 var_2189_groups_0 = const()[name = string("op_2189_groups_0"), val = int32(1)]; + tensor var_2189_strides_0 = const()[name = string("op_2189_strides_0"), val = tensor([1])]; + tensor var_2189_pad_0 = const()[name = string("op_2189_pad_0"), val = tensor([0, 0])]; + tensor var_2189_dilations_0 = const()[name = string("op_2189_dilations_0"), val = tensor([1])]; + tensor var_2174 = transpose(perm = var_2173, x = attn_output_43)[name = string("transpose_33")]; + tensor var_2189 = conv(dilations = var_2189_dilations_0, groups = var_2189_groups_0, pad = var_2189_pad_0, pad_type = var_2189_pad_type_0, strides = var_2189_strides_0, weight = squeeze_3_palettized, x = var_2174)[name = string("op_2189")]; + tensor var_2193 = const()[name = string("op_2193"), val = tensor([0, 2, 1])]; + tensor attn_output_47 = transpose(perm = var_2193, x = var_2189)[name = string("transpose_32")]; + tensor hidden_states_23_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = attn_output_47)[name = string("hidden_states_23_cast_fp16")]; + int32 var_2206 = const()[name = string("op_2206"), val = int32(-1)]; + fp16 const_84_promoted_to_fp16 = const()[name = string("const_84_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2208_cast_fp16 = mul(x = hidden_states_23_cast_fp16, y = const_84_promoted_to_fp16)[name = string("op_2208_cast_fp16")]; + bool input_49_interleave_0 = const()[name = string("input_49_interleave_0"), val = bool(false)]; + tensor input_49_cast_fp16 = concat(axis = var_2206, interleave = input_49_interleave_0, values = (hidden_states_23_cast_fp16, var_2208_cast_fp16))[name = string("input_49_cast_fp16")]; + tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; + fp16 var_2203_to_fp16 = const()[name = string("op_2203_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_2203_to_fp16, x = input_49_cast_fp16)[name = string("normed_29_cast_fp16")]; + tensor normed_31_begin_0 = const()[name = string("normed_31_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_31_end_0 = const()[name = string("normed_31_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_31_end_mask_0 = const()[name = string("normed_31_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_31_cast_fp16 = slice_by_index(begin = normed_31_begin_0, end = normed_31_end_0, end_mask = normed_31_end_mask_0, x = normed_29_cast_fp16)[name = string("normed_31_cast_fp16")]; + tensor const_87_promoted_to_fp16 = const()[name = string("const_87_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315558464)))]; + tensor x_61_cast_fp16 = mul(x = normed_31_cast_fp16, y = const_87_promoted_to_fp16)[name = string("x_61_cast_fp16")]; + tensor var_2233 = const()[name = string("op_2233"), val = tensor([0, 2, 1])]; + tensor input_51_axes_0 = const()[name = string("input_51_axes_0"), val = tensor([2])]; + tensor var_2234 = transpose(perm = var_2233, x = x_61_cast_fp16)[name = string("transpose_31")]; + tensor input_51 = expand_dims(axes = input_51_axes_0, x = var_2234)[name = string("input_51")]; + string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; + tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; + tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; + int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; + tensor input_53 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = model_model_layers_13_mlp_gate_proj_weight_palettized, x = input_51)[name = string("input_53")]; + string b_7_pad_type_0 = const()[name = string("b_7_pad_type_0"), val = string("valid")]; + tensor b_7_strides_0 = const()[name = string("b_7_strides_0"), val = tensor([1, 1])]; + tensor b_7_pad_0 = const()[name = string("b_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_7_dilations_0 = const()[name = string("b_7_dilations_0"), val = tensor([1, 1])]; + int32 b_7_groups_0 = const()[name = string("b_7_groups_0"), val = int32(1)]; + tensor b_7 = conv(dilations = b_7_dilations_0, groups = b_7_groups_0, pad = b_7_pad_0, pad_type = b_7_pad_type_0, strides = b_7_strides_0, weight = model_model_layers_13_mlp_up_proj_weight_palettized, x = input_51)[name = string("b_7")]; + tensor c_7 = silu(x = input_53)[name = string("c_7")]; + tensor input_55 = mul(x = c_7, y = b_7)[name = string("input_55")]; + string e_7_pad_type_0 = const()[name = string("e_7_pad_type_0"), val = string("valid")]; + tensor e_7_strides_0 = const()[name = string("e_7_strides_0"), val = tensor([1, 1])]; + tensor e_7_pad_0 = const()[name = string("e_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_7_dilations_0 = const()[name = string("e_7_dilations_0"), val = tensor([1, 1])]; + int32 e_7_groups_0 = const()[name = string("e_7_groups_0"), val = int32(1)]; + tensor e_7 = conv(dilations = e_7_dilations_0, groups = e_7_groups_0, pad = e_7_pad_0, pad_type = e_7_pad_type_0, strides = e_7_strides_0, weight = model_model_layers_13_mlp_down_proj_weight_palettized, x = input_55)[name = string("e_7")]; + tensor var_2256_axes_0 = const()[name = string("op_2256_axes_0"), val = tensor([2])]; + tensor var_2256 = squeeze(axes = var_2256_axes_0, x = e_7)[name = string("op_2256")]; + tensor var_2257 = const()[name = string("op_2257"), val = tensor([0, 2, 1])]; + tensor var_2258 = transpose(perm = var_2257, x = var_2256)[name = string("transpose_30")]; + tensor hidden_states_25_cast_fp16 = add(x = hidden_states_23_cast_fp16, y = var_2258)[name = string("hidden_states_25_cast_fp16")]; + int32 var_2270 = const()[name = string("op_2270"), val = int32(-1)]; + fp16 const_88_promoted_to_fp16 = const()[name = string("const_88_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2272_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_88_promoted_to_fp16)[name = string("op_2272_cast_fp16")]; + bool input_57_interleave_0 = const()[name = string("input_57_interleave_0"), val = bool(false)]; + tensor input_57_cast_fp16 = concat(axis = var_2270, interleave = input_57_interleave_0, values = (hidden_states_25_cast_fp16, var_2272_cast_fp16))[name = string("input_57_cast_fp16")]; + tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; + fp16 var_2267_to_fp16 = const()[name = string("op_2267_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_2267_to_fp16, x = input_57_cast_fp16)[name = string("normed_33_cast_fp16")]; + tensor normed_35_begin_0 = const()[name = string("normed_35_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_35_end_0 = const()[name = string("normed_35_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_35_end_mask_0 = const()[name = string("normed_35_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_35_cast_fp16 = slice_by_index(begin = normed_35_begin_0, end = normed_35_end_0, end_mask = normed_35_end_mask_0, x = normed_33_cast_fp16)[name = string("normed_35_cast_fp16")]; + tensor const_91_promoted_to_fp16 = const()[name = string("const_91_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315561600)))]; + tensor hidden_states_27_cast_fp16 = mul(x = normed_35_cast_fp16, y = const_91_promoted_to_fp16)[name = string("hidden_states_27_cast_fp16")]; + tensor var_2289 = const()[name = string("op_2289"), val = tensor([0, 2, 1])]; + tensor var_2292_axes_0 = const()[name = string("op_2292_axes_0"), val = tensor([2])]; + tensor var_2290_cast_fp16 = transpose(perm = var_2289, x = hidden_states_27_cast_fp16)[name = string("transpose_29")]; + tensor var_2292_cast_fp16 = expand_dims(axes = var_2292_axes_0, x = var_2290_cast_fp16)[name = string("op_2292_cast_fp16")]; + string var_2308_pad_type_0 = const()[name = string("op_2308_pad_type_0"), val = string("valid")]; + tensor var_2308_strides_0 = const()[name = string("op_2308_strides_0"), val = tensor([1, 1])]; + tensor var_2308_pad_0 = const()[name = string("op_2308_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2308_dilations_0 = const()[name = string("op_2308_dilations_0"), val = tensor([1, 1])]; + int32 var_2308_groups_0 = const()[name = string("op_2308_groups_0"), val = int32(1)]; + tensor var_2308 = conv(bias = model_model_layers_14_self_attn_q_proj_bias, dilations = var_2308_dilations_0, groups = var_2308_groups_0, pad = var_2308_pad_0, pad_type = var_2308_pad_type_0, strides = var_2308_strides_0, weight = model_model_layers_14_self_attn_q_proj_weight_palettized, x = var_2292_cast_fp16)[name = string("op_2308")]; + tensor var_2313 = const()[name = string("op_2313"), val = tensor([1, 12, 1, 128])]; + tensor var_2314 = reshape(shape = var_2313, x = var_2308)[name = string("op_2314")]; + string var_2330_pad_type_0 = const()[name = string("op_2330_pad_type_0"), val = string("valid")]; + tensor var_2330_strides_0 = const()[name = string("op_2330_strides_0"), val = tensor([1, 1])]; + tensor var_2330_pad_0 = const()[name = string("op_2330_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2330_dilations_0 = const()[name = string("op_2330_dilations_0"), val = tensor([1, 1])]; + int32 var_2330_groups_0 = const()[name = string("op_2330_groups_0"), val = int32(1)]; + tensor var_2330 = conv(bias = model_model_layers_14_self_attn_k_proj_bias, dilations = var_2330_dilations_0, groups = var_2330_groups_0, pad = var_2330_pad_0, pad_type = var_2330_pad_type_0, strides = var_2330_strides_0, weight = model_model_layers_14_self_attn_k_proj_weight_palettized, x = var_2292_cast_fp16)[name = string("op_2330")]; + tensor var_2335 = const()[name = string("op_2335"), val = tensor([1, 2, 1, 128])]; + tensor var_2336 = reshape(shape = var_2335, x = var_2330)[name = string("op_2336")]; + string var_2352_pad_type_0 = const()[name = string("op_2352_pad_type_0"), val = string("valid")]; + tensor var_2352_strides_0 = const()[name = string("op_2352_strides_0"), val = tensor([1, 1])]; + tensor var_2352_pad_0 = const()[name = string("op_2352_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2352_dilations_0 = const()[name = string("op_2352_dilations_0"), val = tensor([1, 1])]; + int32 var_2352_groups_0 = const()[name = string("op_2352_groups_0"), val = int32(1)]; + tensor var_2352 = conv(bias = model_model_layers_14_self_attn_v_proj_bias, dilations = var_2352_dilations_0, groups = var_2352_groups_0, pad = var_2352_pad_0, pad_type = var_2352_pad_type_0, strides = var_2352_strides_0, weight = model_model_layers_14_self_attn_v_proj_weight_palettized, x = var_2292_cast_fp16)[name = string("op_2352")]; + tensor var_2357 = const()[name = string("op_2357"), val = tensor([1, 2, 1, 128])]; + tensor var_2358 = reshape(shape = var_2357, x = var_2352)[name = string("op_2358")]; + tensor var_2364 = mul(x = var_2314, y = cos_1_cast_fp16)[name = string("op_2364")]; + tensor x1_17_begin_0 = const()[name = string("x1_17_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_17_end_0 = const()[name = string("x1_17_end_0"), val = tensor([1, 12, 1, 64])]; + tensor x1_17_end_mask_0 = const()[name = string("x1_17_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_17 = slice_by_index(begin = x1_17_begin_0, end = x1_17_end_0, end_mask = x1_17_end_mask_0, x = var_2314)[name = string("x1_17")]; + tensor x2_17_begin_0 = const()[name = string("x2_17_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_17_end_0 = const()[name = string("x2_17_end_0"), val = tensor([1, 12, 1, 128])]; + tensor x2_17_end_mask_0 = const()[name = string("x2_17_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_17 = slice_by_index(begin = x2_17_begin_0, end = x2_17_end_0, end_mask = x2_17_end_mask_0, x = var_2314)[name = string("x2_17")]; + fp16 const_94_promoted = const()[name = string("const_94_promoted"), val = fp16(-0x1p+0)]; + tensor var_2385 = mul(x = x2_17, y = const_94_promoted)[name = string("op_2385")]; + int32 var_2387 = const()[name = string("op_2387"), val = int32(-1)]; + bool var_2388_interleave_0 = const()[name = string("op_2388_interleave_0"), val = bool(false)]; + tensor var_2388 = concat(axis = var_2387, interleave = var_2388_interleave_0, values = (var_2385, x1_17))[name = string("op_2388")]; + tensor var_2389 = mul(x = var_2388, y = sin_1_cast_fp16)[name = string("op_2389")]; + tensor query_states_9 = add(x = var_2364, y = var_2389)[name = string("query_states_9")]; + tensor var_2392 = mul(x = var_2336, y = cos_1_cast_fp16)[name = string("op_2392")]; + tensor x1_19_begin_0 = const()[name = string("x1_19_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_19_end_0 = const()[name = string("x1_19_end_0"), val = tensor([1, 2, 1, 64])]; + tensor x1_19_end_mask_0 = const()[name = string("x1_19_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_19 = slice_by_index(begin = x1_19_begin_0, end = x1_19_end_0, end_mask = x1_19_end_mask_0, x = var_2336)[name = string("x1_19")]; + tensor x2_19_begin_0 = const()[name = string("x2_19_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_19_end_0 = const()[name = string("x2_19_end_0"), val = tensor([1, 2, 1, 128])]; + tensor x2_19_end_mask_0 = const()[name = string("x2_19_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_19 = slice_by_index(begin = x2_19_begin_0, end = x2_19_end_0, end_mask = x2_19_end_mask_0, x = var_2336)[name = string("x2_19")]; + fp16 const_97_promoted = const()[name = string("const_97_promoted"), val = fp16(-0x1p+0)]; + tensor var_2413 = mul(x = x2_19, y = const_97_promoted)[name = string("op_2413")]; + int32 var_2415 = const()[name = string("op_2415"), val = int32(-1)]; + bool var_2416_interleave_0 = const()[name = string("op_2416_interleave_0"), val = bool(false)]; + tensor var_2416 = concat(axis = var_2415, interleave = var_2416_interleave_0, values = (var_2413, x1_19))[name = string("op_2416")]; + tensor var_2417 = mul(x = var_2416, y = sin_1_cast_fp16)[name = string("op_2417")]; + tensor key_states_17 = add(x = var_2392, y = var_2417)[name = string("key_states_17")]; + tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([14])]; + tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; + tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; + tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([15])]; + int32 concat_34_axis_0 = const()[name = string("concat_34_axis_0"), val = int32(0)]; + bool concat_34_interleave_0 = const()[name = string("concat_34_interleave_0"), val = bool(false)]; + tensor concat_34 = concat(axis = concat_34_axis_0, interleave = concat_34_interleave_0, values = (expand_dims_48, expand_dims_49, current_pos, expand_dims_51))[name = string("concat_34")]; + tensor concat_35_values1_0 = const()[name = string("concat_35_values1_0"), val = tensor([0])]; + tensor concat_35_values3_0 = const()[name = string("concat_35_values3_0"), val = tensor([0])]; + int32 concat_35_axis_0 = const()[name = string("concat_35_axis_0"), val = int32(0)]; + bool concat_35_interleave_0 = const()[name = string("concat_35_interleave_0"), val = bool(false)]; + tensor concat_35 = concat(axis = concat_35_axis_0, interleave = concat_35_interleave_0, values = (expand_dims_52, concat_35_values1_0, var_578, concat_35_values3_0))[name = string("concat_35")]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_34, begin_mask = model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0, end = concat_35, end_mask = model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_9_stride_0, update = key_states_17, x = coreml_update_state_25)[name = string("model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_8_write_state")]; + tensor coreml_update_state_26 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_8")]; + tensor expand_dims_54 = const()[name = string("expand_dims_54"), val = tensor([42])]; + tensor expand_dims_55 = const()[name = string("expand_dims_55"), val = tensor([0])]; + tensor expand_dims_57 = const()[name = string("expand_dims_57"), val = tensor([0])]; + tensor expand_dims_58 = const()[name = string("expand_dims_58"), val = tensor([43])]; + int32 concat_38_axis_0 = const()[name = string("concat_38_axis_0"), val = int32(0)]; + bool concat_38_interleave_0 = const()[name = string("concat_38_interleave_0"), val = bool(false)]; + tensor concat_38 = concat(axis = concat_38_axis_0, interleave = concat_38_interleave_0, values = (expand_dims_54, expand_dims_55, current_pos, expand_dims_57))[name = string("concat_38")]; + tensor concat_39_values1_0 = const()[name = string("concat_39_values1_0"), val = tensor([0])]; + tensor concat_39_values3_0 = const()[name = string("concat_39_values3_0"), val = tensor([0])]; + int32 concat_39_axis_0 = const()[name = string("concat_39_axis_0"), val = int32(0)]; + bool concat_39_interleave_0 = const()[name = string("concat_39_interleave_0"), val = bool(false)]; + tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (expand_dims_58, concat_39_values1_0, var_578, concat_39_values3_0))[name = string("concat_39")]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_38, begin_mask = model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0, end = concat_39, end_mask = model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_10_stride_0, update = var_2358, x = coreml_update_state_26)[name = string("model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_9_write_state")]; + tensor coreml_update_state_27 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_9")]; + tensor var_2472_begin_0 = const()[name = string("op_2472_begin_0"), val = tensor([14, 0, 0, 0])]; + tensor var_2472_end_0 = const()[name = string("op_2472_end_0"), val = tensor([15, 2, 2048, 128])]; + tensor var_2472_end_mask_0 = const()[name = string("op_2472_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2472_cast_fp16 = slice_by_index(begin = var_2472_begin_0, end = var_2472_end_0, end_mask = var_2472_end_mask_0, x = coreml_update_state_27)[name = string("op_2472_cast_fp16")]; + tensor K_layer_cache_9_axes_0 = const()[name = string("K_layer_cache_9_axes_0"), val = tensor([0])]; + tensor K_layer_cache_9_cast_fp16 = squeeze(axes = K_layer_cache_9_axes_0, x = var_2472_cast_fp16)[name = string("K_layer_cache_9_cast_fp16")]; + tensor var_2479_begin_0 = const()[name = string("op_2479_begin_0"), val = tensor([42, 0, 0, 0])]; + tensor var_2479_end_0 = const()[name = string("op_2479_end_0"), val = tensor([43, 2, 2048, 128])]; + tensor var_2479_end_mask_0 = const()[name = string("op_2479_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2479_cast_fp16 = slice_by_index(begin = var_2479_begin_0, end = var_2479_end_0, end_mask = var_2479_end_mask_0, x = coreml_update_state_27)[name = string("op_2479_cast_fp16")]; + tensor V_layer_cache_9_axes_0 = const()[name = string("V_layer_cache_9_axes_0"), val = tensor([0])]; + tensor V_layer_cache_9_cast_fp16 = squeeze(axes = V_layer_cache_9_axes_0, x = var_2479_cast_fp16)[name = string("V_layer_cache_9_cast_fp16")]; + tensor x_67_axes_0 = const()[name = string("x_67_axes_0"), val = tensor([1])]; + tensor x_67_cast_fp16 = expand_dims(axes = x_67_axes_0, x = K_layer_cache_9_cast_fp16)[name = string("x_67_cast_fp16")]; + tensor var_2516 = const()[name = string("op_2516"), val = tensor([1, 6, 1, 1])]; + tensor x_69_cast_fp16 = tile(reps = var_2516, x = x_67_cast_fp16)[name = string("x_69_cast_fp16")]; + tensor var_2528 = const()[name = string("op_2528"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_19_cast_fp16 = reshape(shape = var_2528, x = x_69_cast_fp16)[name = string("key_states_19_cast_fp16")]; + tensor x_73_axes_0 = const()[name = string("x_73_axes_0"), val = tensor([1])]; + tensor x_73_cast_fp16 = expand_dims(axes = x_73_axes_0, x = V_layer_cache_9_cast_fp16)[name = string("x_73_cast_fp16")]; + tensor var_2536 = const()[name = string("op_2536"), val = tensor([1, 6, 1, 1])]; + tensor x_75_cast_fp16 = tile(reps = var_2536, x = x_73_cast_fp16)[name = string("x_75_cast_fp16")]; + tensor var_2548 = const()[name = string("op_2548"), val = tensor([1, -1, 2048, 128])]; + tensor value_states_19_cast_fp16 = reshape(shape = var_2548, x = x_75_cast_fp16)[name = string("value_states_19_cast_fp16")]; + bool var_2571_transpose_x_1 = const()[name = string("op_2571_transpose_x_1"), val = bool(false)]; + bool var_2571_transpose_y_1 = const()[name = string("op_2571_transpose_y_1"), val = bool(true)]; + tensor var_2571_cast_fp16 = matmul(transpose_x = var_2571_transpose_x_1, transpose_y = var_2571_transpose_y_1, x = query_states_9, y = key_states_19_cast_fp16)[name = string("op_2571_cast_fp16")]; + fp16 var_2572_to_fp16 = const()[name = string("op_2572_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_17_cast_fp16 = mul(x = var_2571_cast_fp16, y = var_2572_to_fp16)[name = string("attn_logits_17_cast_fp16")]; + tensor attn_logits_19_cast_fp16 = add(x = attn_logits_17_cast_fp16, y = causal_mask)[name = string("attn_logits_19_cast_fp16")]; + int32 var_2599 = const()[name = string("op_2599"), val = int32(-1)]; + tensor var_2601_cast_fp16 = softmax(axis = var_2599, x = attn_logits_19_cast_fp16)[name = string("op_2601_cast_fp16")]; + bool attn_output_49_transpose_x_0 = const()[name = string("attn_output_49_transpose_x_0"), val = bool(false)]; + bool attn_output_49_transpose_y_0 = const()[name = string("attn_output_49_transpose_y_0"), val = bool(false)]; + tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_0, transpose_y = attn_output_49_transpose_y_0, x = var_2601_cast_fp16, y = value_states_19_cast_fp16)[name = string("attn_output_49_cast_fp16")]; + tensor var_2625_perm_0 = const()[name = string("op_2625_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2629 = const()[name = string("op_2629"), val = tensor([1, 1, 1536])]; + tensor var_2625 = transpose(perm = var_2625_perm_0, x = attn_output_49_cast_fp16)[name = string("transpose_28")]; + tensor attn_output_55 = reshape(shape = var_2629, x = var_2625)[name = string("attn_output_55")]; + tensor var_2634 = const()[name = string("op_2634"), val = tensor([0, 2, 1])]; + tensor squeeze_4_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315564736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(317334272))))[name = string("squeeze_4_palettized")]; + string var_2650_pad_type_0 = const()[name = string("op_2650_pad_type_0"), val = string("valid")]; + int32 var_2650_groups_0 = const()[name = string("op_2650_groups_0"), val = int32(1)]; + tensor var_2650_strides_0 = const()[name = string("op_2650_strides_0"), val = tensor([1])]; + tensor var_2650_pad_0 = const()[name = string("op_2650_pad_0"), val = tensor([0, 0])]; + tensor var_2650_dilations_0 = const()[name = string("op_2650_dilations_0"), val = tensor([1])]; + tensor var_2635 = transpose(perm = var_2634, x = attn_output_55)[name = string("transpose_27")]; + tensor var_2650 = conv(dilations = var_2650_dilations_0, groups = var_2650_groups_0, pad = var_2650_pad_0, pad_type = var_2650_pad_type_0, strides = var_2650_strides_0, weight = squeeze_4_palettized, x = var_2635)[name = string("op_2650")]; + tensor var_2654 = const()[name = string("op_2654"), val = tensor([0, 2, 1])]; + tensor attn_output_59 = transpose(perm = var_2654, x = var_2650)[name = string("transpose_26")]; + tensor hidden_states_29_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = attn_output_59)[name = string("hidden_states_29_cast_fp16")]; + int32 var_2667 = const()[name = string("op_2667"), val = int32(-1)]; + fp16 const_106_promoted_to_fp16 = const()[name = string("const_106_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2669_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_106_promoted_to_fp16)[name = string("op_2669_cast_fp16")]; + bool input_63_interleave_0 = const()[name = string("input_63_interleave_0"), val = bool(false)]; + tensor input_63_cast_fp16 = concat(axis = var_2667, interleave = input_63_interleave_0, values = (hidden_states_29_cast_fp16, var_2669_cast_fp16))[name = string("input_63_cast_fp16")]; + tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; + fp16 var_2664_to_fp16 = const()[name = string("op_2664_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_2664_to_fp16, x = input_63_cast_fp16)[name = string("normed_37_cast_fp16")]; + tensor normed_39_begin_0 = const()[name = string("normed_39_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_39_end_0 = const()[name = string("normed_39_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_39_end_mask_0 = const()[name = string("normed_39_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_39_cast_fp16 = slice_by_index(begin = normed_39_begin_0, end = normed_39_end_0, end_mask = normed_39_end_mask_0, x = normed_37_cast_fp16)[name = string("normed_39_cast_fp16")]; + tensor const_109_promoted_to_fp16 = const()[name = string("const_109_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(317383488)))]; + tensor x_77_cast_fp16 = mul(x = normed_39_cast_fp16, y = const_109_promoted_to_fp16)[name = string("x_77_cast_fp16")]; + tensor var_2694 = const()[name = string("op_2694"), val = tensor([0, 2, 1])]; + tensor input_65_axes_0 = const()[name = string("input_65_axes_0"), val = tensor([2])]; + tensor var_2695 = transpose(perm = var_2694, x = x_77_cast_fp16)[name = string("transpose_25")]; + tensor input_65 = expand_dims(axes = input_65_axes_0, x = var_2695)[name = string("input_65")]; + string input_67_pad_type_0 = const()[name = string("input_67_pad_type_0"), val = string("valid")]; + tensor input_67_strides_0 = const()[name = string("input_67_strides_0"), val = tensor([1, 1])]; + tensor input_67_pad_0 = const()[name = string("input_67_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_67_dilations_0 = const()[name = string("input_67_dilations_0"), val = tensor([1, 1])]; + int32 input_67_groups_0 = const()[name = string("input_67_groups_0"), val = int32(1)]; + tensor input_67 = conv(dilations = input_67_dilations_0, groups = input_67_groups_0, pad = input_67_pad_0, pad_type = input_67_pad_type_0, strides = input_67_strides_0, weight = model_model_layers_14_mlp_gate_proj_weight_palettized, x = input_65)[name = string("input_67")]; + string b_9_pad_type_0 = const()[name = string("b_9_pad_type_0"), val = string("valid")]; + tensor b_9_strides_0 = const()[name = string("b_9_strides_0"), val = tensor([1, 1])]; + tensor b_9_pad_0 = const()[name = string("b_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_9_dilations_0 = const()[name = string("b_9_dilations_0"), val = tensor([1, 1])]; + int32 b_9_groups_0 = const()[name = string("b_9_groups_0"), val = int32(1)]; + tensor b_9 = conv(dilations = b_9_dilations_0, groups = b_9_groups_0, pad = b_9_pad_0, pad_type = b_9_pad_type_0, strides = b_9_strides_0, weight = model_model_layers_14_mlp_up_proj_weight_palettized, x = input_65)[name = string("b_9")]; + tensor c_9 = silu(x = input_67)[name = string("c_9")]; + tensor input_69 = mul(x = c_9, y = b_9)[name = string("input_69")]; + string e_9_pad_type_0 = const()[name = string("e_9_pad_type_0"), val = string("valid")]; + tensor e_9_strides_0 = const()[name = string("e_9_strides_0"), val = tensor([1, 1])]; + tensor e_9_pad_0 = const()[name = string("e_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_9_dilations_0 = const()[name = string("e_9_dilations_0"), val = tensor([1, 1])]; + int32 e_9_groups_0 = const()[name = string("e_9_groups_0"), val = int32(1)]; + tensor e_9 = conv(dilations = e_9_dilations_0, groups = e_9_groups_0, pad = e_9_pad_0, pad_type = e_9_pad_type_0, strides = e_9_strides_0, weight = model_model_layers_14_mlp_down_proj_weight_palettized, x = input_69)[name = string("e_9")]; + tensor var_2717_axes_0 = const()[name = string("op_2717_axes_0"), val = tensor([2])]; + tensor var_2717 = squeeze(axes = var_2717_axes_0, x = e_9)[name = string("op_2717")]; + tensor var_2718 = const()[name = string("op_2718"), val = tensor([0, 2, 1])]; + tensor var_2719 = transpose(perm = var_2718, x = var_2717)[name = string("transpose_24")]; + tensor hidden_states_31_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = var_2719)[name = string("hidden_states_31_cast_fp16")]; + int32 var_2731 = const()[name = string("op_2731"), val = int32(-1)]; + fp16 const_110_promoted_to_fp16 = const()[name = string("const_110_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2733_cast_fp16 = mul(x = hidden_states_31_cast_fp16, y = const_110_promoted_to_fp16)[name = string("op_2733_cast_fp16")]; + bool input_71_interleave_0 = const()[name = string("input_71_interleave_0"), val = bool(false)]; + tensor input_71_cast_fp16 = concat(axis = var_2731, interleave = input_71_interleave_0, values = (hidden_states_31_cast_fp16, var_2733_cast_fp16))[name = string("input_71_cast_fp16")]; + tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; + fp16 var_2728_to_fp16 = const()[name = string("op_2728_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_2728_to_fp16, x = input_71_cast_fp16)[name = string("normed_41_cast_fp16")]; + tensor normed_43_begin_0 = const()[name = string("normed_43_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_43_end_0 = const()[name = string("normed_43_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_43_end_mask_0 = const()[name = string("normed_43_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_43_cast_fp16 = slice_by_index(begin = normed_43_begin_0, end = normed_43_end_0, end_mask = normed_43_end_mask_0, x = normed_41_cast_fp16)[name = string("normed_43_cast_fp16")]; + tensor const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(317386624)))]; + tensor hidden_states_33_cast_fp16 = mul(x = normed_43_cast_fp16, y = const_113_promoted_to_fp16)[name = string("hidden_states_33_cast_fp16")]; + tensor var_2750 = const()[name = string("op_2750"), val = tensor([0, 2, 1])]; + tensor var_2753_axes_0 = const()[name = string("op_2753_axes_0"), val = tensor([2])]; + tensor var_2751_cast_fp16 = transpose(perm = var_2750, x = hidden_states_33_cast_fp16)[name = string("transpose_23")]; + tensor var_2753_cast_fp16 = expand_dims(axes = var_2753_axes_0, x = var_2751_cast_fp16)[name = string("op_2753_cast_fp16")]; + string var_2769_pad_type_0 = const()[name = string("op_2769_pad_type_0"), val = string("valid")]; + tensor var_2769_strides_0 = const()[name = string("op_2769_strides_0"), val = tensor([1, 1])]; + tensor var_2769_pad_0 = const()[name = string("op_2769_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2769_dilations_0 = const()[name = string("op_2769_dilations_0"), val = tensor([1, 1])]; + int32 var_2769_groups_0 = const()[name = string("op_2769_groups_0"), val = int32(1)]; + tensor var_2769 = conv(bias = model_model_layers_15_self_attn_q_proj_bias, dilations = var_2769_dilations_0, groups = var_2769_groups_0, pad = var_2769_pad_0, pad_type = var_2769_pad_type_0, strides = var_2769_strides_0, weight = model_model_layers_15_self_attn_q_proj_weight_palettized, x = var_2753_cast_fp16)[name = string("op_2769")]; + tensor var_2774 = const()[name = string("op_2774"), val = tensor([1, 12, 1, 128])]; + tensor var_2775 = reshape(shape = var_2774, x = var_2769)[name = string("op_2775")]; + string var_2791_pad_type_0 = const()[name = string("op_2791_pad_type_0"), val = string("valid")]; + tensor var_2791_strides_0 = const()[name = string("op_2791_strides_0"), val = tensor([1, 1])]; + tensor var_2791_pad_0 = const()[name = string("op_2791_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2791_dilations_0 = const()[name = string("op_2791_dilations_0"), val = tensor([1, 1])]; + int32 var_2791_groups_0 = const()[name = string("op_2791_groups_0"), val = int32(1)]; + tensor var_2791 = conv(bias = model_model_layers_15_self_attn_k_proj_bias, dilations = var_2791_dilations_0, groups = var_2791_groups_0, pad = var_2791_pad_0, pad_type = var_2791_pad_type_0, strides = var_2791_strides_0, weight = model_model_layers_15_self_attn_k_proj_weight_palettized, x = var_2753_cast_fp16)[name = string("op_2791")]; + tensor var_2796 = const()[name = string("op_2796"), val = tensor([1, 2, 1, 128])]; + tensor var_2797 = reshape(shape = var_2796, x = var_2791)[name = string("op_2797")]; + string var_2813_pad_type_0 = const()[name = string("op_2813_pad_type_0"), val = string("valid")]; + tensor var_2813_strides_0 = const()[name = string("op_2813_strides_0"), val = tensor([1, 1])]; + tensor var_2813_pad_0 = const()[name = string("op_2813_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2813_dilations_0 = const()[name = string("op_2813_dilations_0"), val = tensor([1, 1])]; + int32 var_2813_groups_0 = const()[name = string("op_2813_groups_0"), val = int32(1)]; + tensor var_2813 = conv(bias = model_model_layers_15_self_attn_v_proj_bias, dilations = var_2813_dilations_0, groups = var_2813_groups_0, pad = var_2813_pad_0, pad_type = var_2813_pad_type_0, strides = var_2813_strides_0, weight = model_model_layers_15_self_attn_v_proj_weight_palettized, x = var_2753_cast_fp16)[name = string("op_2813")]; + tensor var_2818 = const()[name = string("op_2818"), val = tensor([1, 2, 1, 128])]; + tensor var_2819 = reshape(shape = var_2818, x = var_2813)[name = string("op_2819")]; + tensor var_2825 = mul(x = var_2775, y = cos_1_cast_fp16)[name = string("op_2825")]; + tensor x1_21_begin_0 = const()[name = string("x1_21_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_21_end_0 = const()[name = string("x1_21_end_0"), val = tensor([1, 12, 1, 64])]; + tensor x1_21_end_mask_0 = const()[name = string("x1_21_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_21 = slice_by_index(begin = x1_21_begin_0, end = x1_21_end_0, end_mask = x1_21_end_mask_0, x = var_2775)[name = string("x1_21")]; + tensor x2_21_begin_0 = const()[name = string("x2_21_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_21_end_0 = const()[name = string("x2_21_end_0"), val = tensor([1, 12, 1, 128])]; + tensor x2_21_end_mask_0 = const()[name = string("x2_21_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_21 = slice_by_index(begin = x2_21_begin_0, end = x2_21_end_0, end_mask = x2_21_end_mask_0, x = var_2775)[name = string("x2_21")]; + fp16 const_116_promoted = const()[name = string("const_116_promoted"), val = fp16(-0x1p+0)]; + tensor var_2846 = mul(x = x2_21, y = const_116_promoted)[name = string("op_2846")]; + int32 var_2848 = const()[name = string("op_2848"), val = int32(-1)]; + bool var_2849_interleave_0 = const()[name = string("op_2849_interleave_0"), val = bool(false)]; + tensor var_2849 = concat(axis = var_2848, interleave = var_2849_interleave_0, values = (var_2846, x1_21))[name = string("op_2849")]; + tensor var_2850 = mul(x = var_2849, y = sin_1_cast_fp16)[name = string("op_2850")]; + tensor query_states_11 = add(x = var_2825, y = var_2850)[name = string("query_states_11")]; + tensor var_2853 = mul(x = var_2797, y = cos_1_cast_fp16)[name = string("op_2853")]; + tensor x1_23_begin_0 = const()[name = string("x1_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_23_end_0 = const()[name = string("x1_23_end_0"), val = tensor([1, 2, 1, 64])]; + tensor x1_23_end_mask_0 = const()[name = string("x1_23_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_23 = slice_by_index(begin = x1_23_begin_0, end = x1_23_end_0, end_mask = x1_23_end_mask_0, x = var_2797)[name = string("x1_23")]; + tensor x2_23_begin_0 = const()[name = string("x2_23_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_23_end_0 = const()[name = string("x2_23_end_0"), val = tensor([1, 2, 1, 128])]; + tensor x2_23_end_mask_0 = const()[name = string("x2_23_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_23 = slice_by_index(begin = x2_23_begin_0, end = x2_23_end_0, end_mask = x2_23_end_mask_0, x = var_2797)[name = string("x2_23")]; + fp16 const_119_promoted = const()[name = string("const_119_promoted"), val = fp16(-0x1p+0)]; + tensor var_2874 = mul(x = x2_23, y = const_119_promoted)[name = string("op_2874")]; + int32 var_2876 = const()[name = string("op_2876"), val = int32(-1)]; + bool var_2877_interleave_0 = const()[name = string("op_2877_interleave_0"), val = bool(false)]; + tensor var_2877 = concat(axis = var_2876, interleave = var_2877_interleave_0, values = (var_2874, x1_23))[name = string("op_2877")]; + tensor var_2878 = mul(x = var_2877, y = sin_1_cast_fp16)[name = string("op_2878")]; + tensor key_states_21 = add(x = var_2853, y = var_2878)[name = string("key_states_21")]; + tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([15])]; + tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; + tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; + tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([16])]; + int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; + bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; + tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_60, expand_dims_61, current_pos, expand_dims_63))[name = string("concat_42")]; + tensor concat_43_values1_0 = const()[name = string("concat_43_values1_0"), val = tensor([0])]; + tensor concat_43_values3_0 = const()[name = string("concat_43_values3_0"), val = tensor([0])]; + int32 concat_43_axis_0 = const()[name = string("concat_43_axis_0"), val = int32(0)]; + bool concat_43_interleave_0 = const()[name = string("concat_43_interleave_0"), val = bool(false)]; + tensor concat_43 = concat(axis = concat_43_axis_0, interleave = concat_43_interleave_0, values = (expand_dims_64, concat_43_values1_0, var_578, concat_43_values3_0))[name = string("concat_43")]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_42, begin_mask = model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0, end = concat_43, end_mask = model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_11_stride_0, update = key_states_21, x = coreml_update_state_27)[name = string("model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_10_write_state")]; + tensor coreml_update_state_28 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_10")]; + tensor expand_dims_66 = const()[name = string("expand_dims_66"), val = tensor([43])]; + tensor expand_dims_67 = const()[name = string("expand_dims_67"), val = tensor([0])]; + tensor expand_dims_69 = const()[name = string("expand_dims_69"), val = tensor([0])]; + tensor expand_dims_70 = const()[name = string("expand_dims_70"), val = tensor([44])]; + int32 concat_46_axis_0 = const()[name = string("concat_46_axis_0"), val = int32(0)]; + bool concat_46_interleave_0 = const()[name = string("concat_46_interleave_0"), val = bool(false)]; + tensor concat_46 = concat(axis = concat_46_axis_0, interleave = concat_46_interleave_0, values = (expand_dims_66, expand_dims_67, current_pos, expand_dims_69))[name = string("concat_46")]; + tensor concat_47_values1_0 = const()[name = string("concat_47_values1_0"), val = tensor([0])]; + tensor concat_47_values3_0 = const()[name = string("concat_47_values3_0"), val = tensor([0])]; + int32 concat_47_axis_0 = const()[name = string("concat_47_axis_0"), val = int32(0)]; + bool concat_47_interleave_0 = const()[name = string("concat_47_interleave_0"), val = bool(false)]; + tensor concat_47 = concat(axis = concat_47_axis_0, interleave = concat_47_interleave_0, values = (expand_dims_70, concat_47_values1_0, var_578, concat_47_values3_0))[name = string("concat_47")]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_46, begin_mask = model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0, end = concat_47, end_mask = model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_12_stride_0, update = var_2819, x = coreml_update_state_28)[name = string("model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_11_write_state")]; + tensor coreml_update_state_29 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_11")]; + tensor var_2933_begin_0 = const()[name = string("op_2933_begin_0"), val = tensor([15, 0, 0, 0])]; + tensor var_2933_end_0 = const()[name = string("op_2933_end_0"), val = tensor([16, 2, 2048, 128])]; + tensor var_2933_end_mask_0 = const()[name = string("op_2933_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2933_cast_fp16 = slice_by_index(begin = var_2933_begin_0, end = var_2933_end_0, end_mask = var_2933_end_mask_0, x = coreml_update_state_29)[name = string("op_2933_cast_fp16")]; + tensor K_layer_cache_11_axes_0 = const()[name = string("K_layer_cache_11_axes_0"), val = tensor([0])]; + tensor K_layer_cache_11_cast_fp16 = squeeze(axes = K_layer_cache_11_axes_0, x = var_2933_cast_fp16)[name = string("K_layer_cache_11_cast_fp16")]; + tensor var_2940_begin_0 = const()[name = string("op_2940_begin_0"), val = tensor([43, 0, 0, 0])]; + tensor var_2940_end_0 = const()[name = string("op_2940_end_0"), val = tensor([44, 2, 2048, 128])]; + tensor var_2940_end_mask_0 = const()[name = string("op_2940_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2940_cast_fp16 = slice_by_index(begin = var_2940_begin_0, end = var_2940_end_0, end_mask = var_2940_end_mask_0, x = coreml_update_state_29)[name = string("op_2940_cast_fp16")]; + tensor V_layer_cache_11_axes_0 = const()[name = string("V_layer_cache_11_axes_0"), val = tensor([0])]; + tensor V_layer_cache_11_cast_fp16 = squeeze(axes = V_layer_cache_11_axes_0, x = var_2940_cast_fp16)[name = string("V_layer_cache_11_cast_fp16")]; + tensor x_83_axes_0 = const()[name = string("x_83_axes_0"), val = tensor([1])]; + tensor x_83_cast_fp16 = expand_dims(axes = x_83_axes_0, x = K_layer_cache_11_cast_fp16)[name = string("x_83_cast_fp16")]; + tensor var_2977 = const()[name = string("op_2977"), val = tensor([1, 6, 1, 1])]; + tensor x_85_cast_fp16 = tile(reps = var_2977, x = x_83_cast_fp16)[name = string("x_85_cast_fp16")]; + tensor var_2989 = const()[name = string("op_2989"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_23_cast_fp16 = reshape(shape = var_2989, x = x_85_cast_fp16)[name = string("key_states_23_cast_fp16")]; + tensor x_89_axes_0 = const()[name = string("x_89_axes_0"), val = tensor([1])]; + tensor x_89_cast_fp16 = expand_dims(axes = x_89_axes_0, x = V_layer_cache_11_cast_fp16)[name = string("x_89_cast_fp16")]; + tensor var_2997 = const()[name = string("op_2997"), val = tensor([1, 6, 1, 1])]; + tensor x_91_cast_fp16 = tile(reps = var_2997, x = x_89_cast_fp16)[name = string("x_91_cast_fp16")]; + tensor var_3009 = const()[name = string("op_3009"), val = tensor([1, -1, 2048, 128])]; + tensor value_states_23_cast_fp16 = reshape(shape = var_3009, x = x_91_cast_fp16)[name = string("value_states_23_cast_fp16")]; + bool var_3032_transpose_x_1 = const()[name = string("op_3032_transpose_x_1"), val = bool(false)]; + bool var_3032_transpose_y_1 = const()[name = string("op_3032_transpose_y_1"), val = bool(true)]; + tensor var_3032_cast_fp16 = matmul(transpose_x = var_3032_transpose_x_1, transpose_y = var_3032_transpose_y_1, x = query_states_11, y = key_states_23_cast_fp16)[name = string("op_3032_cast_fp16")]; + fp16 var_3033_to_fp16 = const()[name = string("op_3033_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_21_cast_fp16 = mul(x = var_3032_cast_fp16, y = var_3033_to_fp16)[name = string("attn_logits_21_cast_fp16")]; + tensor attn_logits_23_cast_fp16 = add(x = attn_logits_21_cast_fp16, y = causal_mask)[name = string("attn_logits_23_cast_fp16")]; + int32 var_3060 = const()[name = string("op_3060"), val = int32(-1)]; + tensor var_3062_cast_fp16 = softmax(axis = var_3060, x = attn_logits_23_cast_fp16)[name = string("op_3062_cast_fp16")]; + bool attn_output_61_transpose_x_0 = const()[name = string("attn_output_61_transpose_x_0"), val = bool(false)]; + bool attn_output_61_transpose_y_0 = const()[name = string("attn_output_61_transpose_y_0"), val = bool(false)]; + tensor attn_output_61_cast_fp16 = matmul(transpose_x = attn_output_61_transpose_x_0, transpose_y = attn_output_61_transpose_y_0, x = var_3062_cast_fp16, y = value_states_23_cast_fp16)[name = string("attn_output_61_cast_fp16")]; + tensor var_3086_perm_0 = const()[name = string("op_3086_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_3090 = const()[name = string("op_3090"), val = tensor([1, 1, 1536])]; + tensor var_3086 = transpose(perm = var_3086_perm_0, x = attn_output_61_cast_fp16)[name = string("transpose_22")]; + tensor attn_output_67 = reshape(shape = var_3090, x = var_3086)[name = string("attn_output_67")]; + tensor var_3095 = const()[name = string("op_3095"), val = tensor([0, 2, 1])]; + tensor squeeze_5_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(317389760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(319159296))))[name = string("squeeze_5_palettized")]; + string var_3111_pad_type_0 = const()[name = string("op_3111_pad_type_0"), val = string("valid")]; + int32 var_3111_groups_0 = const()[name = string("op_3111_groups_0"), val = int32(1)]; + tensor var_3111_strides_0 = const()[name = string("op_3111_strides_0"), val = tensor([1])]; + tensor var_3111_pad_0 = const()[name = string("op_3111_pad_0"), val = tensor([0, 0])]; + tensor var_3111_dilations_0 = const()[name = string("op_3111_dilations_0"), val = tensor([1])]; + tensor var_3096 = transpose(perm = var_3095, x = attn_output_67)[name = string("transpose_21")]; + tensor var_3111 = conv(dilations = var_3111_dilations_0, groups = var_3111_groups_0, pad = var_3111_pad_0, pad_type = var_3111_pad_type_0, strides = var_3111_strides_0, weight = squeeze_5_palettized, x = var_3096)[name = string("op_3111")]; + tensor var_3115 = const()[name = string("op_3115"), val = tensor([0, 2, 1])]; + tensor attn_output_71 = transpose(perm = var_3115, x = var_3111)[name = string("transpose_20")]; + tensor hidden_states_35_cast_fp16 = add(x = hidden_states_31_cast_fp16, y = attn_output_71)[name = string("hidden_states_35_cast_fp16")]; + int32 var_3128 = const()[name = string("op_3128"), val = int32(-1)]; + fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3130_cast_fp16 = mul(x = hidden_states_35_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_3130_cast_fp16")]; + bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; + tensor input_77_cast_fp16 = concat(axis = var_3128, interleave = input_77_interleave_0, values = (hidden_states_35_cast_fp16, var_3130_cast_fp16))[name = string("input_77_cast_fp16")]; + tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; + fp16 var_3125_to_fp16 = const()[name = string("op_3125_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_3125_to_fp16, x = input_77_cast_fp16)[name = string("normed_45_cast_fp16")]; + tensor normed_47_begin_0 = const()[name = string("normed_47_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_47_end_0 = const()[name = string("normed_47_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_47_end_mask_0 = const()[name = string("normed_47_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_47_cast_fp16 = slice_by_index(begin = normed_47_begin_0, end = normed_47_end_0, end_mask = normed_47_end_mask_0, x = normed_45_cast_fp16)[name = string("normed_47_cast_fp16")]; + tensor const_131_promoted_to_fp16 = const()[name = string("const_131_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(319208512)))]; + tensor x_93_cast_fp16 = mul(x = normed_47_cast_fp16, y = const_131_promoted_to_fp16)[name = string("x_93_cast_fp16")]; + tensor var_3155 = const()[name = string("op_3155"), val = tensor([0, 2, 1])]; + tensor input_79_axes_0 = const()[name = string("input_79_axes_0"), val = tensor([2])]; + tensor var_3156 = transpose(perm = var_3155, x = x_93_cast_fp16)[name = string("transpose_19")]; + tensor input_79 = expand_dims(axes = input_79_axes_0, x = var_3156)[name = string("input_79")]; + string input_81_pad_type_0 = const()[name = string("input_81_pad_type_0"), val = string("valid")]; + tensor input_81_strides_0 = const()[name = string("input_81_strides_0"), val = tensor([1, 1])]; + tensor input_81_pad_0 = const()[name = string("input_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_81_dilations_0 = const()[name = string("input_81_dilations_0"), val = tensor([1, 1])]; + int32 input_81_groups_0 = const()[name = string("input_81_groups_0"), val = int32(1)]; + tensor input_81 = conv(dilations = input_81_dilations_0, groups = input_81_groups_0, pad = input_81_pad_0, pad_type = input_81_pad_type_0, strides = input_81_strides_0, weight = model_model_layers_15_mlp_gate_proj_weight_palettized, x = input_79)[name = string("input_81")]; + string b_11_pad_type_0 = const()[name = string("b_11_pad_type_0"), val = string("valid")]; + tensor b_11_strides_0 = const()[name = string("b_11_strides_0"), val = tensor([1, 1])]; + tensor b_11_pad_0 = const()[name = string("b_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_11_dilations_0 = const()[name = string("b_11_dilations_0"), val = tensor([1, 1])]; + int32 b_11_groups_0 = const()[name = string("b_11_groups_0"), val = int32(1)]; + tensor b_11 = conv(dilations = b_11_dilations_0, groups = b_11_groups_0, pad = b_11_pad_0, pad_type = b_11_pad_type_0, strides = b_11_strides_0, weight = model_model_layers_15_mlp_up_proj_weight_palettized, x = input_79)[name = string("b_11")]; + tensor c_11 = silu(x = input_81)[name = string("c_11")]; + tensor input_83 = mul(x = c_11, y = b_11)[name = string("input_83")]; + string e_11_pad_type_0 = const()[name = string("e_11_pad_type_0"), val = string("valid")]; + tensor e_11_strides_0 = const()[name = string("e_11_strides_0"), val = tensor([1, 1])]; + tensor e_11_pad_0 = const()[name = string("e_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_11_dilations_0 = const()[name = string("e_11_dilations_0"), val = tensor([1, 1])]; + int32 e_11_groups_0 = const()[name = string("e_11_groups_0"), val = int32(1)]; + tensor e_11 = conv(dilations = e_11_dilations_0, groups = e_11_groups_0, pad = e_11_pad_0, pad_type = e_11_pad_type_0, strides = e_11_strides_0, weight = model_model_layers_15_mlp_down_proj_weight_palettized, x = input_83)[name = string("e_11")]; + tensor var_3178_axes_0 = const()[name = string("op_3178_axes_0"), val = tensor([2])]; + tensor var_3178 = squeeze(axes = var_3178_axes_0, x = e_11)[name = string("op_3178")]; + tensor var_3179 = const()[name = string("op_3179"), val = tensor([0, 2, 1])]; + tensor var_3180 = transpose(perm = var_3179, x = var_3178)[name = string("transpose_18")]; + tensor hidden_states_37_cast_fp16 = add(x = hidden_states_35_cast_fp16, y = var_3180)[name = string("hidden_states_37_cast_fp16")]; + int32 var_3192 = const()[name = string("op_3192"), val = int32(-1)]; + fp16 const_132_promoted_to_fp16 = const()[name = string("const_132_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3194_cast_fp16 = mul(x = hidden_states_37_cast_fp16, y = const_132_promoted_to_fp16)[name = string("op_3194_cast_fp16")]; + bool input_85_interleave_0 = const()[name = string("input_85_interleave_0"), val = bool(false)]; + tensor input_85_cast_fp16 = concat(axis = var_3192, interleave = input_85_interleave_0, values = (hidden_states_37_cast_fp16, var_3194_cast_fp16))[name = string("input_85_cast_fp16")]; + tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; + fp16 var_3189_to_fp16 = const()[name = string("op_3189_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_3189_to_fp16, x = input_85_cast_fp16)[name = string("normed_49_cast_fp16")]; + tensor normed_51_begin_0 = const()[name = string("normed_51_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_51_end_0 = const()[name = string("normed_51_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_51_end_mask_0 = const()[name = string("normed_51_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_51_cast_fp16 = slice_by_index(begin = normed_51_begin_0, end = normed_51_end_0, end_mask = normed_51_end_mask_0, x = normed_49_cast_fp16)[name = string("normed_51_cast_fp16")]; + tensor const_135_promoted_to_fp16 = const()[name = string("const_135_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(319211648)))]; + tensor hidden_states_39_cast_fp16 = mul(x = normed_51_cast_fp16, y = const_135_promoted_to_fp16)[name = string("hidden_states_39_cast_fp16")]; + tensor var_3211 = const()[name = string("op_3211"), val = tensor([0, 2, 1])]; + tensor var_3214_axes_0 = const()[name = string("op_3214_axes_0"), val = tensor([2])]; + tensor var_3212_cast_fp16 = transpose(perm = var_3211, x = hidden_states_39_cast_fp16)[name = string("transpose_17")]; + tensor var_3214_cast_fp16 = expand_dims(axes = var_3214_axes_0, x = var_3212_cast_fp16)[name = string("op_3214_cast_fp16")]; + string var_3230_pad_type_0 = const()[name = string("op_3230_pad_type_0"), val = string("valid")]; + tensor var_3230_strides_0 = const()[name = string("op_3230_strides_0"), val = tensor([1, 1])]; + tensor var_3230_pad_0 = const()[name = string("op_3230_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3230_dilations_0 = const()[name = string("op_3230_dilations_0"), val = tensor([1, 1])]; + int32 var_3230_groups_0 = const()[name = string("op_3230_groups_0"), val = int32(1)]; + tensor var_3230 = conv(bias = model_model_layers_16_self_attn_q_proj_bias, dilations = var_3230_dilations_0, groups = var_3230_groups_0, pad = var_3230_pad_0, pad_type = var_3230_pad_type_0, strides = var_3230_strides_0, weight = model_model_layers_16_self_attn_q_proj_weight_palettized, x = var_3214_cast_fp16)[name = string("op_3230")]; + tensor var_3235 = const()[name = string("op_3235"), val = tensor([1, 12, 1, 128])]; + tensor var_3236 = reshape(shape = var_3235, x = var_3230)[name = string("op_3236")]; + string var_3252_pad_type_0 = const()[name = string("op_3252_pad_type_0"), val = string("valid")]; + tensor var_3252_strides_0 = const()[name = string("op_3252_strides_0"), val = tensor([1, 1])]; + tensor var_3252_pad_0 = const()[name = string("op_3252_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3252_dilations_0 = const()[name = string("op_3252_dilations_0"), val = tensor([1, 1])]; + int32 var_3252_groups_0 = const()[name = string("op_3252_groups_0"), val = int32(1)]; + tensor var_3252 = conv(bias = model_model_layers_16_self_attn_k_proj_bias, dilations = var_3252_dilations_0, groups = var_3252_groups_0, pad = var_3252_pad_0, pad_type = var_3252_pad_type_0, strides = var_3252_strides_0, weight = model_model_layers_16_self_attn_k_proj_weight_palettized, x = var_3214_cast_fp16)[name = string("op_3252")]; + tensor var_3257 = const()[name = string("op_3257"), val = tensor([1, 2, 1, 128])]; + tensor var_3258 = reshape(shape = var_3257, x = var_3252)[name = string("op_3258")]; + string var_3274_pad_type_0 = const()[name = string("op_3274_pad_type_0"), val = string("valid")]; + tensor var_3274_strides_0 = const()[name = string("op_3274_strides_0"), val = tensor([1, 1])]; + tensor var_3274_pad_0 = const()[name = string("op_3274_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3274_dilations_0 = const()[name = string("op_3274_dilations_0"), val = tensor([1, 1])]; + int32 var_3274_groups_0 = const()[name = string("op_3274_groups_0"), val = int32(1)]; + tensor var_3274 = conv(bias = model_model_layers_16_self_attn_v_proj_bias, dilations = var_3274_dilations_0, groups = var_3274_groups_0, pad = var_3274_pad_0, pad_type = var_3274_pad_type_0, strides = var_3274_strides_0, weight = model_model_layers_16_self_attn_v_proj_weight_palettized, x = var_3214_cast_fp16)[name = string("op_3274")]; + tensor var_3279 = const()[name = string("op_3279"), val = tensor([1, 2, 1, 128])]; + tensor var_3280 = reshape(shape = var_3279, x = var_3274)[name = string("op_3280")]; + tensor var_3286 = mul(x = var_3236, y = cos_1_cast_fp16)[name = string("op_3286")]; + tensor x1_25_begin_0 = const()[name = string("x1_25_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_25_end_0 = const()[name = string("x1_25_end_0"), val = tensor([1, 12, 1, 64])]; + tensor x1_25_end_mask_0 = const()[name = string("x1_25_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_25 = slice_by_index(begin = x1_25_begin_0, end = x1_25_end_0, end_mask = x1_25_end_mask_0, x = var_3236)[name = string("x1_25")]; + tensor x2_25_begin_0 = const()[name = string("x2_25_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_25_end_0 = const()[name = string("x2_25_end_0"), val = tensor([1, 12, 1, 128])]; + tensor x2_25_end_mask_0 = const()[name = string("x2_25_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_25 = slice_by_index(begin = x2_25_begin_0, end = x2_25_end_0, end_mask = x2_25_end_mask_0, x = var_3236)[name = string("x2_25")]; + fp16 const_138_promoted = const()[name = string("const_138_promoted"), val = fp16(-0x1p+0)]; + tensor var_3307 = mul(x = x2_25, y = const_138_promoted)[name = string("op_3307")]; + int32 var_3309 = const()[name = string("op_3309"), val = int32(-1)]; + bool var_3310_interleave_0 = const()[name = string("op_3310_interleave_0"), val = bool(false)]; + tensor var_3310 = concat(axis = var_3309, interleave = var_3310_interleave_0, values = (var_3307, x1_25))[name = string("op_3310")]; + tensor var_3311 = mul(x = var_3310, y = sin_1_cast_fp16)[name = string("op_3311")]; + tensor query_states_13 = add(x = var_3286, y = var_3311)[name = string("query_states_13")]; + tensor var_3314 = mul(x = var_3258, y = cos_1_cast_fp16)[name = string("op_3314")]; + tensor x1_27_begin_0 = const()[name = string("x1_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_27_end_0 = const()[name = string("x1_27_end_0"), val = tensor([1, 2, 1, 64])]; + tensor x1_27_end_mask_0 = const()[name = string("x1_27_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_27 = slice_by_index(begin = x1_27_begin_0, end = x1_27_end_0, end_mask = x1_27_end_mask_0, x = var_3258)[name = string("x1_27")]; + tensor x2_27_begin_0 = const()[name = string("x2_27_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_27_end_0 = const()[name = string("x2_27_end_0"), val = tensor([1, 2, 1, 128])]; + tensor x2_27_end_mask_0 = const()[name = string("x2_27_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_27 = slice_by_index(begin = x2_27_begin_0, end = x2_27_end_0, end_mask = x2_27_end_mask_0, x = var_3258)[name = string("x2_27")]; + fp16 const_141_promoted = const()[name = string("const_141_promoted"), val = fp16(-0x1p+0)]; + tensor var_3335 = mul(x = x2_27, y = const_141_promoted)[name = string("op_3335")]; + int32 var_3337 = const()[name = string("op_3337"), val = int32(-1)]; + bool var_3338_interleave_0 = const()[name = string("op_3338_interleave_0"), val = bool(false)]; + tensor var_3338 = concat(axis = var_3337, interleave = var_3338_interleave_0, values = (var_3335, x1_27))[name = string("op_3338")]; + tensor var_3339 = mul(x = var_3338, y = sin_1_cast_fp16)[name = string("op_3339")]; + tensor key_states_25 = add(x = var_3314, y = var_3339)[name = string("key_states_25")]; + tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([16])]; + tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; + tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; + tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([17])]; + int32 concat_50_axis_0 = const()[name = string("concat_50_axis_0"), val = int32(0)]; + bool concat_50_interleave_0 = const()[name = string("concat_50_interleave_0"), val = bool(false)]; + tensor concat_50 = concat(axis = concat_50_axis_0, interleave = concat_50_interleave_0, values = (expand_dims_72, expand_dims_73, current_pos, expand_dims_75))[name = string("concat_50")]; + tensor concat_51_values1_0 = const()[name = string("concat_51_values1_0"), val = tensor([0])]; + tensor concat_51_values3_0 = const()[name = string("concat_51_values3_0"), val = tensor([0])]; + int32 concat_51_axis_0 = const()[name = string("concat_51_axis_0"), val = int32(0)]; + bool concat_51_interleave_0 = const()[name = string("concat_51_interleave_0"), val = bool(false)]; + tensor concat_51 = concat(axis = concat_51_axis_0, interleave = concat_51_interleave_0, values = (expand_dims_76, concat_51_values1_0, var_578, concat_51_values3_0))[name = string("concat_51")]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_50, begin_mask = model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0, end = concat_51, end_mask = model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_13_stride_0, update = key_states_25, x = coreml_update_state_29)[name = string("model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_12_write_state")]; + tensor coreml_update_state_30 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_12")]; + tensor expand_dims_78 = const()[name = string("expand_dims_78"), val = tensor([44])]; + tensor expand_dims_79 = const()[name = string("expand_dims_79"), val = tensor([0])]; + tensor expand_dims_81 = const()[name = string("expand_dims_81"), val = tensor([0])]; + tensor expand_dims_82 = const()[name = string("expand_dims_82"), val = tensor([45])]; + int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; + bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; + tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_78, expand_dims_79, current_pos, expand_dims_81))[name = string("concat_54")]; + tensor concat_55_values1_0 = const()[name = string("concat_55_values1_0"), val = tensor([0])]; + tensor concat_55_values3_0 = const()[name = string("concat_55_values3_0"), val = tensor([0])]; + int32 concat_55_axis_0 = const()[name = string("concat_55_axis_0"), val = int32(0)]; + bool concat_55_interleave_0 = const()[name = string("concat_55_interleave_0"), val = bool(false)]; + tensor concat_55 = concat(axis = concat_55_axis_0, interleave = concat_55_interleave_0, values = (expand_dims_82, concat_55_values1_0, var_578, concat_55_values3_0))[name = string("concat_55")]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_54, begin_mask = model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0, end = concat_55, end_mask = model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_14_stride_0, update = var_3280, x = coreml_update_state_30)[name = string("model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_13_write_state")]; + tensor coreml_update_state_31 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_13")]; + tensor var_3394_begin_0 = const()[name = string("op_3394_begin_0"), val = tensor([16, 0, 0, 0])]; + tensor var_3394_end_0 = const()[name = string("op_3394_end_0"), val = tensor([17, 2, 2048, 128])]; + tensor var_3394_end_mask_0 = const()[name = string("op_3394_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3394_cast_fp16 = slice_by_index(begin = var_3394_begin_0, end = var_3394_end_0, end_mask = var_3394_end_mask_0, x = coreml_update_state_31)[name = string("op_3394_cast_fp16")]; + tensor K_layer_cache_13_axes_0 = const()[name = string("K_layer_cache_13_axes_0"), val = tensor([0])]; + tensor K_layer_cache_13_cast_fp16 = squeeze(axes = K_layer_cache_13_axes_0, x = var_3394_cast_fp16)[name = string("K_layer_cache_13_cast_fp16")]; + tensor var_3401_begin_0 = const()[name = string("op_3401_begin_0"), val = tensor([44, 0, 0, 0])]; + tensor var_3401_end_0 = const()[name = string("op_3401_end_0"), val = tensor([45, 2, 2048, 128])]; + tensor var_3401_end_mask_0 = const()[name = string("op_3401_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3401_cast_fp16 = slice_by_index(begin = var_3401_begin_0, end = var_3401_end_0, end_mask = var_3401_end_mask_0, x = coreml_update_state_31)[name = string("op_3401_cast_fp16")]; + tensor V_layer_cache_13_axes_0 = const()[name = string("V_layer_cache_13_axes_0"), val = tensor([0])]; + tensor V_layer_cache_13_cast_fp16 = squeeze(axes = V_layer_cache_13_axes_0, x = var_3401_cast_fp16)[name = string("V_layer_cache_13_cast_fp16")]; + tensor x_99_axes_0 = const()[name = string("x_99_axes_0"), val = tensor([1])]; + tensor x_99_cast_fp16 = expand_dims(axes = x_99_axes_0, x = K_layer_cache_13_cast_fp16)[name = string("x_99_cast_fp16")]; + tensor var_3438 = const()[name = string("op_3438"), val = tensor([1, 6, 1, 1])]; + tensor x_101_cast_fp16 = tile(reps = var_3438, x = x_99_cast_fp16)[name = string("x_101_cast_fp16")]; + tensor var_3450 = const()[name = string("op_3450"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_27_cast_fp16 = reshape(shape = var_3450, x = x_101_cast_fp16)[name = string("key_states_27_cast_fp16")]; + tensor x_105_axes_0 = const()[name = string("x_105_axes_0"), val = tensor([1])]; + tensor x_105_cast_fp16 = expand_dims(axes = x_105_axes_0, x = V_layer_cache_13_cast_fp16)[name = string("x_105_cast_fp16")]; + tensor var_3458 = const()[name = string("op_3458"), val = tensor([1, 6, 1, 1])]; + tensor x_107_cast_fp16 = tile(reps = var_3458, x = x_105_cast_fp16)[name = string("x_107_cast_fp16")]; + tensor var_3470 = const()[name = string("op_3470"), val = tensor([1, -1, 2048, 128])]; + tensor value_states_27_cast_fp16 = reshape(shape = var_3470, x = x_107_cast_fp16)[name = string("value_states_27_cast_fp16")]; + bool var_3493_transpose_x_1 = const()[name = string("op_3493_transpose_x_1"), val = bool(false)]; + bool var_3493_transpose_y_1 = const()[name = string("op_3493_transpose_y_1"), val = bool(true)]; + tensor var_3493_cast_fp16 = matmul(transpose_x = var_3493_transpose_x_1, transpose_y = var_3493_transpose_y_1, x = query_states_13, y = key_states_27_cast_fp16)[name = string("op_3493_cast_fp16")]; + fp16 var_3494_to_fp16 = const()[name = string("op_3494_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_25_cast_fp16 = mul(x = var_3493_cast_fp16, y = var_3494_to_fp16)[name = string("attn_logits_25_cast_fp16")]; + tensor attn_logits_27_cast_fp16 = add(x = attn_logits_25_cast_fp16, y = causal_mask)[name = string("attn_logits_27_cast_fp16")]; + int32 var_3521 = const()[name = string("op_3521"), val = int32(-1)]; + tensor var_3523_cast_fp16 = softmax(axis = var_3521, x = attn_logits_27_cast_fp16)[name = string("op_3523_cast_fp16")]; + bool attn_output_73_transpose_x_0 = const()[name = string("attn_output_73_transpose_x_0"), val = bool(false)]; + bool attn_output_73_transpose_y_0 = const()[name = string("attn_output_73_transpose_y_0"), val = bool(false)]; + tensor attn_output_73_cast_fp16 = matmul(transpose_x = attn_output_73_transpose_x_0, transpose_y = attn_output_73_transpose_y_0, x = var_3523_cast_fp16, y = value_states_27_cast_fp16)[name = string("attn_output_73_cast_fp16")]; + tensor var_3547_perm_0 = const()[name = string("op_3547_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_3551 = const()[name = string("op_3551"), val = tensor([1, 1, 1536])]; + tensor var_3547 = transpose(perm = var_3547_perm_0, x = attn_output_73_cast_fp16)[name = string("transpose_16")]; + tensor attn_output_79 = reshape(shape = var_3551, x = var_3547)[name = string("attn_output_79")]; + tensor var_3556 = const()[name = string("op_3556"), val = tensor([0, 2, 1])]; + tensor squeeze_6_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(319214784))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(320984320))))[name = string("squeeze_6_palettized")]; + string var_3572_pad_type_0 = const()[name = string("op_3572_pad_type_0"), val = string("valid")]; + int32 var_3572_groups_0 = const()[name = string("op_3572_groups_0"), val = int32(1)]; + tensor var_3572_strides_0 = const()[name = string("op_3572_strides_0"), val = tensor([1])]; + tensor var_3572_pad_0 = const()[name = string("op_3572_pad_0"), val = tensor([0, 0])]; + tensor var_3572_dilations_0 = const()[name = string("op_3572_dilations_0"), val = tensor([1])]; + tensor var_3557 = transpose(perm = var_3556, x = attn_output_79)[name = string("transpose_15")]; + tensor var_3572 = conv(dilations = var_3572_dilations_0, groups = var_3572_groups_0, pad = var_3572_pad_0, pad_type = var_3572_pad_type_0, strides = var_3572_strides_0, weight = squeeze_6_palettized, x = var_3557)[name = string("op_3572")]; + tensor var_3576 = const()[name = string("op_3576"), val = tensor([0, 2, 1])]; + tensor attn_output_83 = transpose(perm = var_3576, x = var_3572)[name = string("transpose_14")]; + tensor hidden_states_41_cast_fp16 = add(x = hidden_states_37_cast_fp16, y = attn_output_83)[name = string("hidden_states_41_cast_fp16")]; + int32 var_3589 = const()[name = string("op_3589"), val = int32(-1)]; + fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3591_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_3591_cast_fp16")]; + bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; + tensor input_91_cast_fp16 = concat(axis = var_3589, interleave = input_91_interleave_0, values = (hidden_states_41_cast_fp16, var_3591_cast_fp16))[name = string("input_91_cast_fp16")]; + tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; + fp16 var_3586_to_fp16 = const()[name = string("op_3586_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_3586_to_fp16, x = input_91_cast_fp16)[name = string("normed_53_cast_fp16")]; + tensor normed_55_begin_0 = const()[name = string("normed_55_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_55_end_0 = const()[name = string("normed_55_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_55_end_mask_0 = const()[name = string("normed_55_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_55_cast_fp16 = slice_by_index(begin = normed_55_begin_0, end = normed_55_end_0, end_mask = normed_55_end_mask_0, x = normed_53_cast_fp16)[name = string("normed_55_cast_fp16")]; + tensor const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321033536)))]; + tensor x_109_cast_fp16 = mul(x = normed_55_cast_fp16, y = const_153_promoted_to_fp16)[name = string("x_109_cast_fp16")]; + tensor var_3616 = const()[name = string("op_3616"), val = tensor([0, 2, 1])]; + tensor input_93_axes_0 = const()[name = string("input_93_axes_0"), val = tensor([2])]; + tensor var_3617 = transpose(perm = var_3616, x = x_109_cast_fp16)[name = string("transpose_13")]; + tensor input_93 = expand_dims(axes = input_93_axes_0, x = var_3617)[name = string("input_93")]; + string input_95_pad_type_0 = const()[name = string("input_95_pad_type_0"), val = string("valid")]; + tensor input_95_strides_0 = const()[name = string("input_95_strides_0"), val = tensor([1, 1])]; + tensor input_95_pad_0 = const()[name = string("input_95_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_95_dilations_0 = const()[name = string("input_95_dilations_0"), val = tensor([1, 1])]; + int32 input_95_groups_0 = const()[name = string("input_95_groups_0"), val = int32(1)]; + tensor input_95 = conv(dilations = input_95_dilations_0, groups = input_95_groups_0, pad = input_95_pad_0, pad_type = input_95_pad_type_0, strides = input_95_strides_0, weight = model_model_layers_16_mlp_gate_proj_weight_palettized, x = input_93)[name = string("input_95")]; + string b_13_pad_type_0 = const()[name = string("b_13_pad_type_0"), val = string("valid")]; + tensor b_13_strides_0 = const()[name = string("b_13_strides_0"), val = tensor([1, 1])]; + tensor b_13_pad_0 = const()[name = string("b_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_13_dilations_0 = const()[name = string("b_13_dilations_0"), val = tensor([1, 1])]; + int32 b_13_groups_0 = const()[name = string("b_13_groups_0"), val = int32(1)]; + tensor b_13 = conv(dilations = b_13_dilations_0, groups = b_13_groups_0, pad = b_13_pad_0, pad_type = b_13_pad_type_0, strides = b_13_strides_0, weight = model_model_layers_16_mlp_up_proj_weight_palettized, x = input_93)[name = string("b_13")]; + tensor c_13 = silu(x = input_95)[name = string("c_13")]; + tensor input_97 = mul(x = c_13, y = b_13)[name = string("input_97")]; + string e_13_pad_type_0 = const()[name = string("e_13_pad_type_0"), val = string("valid")]; + tensor e_13_strides_0 = const()[name = string("e_13_strides_0"), val = tensor([1, 1])]; + tensor e_13_pad_0 = const()[name = string("e_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_13_dilations_0 = const()[name = string("e_13_dilations_0"), val = tensor([1, 1])]; + int32 e_13_groups_0 = const()[name = string("e_13_groups_0"), val = int32(1)]; + tensor e_13 = conv(dilations = e_13_dilations_0, groups = e_13_groups_0, pad = e_13_pad_0, pad_type = e_13_pad_type_0, strides = e_13_strides_0, weight = model_model_layers_16_mlp_down_proj_weight_palettized, x = input_97)[name = string("e_13")]; + tensor var_3639_axes_0 = const()[name = string("op_3639_axes_0"), val = tensor([2])]; + tensor var_3639 = squeeze(axes = var_3639_axes_0, x = e_13)[name = string("op_3639")]; + tensor var_3640 = const()[name = string("op_3640"), val = tensor([0, 2, 1])]; + tensor var_3641 = transpose(perm = var_3640, x = var_3639)[name = string("transpose_12")]; + tensor hidden_states_43_cast_fp16 = add(x = hidden_states_41_cast_fp16, y = var_3641)[name = string("hidden_states_43_cast_fp16")]; + int32 var_3653 = const()[name = string("op_3653"), val = int32(-1)]; + fp16 const_154_promoted_to_fp16 = const()[name = string("const_154_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3655_cast_fp16 = mul(x = hidden_states_43_cast_fp16, y = const_154_promoted_to_fp16)[name = string("op_3655_cast_fp16")]; + bool input_99_interleave_0 = const()[name = string("input_99_interleave_0"), val = bool(false)]; + tensor input_99_cast_fp16 = concat(axis = var_3653, interleave = input_99_interleave_0, values = (hidden_states_43_cast_fp16, var_3655_cast_fp16))[name = string("input_99_cast_fp16")]; + tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; + fp16 var_3650_to_fp16 = const()[name = string("op_3650_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_3650_to_fp16, x = input_99_cast_fp16)[name = string("normed_57_cast_fp16")]; + tensor normed_59_begin_0 = const()[name = string("normed_59_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_59_end_0 = const()[name = string("normed_59_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_59_end_mask_0 = const()[name = string("normed_59_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_59_cast_fp16 = slice_by_index(begin = normed_59_begin_0, end = normed_59_end_0, end_mask = normed_59_end_mask_0, x = normed_57_cast_fp16)[name = string("normed_59_cast_fp16")]; + tensor const_157_promoted_to_fp16 = const()[name = string("const_157_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321036672)))]; + tensor hidden_states_45_cast_fp16 = mul(x = normed_59_cast_fp16, y = const_157_promoted_to_fp16)[name = string("hidden_states_45_cast_fp16")]; + tensor var_3672 = const()[name = string("op_3672"), val = tensor([0, 2, 1])]; + tensor var_3675_axes_0 = const()[name = string("op_3675_axes_0"), val = tensor([2])]; + tensor var_3673_cast_fp16 = transpose(perm = var_3672, x = hidden_states_45_cast_fp16)[name = string("transpose_11")]; + tensor var_3675_cast_fp16 = expand_dims(axes = var_3675_axes_0, x = var_3673_cast_fp16)[name = string("op_3675_cast_fp16")]; + string var_3691_pad_type_0 = const()[name = string("op_3691_pad_type_0"), val = string("valid")]; + tensor var_3691_strides_0 = const()[name = string("op_3691_strides_0"), val = tensor([1, 1])]; + tensor var_3691_pad_0 = const()[name = string("op_3691_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3691_dilations_0 = const()[name = string("op_3691_dilations_0"), val = tensor([1, 1])]; + int32 var_3691_groups_0 = const()[name = string("op_3691_groups_0"), val = int32(1)]; + tensor var_3691 = conv(bias = model_model_layers_17_self_attn_q_proj_bias, dilations = var_3691_dilations_0, groups = var_3691_groups_0, pad = var_3691_pad_0, pad_type = var_3691_pad_type_0, strides = var_3691_strides_0, weight = model_model_layers_17_self_attn_q_proj_weight_palettized, x = var_3675_cast_fp16)[name = string("op_3691")]; + tensor var_3696 = const()[name = string("op_3696"), val = tensor([1, 12, 1, 128])]; + tensor var_3697 = reshape(shape = var_3696, x = var_3691)[name = string("op_3697")]; + string var_3713_pad_type_0 = const()[name = string("op_3713_pad_type_0"), val = string("valid")]; + tensor var_3713_strides_0 = const()[name = string("op_3713_strides_0"), val = tensor([1, 1])]; + tensor var_3713_pad_0 = const()[name = string("op_3713_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3713_dilations_0 = const()[name = string("op_3713_dilations_0"), val = tensor([1, 1])]; + int32 var_3713_groups_0 = const()[name = string("op_3713_groups_0"), val = int32(1)]; + tensor var_3713 = conv(bias = model_model_layers_17_self_attn_k_proj_bias, dilations = var_3713_dilations_0, groups = var_3713_groups_0, pad = var_3713_pad_0, pad_type = var_3713_pad_type_0, strides = var_3713_strides_0, weight = model_model_layers_17_self_attn_k_proj_weight_palettized, x = var_3675_cast_fp16)[name = string("op_3713")]; + tensor var_3718 = const()[name = string("op_3718"), val = tensor([1, 2, 1, 128])]; + tensor var_3719 = reshape(shape = var_3718, x = var_3713)[name = string("op_3719")]; + string var_3735_pad_type_0 = const()[name = string("op_3735_pad_type_0"), val = string("valid")]; + tensor var_3735_strides_0 = const()[name = string("op_3735_strides_0"), val = tensor([1, 1])]; + tensor var_3735_pad_0 = const()[name = string("op_3735_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3735_dilations_0 = const()[name = string("op_3735_dilations_0"), val = tensor([1, 1])]; + int32 var_3735_groups_0 = const()[name = string("op_3735_groups_0"), val = int32(1)]; + tensor var_3735 = conv(bias = model_model_layers_17_self_attn_v_proj_bias, dilations = var_3735_dilations_0, groups = var_3735_groups_0, pad = var_3735_pad_0, pad_type = var_3735_pad_type_0, strides = var_3735_strides_0, weight = model_model_layers_17_self_attn_v_proj_weight_palettized, x = var_3675_cast_fp16)[name = string("op_3735")]; + tensor var_3740 = const()[name = string("op_3740"), val = tensor([1, 2, 1, 128])]; + tensor var_3741 = reshape(shape = var_3740, x = var_3735)[name = string("op_3741")]; + tensor var_3747 = mul(x = var_3697, y = cos_1_cast_fp16)[name = string("op_3747")]; + tensor x1_29_begin_0 = const()[name = string("x1_29_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_29_end_0 = const()[name = string("x1_29_end_0"), val = tensor([1, 12, 1, 64])]; + tensor x1_29_end_mask_0 = const()[name = string("x1_29_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_29 = slice_by_index(begin = x1_29_begin_0, end = x1_29_end_0, end_mask = x1_29_end_mask_0, x = var_3697)[name = string("x1_29")]; + tensor x2_29_begin_0 = const()[name = string("x2_29_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_29_end_0 = const()[name = string("x2_29_end_0"), val = tensor([1, 12, 1, 128])]; + tensor x2_29_end_mask_0 = const()[name = string("x2_29_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_29 = slice_by_index(begin = x2_29_begin_0, end = x2_29_end_0, end_mask = x2_29_end_mask_0, x = var_3697)[name = string("x2_29")]; + fp16 const_160_promoted = const()[name = string("const_160_promoted"), val = fp16(-0x1p+0)]; + tensor var_3768 = mul(x = x2_29, y = const_160_promoted)[name = string("op_3768")]; + int32 var_3770 = const()[name = string("op_3770"), val = int32(-1)]; + bool var_3771_interleave_0 = const()[name = string("op_3771_interleave_0"), val = bool(false)]; + tensor var_3771 = concat(axis = var_3770, interleave = var_3771_interleave_0, values = (var_3768, x1_29))[name = string("op_3771")]; + tensor var_3772 = mul(x = var_3771, y = sin_1_cast_fp16)[name = string("op_3772")]; + tensor query_states_15 = add(x = var_3747, y = var_3772)[name = string("query_states_15")]; + tensor var_3775 = mul(x = var_3719, y = cos_1_cast_fp16)[name = string("op_3775")]; + tensor x1_31_begin_0 = const()[name = string("x1_31_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_31_end_0 = const()[name = string("x1_31_end_0"), val = tensor([1, 2, 1, 64])]; + tensor x1_31_end_mask_0 = const()[name = string("x1_31_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_31 = slice_by_index(begin = x1_31_begin_0, end = x1_31_end_0, end_mask = x1_31_end_mask_0, x = var_3719)[name = string("x1_31")]; + tensor x2_31_begin_0 = const()[name = string("x2_31_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_31_end_0 = const()[name = string("x2_31_end_0"), val = tensor([1, 2, 1, 128])]; + tensor x2_31_end_mask_0 = const()[name = string("x2_31_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_31 = slice_by_index(begin = x2_31_begin_0, end = x2_31_end_0, end_mask = x2_31_end_mask_0, x = var_3719)[name = string("x2_31")]; + fp16 const_163_promoted = const()[name = string("const_163_promoted"), val = fp16(-0x1p+0)]; + tensor var_3796 = mul(x = x2_31, y = const_163_promoted)[name = string("op_3796")]; + int32 var_3798 = const()[name = string("op_3798"), val = int32(-1)]; + bool var_3799_interleave_0 = const()[name = string("op_3799_interleave_0"), val = bool(false)]; + tensor var_3799 = concat(axis = var_3798, interleave = var_3799_interleave_0, values = (var_3796, x1_31))[name = string("op_3799")]; + tensor var_3800 = mul(x = var_3799, y = sin_1_cast_fp16)[name = string("op_3800")]; + tensor key_states_29 = add(x = var_3775, y = var_3800)[name = string("key_states_29")]; + tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([17])]; + tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; + tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; + tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([18])]; + int32 concat_58_axis_0 = const()[name = string("concat_58_axis_0"), val = int32(0)]; + bool concat_58_interleave_0 = const()[name = string("concat_58_interleave_0"), val = bool(false)]; + tensor concat_58 = concat(axis = concat_58_axis_0, interleave = concat_58_interleave_0, values = (expand_dims_84, expand_dims_85, current_pos, expand_dims_87))[name = string("concat_58")]; + tensor concat_59_values1_0 = const()[name = string("concat_59_values1_0"), val = tensor([0])]; + tensor concat_59_values3_0 = const()[name = string("concat_59_values3_0"), val = tensor([0])]; + int32 concat_59_axis_0 = const()[name = string("concat_59_axis_0"), val = int32(0)]; + bool concat_59_interleave_0 = const()[name = string("concat_59_interleave_0"), val = bool(false)]; + tensor concat_59 = concat(axis = concat_59_axis_0, interleave = concat_59_interleave_0, values = (expand_dims_88, concat_59_values1_0, var_578, concat_59_values3_0))[name = string("concat_59")]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_58, begin_mask = model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0, end = concat_59, end_mask = model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_15_stride_0, update = key_states_29, x = coreml_update_state_31)[name = string("model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_14_write_state")]; + tensor coreml_update_state_32 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_14")]; + tensor expand_dims_90 = const()[name = string("expand_dims_90"), val = tensor([45])]; + tensor expand_dims_91 = const()[name = string("expand_dims_91"), val = tensor([0])]; + tensor expand_dims_93 = const()[name = string("expand_dims_93"), val = tensor([0])]; + tensor expand_dims_94 = const()[name = string("expand_dims_94"), val = tensor([46])]; + int32 concat_62_axis_0 = const()[name = string("concat_62_axis_0"), val = int32(0)]; + bool concat_62_interleave_0 = const()[name = string("concat_62_interleave_0"), val = bool(false)]; + tensor concat_62 = concat(axis = concat_62_axis_0, interleave = concat_62_interleave_0, values = (expand_dims_90, expand_dims_91, current_pos, expand_dims_93))[name = string("concat_62")]; + tensor concat_63_values1_0 = const()[name = string("concat_63_values1_0"), val = tensor([0])]; + tensor concat_63_values3_0 = const()[name = string("concat_63_values3_0"), val = tensor([0])]; + int32 concat_63_axis_0 = const()[name = string("concat_63_axis_0"), val = int32(0)]; + bool concat_63_interleave_0 = const()[name = string("concat_63_interleave_0"), val = bool(false)]; + tensor concat_63 = concat(axis = concat_63_axis_0, interleave = concat_63_interleave_0, values = (expand_dims_94, concat_63_values1_0, var_578, concat_63_values3_0))[name = string("concat_63")]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_62, begin_mask = model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0, end = concat_63, end_mask = model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_16_stride_0, update = var_3741, x = coreml_update_state_32)[name = string("model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_15_write_state")]; + tensor coreml_update_state_33 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_15")]; + tensor var_3855_begin_0 = const()[name = string("op_3855_begin_0"), val = tensor([17, 0, 0, 0])]; + tensor var_3855_end_0 = const()[name = string("op_3855_end_0"), val = tensor([18, 2, 2048, 128])]; + tensor var_3855_end_mask_0 = const()[name = string("op_3855_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3855_cast_fp16 = slice_by_index(begin = var_3855_begin_0, end = var_3855_end_0, end_mask = var_3855_end_mask_0, x = coreml_update_state_33)[name = string("op_3855_cast_fp16")]; + tensor K_layer_cache_15_axes_0 = const()[name = string("K_layer_cache_15_axes_0"), val = tensor([0])]; + tensor K_layer_cache_15_cast_fp16 = squeeze(axes = K_layer_cache_15_axes_0, x = var_3855_cast_fp16)[name = string("K_layer_cache_15_cast_fp16")]; + tensor var_3862_begin_0 = const()[name = string("op_3862_begin_0"), val = tensor([45, 0, 0, 0])]; + tensor var_3862_end_0 = const()[name = string("op_3862_end_0"), val = tensor([46, 2, 2048, 128])]; + tensor var_3862_end_mask_0 = const()[name = string("op_3862_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3862_cast_fp16 = slice_by_index(begin = var_3862_begin_0, end = var_3862_end_0, end_mask = var_3862_end_mask_0, x = coreml_update_state_33)[name = string("op_3862_cast_fp16")]; + tensor V_layer_cache_15_axes_0 = const()[name = string("V_layer_cache_15_axes_0"), val = tensor([0])]; + tensor V_layer_cache_15_cast_fp16 = squeeze(axes = V_layer_cache_15_axes_0, x = var_3862_cast_fp16)[name = string("V_layer_cache_15_cast_fp16")]; + tensor x_115_axes_0 = const()[name = string("x_115_axes_0"), val = tensor([1])]; + tensor x_115_cast_fp16 = expand_dims(axes = x_115_axes_0, x = K_layer_cache_15_cast_fp16)[name = string("x_115_cast_fp16")]; + tensor var_3899 = const()[name = string("op_3899"), val = tensor([1, 6, 1, 1])]; + tensor x_117_cast_fp16 = tile(reps = var_3899, x = x_115_cast_fp16)[name = string("x_117_cast_fp16")]; + tensor var_3911 = const()[name = string("op_3911"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_31_cast_fp16 = reshape(shape = var_3911, x = x_117_cast_fp16)[name = string("key_states_31_cast_fp16")]; + tensor x_121_axes_0 = const()[name = string("x_121_axes_0"), val = tensor([1])]; + tensor x_121_cast_fp16 = expand_dims(axes = x_121_axes_0, x = V_layer_cache_15_cast_fp16)[name = string("x_121_cast_fp16")]; + tensor var_3919 = const()[name = string("op_3919"), val = tensor([1, 6, 1, 1])]; + tensor x_123_cast_fp16 = tile(reps = var_3919, x = x_121_cast_fp16)[name = string("x_123_cast_fp16")]; + tensor var_3931 = const()[name = string("op_3931"), val = tensor([1, -1, 2048, 128])]; + tensor value_states_31_cast_fp16 = reshape(shape = var_3931, x = x_123_cast_fp16)[name = string("value_states_31_cast_fp16")]; + bool var_3954_transpose_x_1 = const()[name = string("op_3954_transpose_x_1"), val = bool(false)]; + bool var_3954_transpose_y_1 = const()[name = string("op_3954_transpose_y_1"), val = bool(true)]; + tensor var_3954_cast_fp16 = matmul(transpose_x = var_3954_transpose_x_1, transpose_y = var_3954_transpose_y_1, x = query_states_15, y = key_states_31_cast_fp16)[name = string("op_3954_cast_fp16")]; + fp16 var_3955_to_fp16 = const()[name = string("op_3955_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_29_cast_fp16 = mul(x = var_3954_cast_fp16, y = var_3955_to_fp16)[name = string("attn_logits_29_cast_fp16")]; + tensor attn_logits_31_cast_fp16 = add(x = attn_logits_29_cast_fp16, y = causal_mask)[name = string("attn_logits_31_cast_fp16")]; + int32 var_3982 = const()[name = string("op_3982"), val = int32(-1)]; + tensor var_3984_cast_fp16 = softmax(axis = var_3982, x = attn_logits_31_cast_fp16)[name = string("op_3984_cast_fp16")]; + bool attn_output_85_transpose_x_0 = const()[name = string("attn_output_85_transpose_x_0"), val = bool(false)]; + bool attn_output_85_transpose_y_0 = const()[name = string("attn_output_85_transpose_y_0"), val = bool(false)]; + tensor attn_output_85_cast_fp16 = matmul(transpose_x = attn_output_85_transpose_x_0, transpose_y = attn_output_85_transpose_y_0, x = var_3984_cast_fp16, y = value_states_31_cast_fp16)[name = string("attn_output_85_cast_fp16")]; + tensor var_4008_perm_0 = const()[name = string("op_4008_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_4012 = const()[name = string("op_4012"), val = tensor([1, 1, 1536])]; + tensor var_4008 = transpose(perm = var_4008_perm_0, x = attn_output_85_cast_fp16)[name = string("transpose_10")]; + tensor attn_output_91 = reshape(shape = var_4012, x = var_4008)[name = string("attn_output_91")]; + tensor var_4017 = const()[name = string("op_4017"), val = tensor([0, 2, 1])]; + tensor squeeze_7_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321039808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(322809344))))[name = string("squeeze_7_palettized")]; + string var_4033_pad_type_0 = const()[name = string("op_4033_pad_type_0"), val = string("valid")]; + int32 var_4033_groups_0 = const()[name = string("op_4033_groups_0"), val = int32(1)]; + tensor var_4033_strides_0 = const()[name = string("op_4033_strides_0"), val = tensor([1])]; + tensor var_4033_pad_0 = const()[name = string("op_4033_pad_0"), val = tensor([0, 0])]; + tensor var_4033_dilations_0 = const()[name = string("op_4033_dilations_0"), val = tensor([1])]; + tensor var_4018 = transpose(perm = var_4017, x = attn_output_91)[name = string("transpose_9")]; + tensor var_4033 = conv(dilations = var_4033_dilations_0, groups = var_4033_groups_0, pad = var_4033_pad_0, pad_type = var_4033_pad_type_0, strides = var_4033_strides_0, weight = squeeze_7_palettized, x = var_4018)[name = string("op_4033")]; + tensor var_4037 = const()[name = string("op_4037"), val = tensor([0, 2, 1])]; + tensor attn_output_95 = transpose(perm = var_4037, x = var_4033)[name = string("transpose_8")]; + tensor hidden_states_47_cast_fp16 = add(x = hidden_states_43_cast_fp16, y = attn_output_95)[name = string("hidden_states_47_cast_fp16")]; + int32 var_4050 = const()[name = string("op_4050"), val = int32(-1)]; + fp16 const_172_promoted_to_fp16 = const()[name = string("const_172_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4052_cast_fp16 = mul(x = hidden_states_47_cast_fp16, y = const_172_promoted_to_fp16)[name = string("op_4052_cast_fp16")]; + bool input_105_interleave_0 = const()[name = string("input_105_interleave_0"), val = bool(false)]; + tensor input_105_cast_fp16 = concat(axis = var_4050, interleave = input_105_interleave_0, values = (hidden_states_47_cast_fp16, var_4052_cast_fp16))[name = string("input_105_cast_fp16")]; + tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; + fp16 var_4047_to_fp16 = const()[name = string("op_4047_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_4047_to_fp16, x = input_105_cast_fp16)[name = string("normed_61_cast_fp16")]; + tensor normed_63_begin_0 = const()[name = string("normed_63_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_63_end_0 = const()[name = string("normed_63_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_63_end_mask_0 = const()[name = string("normed_63_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_63_cast_fp16 = slice_by_index(begin = normed_63_begin_0, end = normed_63_end_0, end_mask = normed_63_end_mask_0, x = normed_61_cast_fp16)[name = string("normed_63_cast_fp16")]; + tensor const_175_promoted_to_fp16 = const()[name = string("const_175_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(322858560)))]; + tensor x_125_cast_fp16 = mul(x = normed_63_cast_fp16, y = const_175_promoted_to_fp16)[name = string("x_125_cast_fp16")]; + tensor var_4077 = const()[name = string("op_4077"), val = tensor([0, 2, 1])]; + tensor input_107_axes_0 = const()[name = string("input_107_axes_0"), val = tensor([2])]; + tensor var_4078 = transpose(perm = var_4077, x = x_125_cast_fp16)[name = string("transpose_7")]; + tensor input_107 = expand_dims(axes = input_107_axes_0, x = var_4078)[name = string("input_107")]; + string input_109_pad_type_0 = const()[name = string("input_109_pad_type_0"), val = string("valid")]; + tensor input_109_strides_0 = const()[name = string("input_109_strides_0"), val = tensor([1, 1])]; + tensor input_109_pad_0 = const()[name = string("input_109_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_109_dilations_0 = const()[name = string("input_109_dilations_0"), val = tensor([1, 1])]; + int32 input_109_groups_0 = const()[name = string("input_109_groups_0"), val = int32(1)]; + tensor input_109 = conv(dilations = input_109_dilations_0, groups = input_109_groups_0, pad = input_109_pad_0, pad_type = input_109_pad_type_0, strides = input_109_strides_0, weight = model_model_layers_17_mlp_gate_proj_weight_palettized, x = input_107)[name = string("input_109")]; + string b_15_pad_type_0 = const()[name = string("b_15_pad_type_0"), val = string("valid")]; + tensor b_15_strides_0 = const()[name = string("b_15_strides_0"), val = tensor([1, 1])]; + tensor b_15_pad_0 = const()[name = string("b_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_15_dilations_0 = const()[name = string("b_15_dilations_0"), val = tensor([1, 1])]; + int32 b_15_groups_0 = const()[name = string("b_15_groups_0"), val = int32(1)]; + tensor b_15 = conv(dilations = b_15_dilations_0, groups = b_15_groups_0, pad = b_15_pad_0, pad_type = b_15_pad_type_0, strides = b_15_strides_0, weight = model_model_layers_17_mlp_up_proj_weight_palettized, x = input_107)[name = string("b_15")]; + tensor c_15 = silu(x = input_109)[name = string("c_15")]; + tensor input_111 = mul(x = c_15, y = b_15)[name = string("input_111")]; + string e_15_pad_type_0 = const()[name = string("e_15_pad_type_0"), val = string("valid")]; + tensor e_15_strides_0 = const()[name = string("e_15_strides_0"), val = tensor([1, 1])]; + tensor e_15_pad_0 = const()[name = string("e_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_15_dilations_0 = const()[name = string("e_15_dilations_0"), val = tensor([1, 1])]; + int32 e_15_groups_0 = const()[name = string("e_15_groups_0"), val = int32(1)]; + tensor e_15 = conv(dilations = e_15_dilations_0, groups = e_15_groups_0, pad = e_15_pad_0, pad_type = e_15_pad_type_0, strides = e_15_strides_0, weight = model_model_layers_17_mlp_down_proj_weight_palettized, x = input_111)[name = string("e_15")]; + tensor var_4100_axes_0 = const()[name = string("op_4100_axes_0"), val = tensor([2])]; + tensor var_4100 = squeeze(axes = var_4100_axes_0, x = e_15)[name = string("op_4100")]; + tensor var_4101 = const()[name = string("op_4101"), val = tensor([0, 2, 1])]; + tensor var_4102 = transpose(perm = var_4101, x = var_4100)[name = string("transpose_6")]; + tensor hidden_states_49_cast_fp16 = add(x = hidden_states_47_cast_fp16, y = var_4102)[name = string("hidden_states_49_cast_fp16")]; + int32 var_4114 = const()[name = string("op_4114"), val = int32(-1)]; + fp16 const_176_promoted_to_fp16 = const()[name = string("const_176_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4116_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_176_promoted_to_fp16)[name = string("op_4116_cast_fp16")]; + bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; + tensor input_113_cast_fp16 = concat(axis = var_4114, interleave = input_113_interleave_0, values = (hidden_states_49_cast_fp16, var_4116_cast_fp16))[name = string("input_113_cast_fp16")]; + tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; + fp16 var_4111_to_fp16 = const()[name = string("op_4111_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_4111_to_fp16, x = input_113_cast_fp16)[name = string("normed_65_cast_fp16")]; + tensor normed_67_begin_0 = const()[name = string("normed_67_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_67_end_0 = const()[name = string("normed_67_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_67_end_mask_0 = const()[name = string("normed_67_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_67_cast_fp16 = slice_by_index(begin = normed_67_begin_0, end = normed_67_end_0, end_mask = normed_67_end_mask_0, x = normed_65_cast_fp16)[name = string("normed_67_cast_fp16")]; + tensor const_179_promoted_to_fp16 = const()[name = string("const_179_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(322861696)))]; + tensor hidden_states_51_cast_fp16 = mul(x = normed_67_cast_fp16, y = const_179_promoted_to_fp16)[name = string("hidden_states_51_cast_fp16")]; + tensor var_4133 = const()[name = string("op_4133"), val = tensor([0, 2, 1])]; + tensor var_4136_axes_0 = const()[name = string("op_4136_axes_0"), val = tensor([2])]; + tensor var_4134_cast_fp16 = transpose(perm = var_4133, x = hidden_states_51_cast_fp16)[name = string("transpose_5")]; + tensor var_4136_cast_fp16 = expand_dims(axes = var_4136_axes_0, x = var_4134_cast_fp16)[name = string("op_4136_cast_fp16")]; + string var_4152_pad_type_0 = const()[name = string("op_4152_pad_type_0"), val = string("valid")]; + tensor var_4152_strides_0 = const()[name = string("op_4152_strides_0"), val = tensor([1, 1])]; + tensor var_4152_pad_0 = const()[name = string("op_4152_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4152_dilations_0 = const()[name = string("op_4152_dilations_0"), val = tensor([1, 1])]; + int32 var_4152_groups_0 = const()[name = string("op_4152_groups_0"), val = int32(1)]; + tensor var_4152 = conv(bias = model_model_layers_18_self_attn_q_proj_bias, dilations = var_4152_dilations_0, groups = var_4152_groups_0, pad = var_4152_pad_0, pad_type = var_4152_pad_type_0, strides = var_4152_strides_0, weight = model_model_layers_18_self_attn_q_proj_weight_palettized, x = var_4136_cast_fp16)[name = string("op_4152")]; + tensor var_4157 = const()[name = string("op_4157"), val = tensor([1, 12, 1, 128])]; + tensor var_4158 = reshape(shape = var_4157, x = var_4152)[name = string("op_4158")]; + string var_4174_pad_type_0 = const()[name = string("op_4174_pad_type_0"), val = string("valid")]; + tensor var_4174_strides_0 = const()[name = string("op_4174_strides_0"), val = tensor([1, 1])]; + tensor var_4174_pad_0 = const()[name = string("op_4174_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4174_dilations_0 = const()[name = string("op_4174_dilations_0"), val = tensor([1, 1])]; + int32 var_4174_groups_0 = const()[name = string("op_4174_groups_0"), val = int32(1)]; + tensor var_4174 = conv(bias = model_model_layers_18_self_attn_k_proj_bias, dilations = var_4174_dilations_0, groups = var_4174_groups_0, pad = var_4174_pad_0, pad_type = var_4174_pad_type_0, strides = var_4174_strides_0, weight = model_model_layers_18_self_attn_k_proj_weight_palettized, x = var_4136_cast_fp16)[name = string("op_4174")]; + tensor var_4179 = const()[name = string("op_4179"), val = tensor([1, 2, 1, 128])]; + tensor var_4180 = reshape(shape = var_4179, x = var_4174)[name = string("op_4180")]; + string var_4196_pad_type_0 = const()[name = string("op_4196_pad_type_0"), val = string("valid")]; + tensor var_4196_strides_0 = const()[name = string("op_4196_strides_0"), val = tensor([1, 1])]; + tensor var_4196_pad_0 = const()[name = string("op_4196_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4196_dilations_0 = const()[name = string("op_4196_dilations_0"), val = tensor([1, 1])]; + int32 var_4196_groups_0 = const()[name = string("op_4196_groups_0"), val = int32(1)]; + tensor var_4196 = conv(bias = model_model_layers_18_self_attn_v_proj_bias, dilations = var_4196_dilations_0, groups = var_4196_groups_0, pad = var_4196_pad_0, pad_type = var_4196_pad_type_0, strides = var_4196_strides_0, weight = model_model_layers_18_self_attn_v_proj_weight_palettized, x = var_4136_cast_fp16)[name = string("op_4196")]; + tensor var_4201 = const()[name = string("op_4201"), val = tensor([1, 2, 1, 128])]; + tensor var_4202 = reshape(shape = var_4201, x = var_4196)[name = string("op_4202")]; + tensor var_4208 = mul(x = var_4158, y = cos_1_cast_fp16)[name = string("op_4208")]; + tensor x1_33_begin_0 = const()[name = string("x1_33_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_33_end_0 = const()[name = string("x1_33_end_0"), val = tensor([1, 12, 1, 64])]; + tensor x1_33_end_mask_0 = const()[name = string("x1_33_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_33 = slice_by_index(begin = x1_33_begin_0, end = x1_33_end_0, end_mask = x1_33_end_mask_0, x = var_4158)[name = string("x1_33")]; + tensor x2_33_begin_0 = const()[name = string("x2_33_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_33_end_0 = const()[name = string("x2_33_end_0"), val = tensor([1, 12, 1, 128])]; + tensor x2_33_end_mask_0 = const()[name = string("x2_33_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_33 = slice_by_index(begin = x2_33_begin_0, end = x2_33_end_0, end_mask = x2_33_end_mask_0, x = var_4158)[name = string("x2_33")]; + fp16 const_182_promoted = const()[name = string("const_182_promoted"), val = fp16(-0x1p+0)]; + tensor var_4229 = mul(x = x2_33, y = const_182_promoted)[name = string("op_4229")]; + int32 var_4231 = const()[name = string("op_4231"), val = int32(-1)]; + bool var_4232_interleave_0 = const()[name = string("op_4232_interleave_0"), val = bool(false)]; + tensor var_4232 = concat(axis = var_4231, interleave = var_4232_interleave_0, values = (var_4229, x1_33))[name = string("op_4232")]; + tensor var_4233 = mul(x = var_4232, y = sin_1_cast_fp16)[name = string("op_4233")]; + tensor query_states = add(x = var_4208, y = var_4233)[name = string("query_states")]; + tensor var_4236 = mul(x = var_4180, y = cos_1_cast_fp16)[name = string("op_4236")]; + tensor x1_begin_0 = const()[name = string("x1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_end_0 = const()[name = string("x1_end_0"), val = tensor([1, 2, 1, 64])]; + tensor x1_end_mask_0 = const()[name = string("x1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1 = slice_by_index(begin = x1_begin_0, end = x1_end_0, end_mask = x1_end_mask_0, x = var_4180)[name = string("x1")]; + tensor x2_begin_0 = const()[name = string("x2_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_end_0 = const()[name = string("x2_end_0"), val = tensor([1, 2, 1, 128])]; + tensor x2_end_mask_0 = const()[name = string("x2_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2 = slice_by_index(begin = x2_begin_0, end = x2_end_0, end_mask = x2_end_mask_0, x = var_4180)[name = string("x2")]; + fp16 const_185_promoted = const()[name = string("const_185_promoted"), val = fp16(-0x1p+0)]; + tensor var_4257 = mul(x = x2, y = const_185_promoted)[name = string("op_4257")]; + int32 var_4259 = const()[name = string("op_4259"), val = int32(-1)]; + bool var_4260_interleave_0 = const()[name = string("op_4260_interleave_0"), val = bool(false)]; + tensor var_4260 = concat(axis = var_4259, interleave = var_4260_interleave_0, values = (var_4257, x1))[name = string("op_4260")]; + tensor var_4261 = mul(x = var_4260, y = sin_1_cast_fp16)[name = string("op_4261")]; + tensor key_states_33 = add(x = var_4236, y = var_4261)[name = string("key_states_33")]; + tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([18])]; + tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; + tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; + tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([19])]; + int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; + bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; + tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_96, expand_dims_97, current_pos, expand_dims_99))[name = string("concat_66")]; + tensor concat_67_values1_0 = const()[name = string("concat_67_values1_0"), val = tensor([0])]; + tensor concat_67_values3_0 = const()[name = string("concat_67_values3_0"), val = tensor([0])]; + int32 concat_67_axis_0 = const()[name = string("concat_67_axis_0"), val = int32(0)]; + bool concat_67_interleave_0 = const()[name = string("concat_67_interleave_0"), val = bool(false)]; + tensor concat_67 = concat(axis = concat_67_axis_0, interleave = concat_67_interleave_0, values = (expand_dims_100, concat_67_values1_0, var_578, concat_67_values3_0))[name = string("concat_67")]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_66, begin_mask = model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0, end = concat_67, end_mask = model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_17_stride_0, update = key_states_33, x = coreml_update_state_33)[name = string("model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_16_write_state")]; + tensor coreml_update_state_34 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_16")]; + tensor expand_dims_102 = const()[name = string("expand_dims_102"), val = tensor([46])]; + tensor expand_dims_103 = const()[name = string("expand_dims_103"), val = tensor([0])]; + tensor expand_dims_105 = const()[name = string("expand_dims_105"), val = tensor([0])]; + tensor expand_dims_106 = const()[name = string("expand_dims_106"), val = tensor([47])]; + int32 concat_70_axis_0 = const()[name = string("concat_70_axis_0"), val = int32(0)]; + bool concat_70_interleave_0 = const()[name = string("concat_70_interleave_0"), val = bool(false)]; + tensor concat_70 = concat(axis = concat_70_axis_0, interleave = concat_70_interleave_0, values = (expand_dims_102, expand_dims_103, current_pos, expand_dims_105))[name = string("concat_70")]; + tensor concat_71_values1_0 = const()[name = string("concat_71_values1_0"), val = tensor([0])]; + tensor concat_71_values3_0 = const()[name = string("concat_71_values3_0"), val = tensor([0])]; + int32 concat_71_axis_0 = const()[name = string("concat_71_axis_0"), val = int32(0)]; + bool concat_71_interleave_0 = const()[name = string("concat_71_interleave_0"), val = bool(false)]; + tensor concat_71 = concat(axis = concat_71_axis_0, interleave = concat_71_interleave_0, values = (expand_dims_106, concat_71_values1_0, var_578, concat_71_values3_0))[name = string("concat_71")]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_70, begin_mask = model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0, end = concat_71, end_mask = model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_18_stride_0, update = var_4202, x = coreml_update_state_34)[name = string("model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_17_write_state")]; + tensor coreml_update_state_35 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_17")]; + tensor var_4316_begin_0 = const()[name = string("op_4316_begin_0"), val = tensor([18, 0, 0, 0])]; + tensor var_4316_end_0 = const()[name = string("op_4316_end_0"), val = tensor([19, 2, 2048, 128])]; + tensor var_4316_end_mask_0 = const()[name = string("op_4316_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4316_cast_fp16 = slice_by_index(begin = var_4316_begin_0, end = var_4316_end_0, end_mask = var_4316_end_mask_0, x = coreml_update_state_35)[name = string("op_4316_cast_fp16")]; + tensor K_layer_cache_axes_0 = const()[name = string("K_layer_cache_axes_0"), val = tensor([0])]; + tensor K_layer_cache_cast_fp16 = squeeze(axes = K_layer_cache_axes_0, x = var_4316_cast_fp16)[name = string("K_layer_cache_cast_fp16")]; + tensor var_4323_begin_0 = const()[name = string("op_4323_begin_0"), val = tensor([46, 0, 0, 0])]; + tensor var_4323_end_0 = const()[name = string("op_4323_end_0"), val = tensor([47, 2, 2048, 128])]; + tensor var_4323_end_mask_0 = const()[name = string("op_4323_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4323_cast_fp16 = slice_by_index(begin = var_4323_begin_0, end = var_4323_end_0, end_mask = var_4323_end_mask_0, x = coreml_update_state_35)[name = string("op_4323_cast_fp16")]; + tensor V_layer_cache_axes_0 = const()[name = string("V_layer_cache_axes_0"), val = tensor([0])]; + tensor V_layer_cache_cast_fp16 = squeeze(axes = V_layer_cache_axes_0, x = var_4323_cast_fp16)[name = string("V_layer_cache_cast_fp16")]; + tensor x_131_axes_0 = const()[name = string("x_131_axes_0"), val = tensor([1])]; + tensor x_131_cast_fp16 = expand_dims(axes = x_131_axes_0, x = K_layer_cache_cast_fp16)[name = string("x_131_cast_fp16")]; + tensor var_4360 = const()[name = string("op_4360"), val = tensor([1, 6, 1, 1])]; + tensor x_133_cast_fp16 = tile(reps = var_4360, x = x_131_cast_fp16)[name = string("x_133_cast_fp16")]; + tensor var_4372 = const()[name = string("op_4372"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_cast_fp16 = reshape(shape = var_4372, x = x_133_cast_fp16)[name = string("key_states_cast_fp16")]; + tensor x_137_axes_0 = const()[name = string("x_137_axes_0"), val = tensor([1])]; + tensor x_137_cast_fp16 = expand_dims(axes = x_137_axes_0, x = V_layer_cache_cast_fp16)[name = string("x_137_cast_fp16")]; + tensor var_4380 = const()[name = string("op_4380"), val = tensor([1, 6, 1, 1])]; + tensor x_139_cast_fp16 = tile(reps = var_4380, x = x_137_cast_fp16)[name = string("x_139_cast_fp16")]; + tensor var_4392 = const()[name = string("op_4392"), val = tensor([1, -1, 2048, 128])]; + tensor value_states_cast_fp16 = reshape(shape = var_4392, x = x_139_cast_fp16)[name = string("value_states_cast_fp16")]; + bool var_4415_transpose_x_1 = const()[name = string("op_4415_transpose_x_1"), val = bool(false)]; + bool var_4415_transpose_y_1 = const()[name = string("op_4415_transpose_y_1"), val = bool(true)]; + tensor var_4415_cast_fp16 = matmul(transpose_x = var_4415_transpose_x_1, transpose_y = var_4415_transpose_y_1, x = query_states, y = key_states_cast_fp16)[name = string("op_4415_cast_fp16")]; + fp16 var_4416_to_fp16 = const()[name = string("op_4416_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_33_cast_fp16 = mul(x = var_4415_cast_fp16, y = var_4416_to_fp16)[name = string("attn_logits_33_cast_fp16")]; + tensor attn_logits_cast_fp16 = add(x = attn_logits_33_cast_fp16, y = causal_mask)[name = string("attn_logits_cast_fp16")]; + int32 var_4443 = const()[name = string("op_4443"), val = int32(-1)]; + tensor var_4445_cast_fp16 = softmax(axis = var_4443, x = attn_logits_cast_fp16)[name = string("op_4445_cast_fp16")]; + bool attn_output_97_transpose_x_0 = const()[name = string("attn_output_97_transpose_x_0"), val = bool(false)]; + bool attn_output_97_transpose_y_0 = const()[name = string("attn_output_97_transpose_y_0"), val = bool(false)]; + tensor attn_output_97_cast_fp16 = matmul(transpose_x = attn_output_97_transpose_x_0, transpose_y = attn_output_97_transpose_y_0, x = var_4445_cast_fp16, y = value_states_cast_fp16)[name = string("attn_output_97_cast_fp16")]; + tensor var_4469_perm_0 = const()[name = string("op_4469_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_4473 = const()[name = string("op_4473"), val = tensor([1, 1, 1536])]; + tensor var_4469 = transpose(perm = var_4469_perm_0, x = attn_output_97_cast_fp16)[name = string("transpose_4")]; + tensor attn_output_103 = reshape(shape = var_4473, x = var_4469)[name = string("attn_output_103")]; + tensor var_4478 = const()[name = string("op_4478"), val = tensor([0, 2, 1])]; + tensor squeeze_8_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(322864832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324634368))))[name = string("squeeze_8_palettized")]; + string var_4494_pad_type_0 = const()[name = string("op_4494_pad_type_0"), val = string("valid")]; + int32 var_4494_groups_0 = const()[name = string("op_4494_groups_0"), val = int32(1)]; + tensor var_4494_strides_0 = const()[name = string("op_4494_strides_0"), val = tensor([1])]; + tensor var_4494_pad_0 = const()[name = string("op_4494_pad_0"), val = tensor([0, 0])]; + tensor var_4494_dilations_0 = const()[name = string("op_4494_dilations_0"), val = tensor([1])]; + tensor var_4479 = transpose(perm = var_4478, x = attn_output_103)[name = string("transpose_3")]; + tensor var_4494 = conv(dilations = var_4494_dilations_0, groups = var_4494_groups_0, pad = var_4494_pad_0, pad_type = var_4494_pad_type_0, strides = var_4494_strides_0, weight = squeeze_8_palettized, x = var_4479)[name = string("op_4494")]; + tensor var_4498 = const()[name = string("op_4498"), val = tensor([0, 2, 1])]; + tensor attn_output = transpose(perm = var_4498, x = var_4494)[name = string("transpose_2")]; + tensor hidden_states_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = attn_output)[name = string("hidden_states_cast_fp16")]; + int32 var_4511 = const()[name = string("op_4511"), val = int32(-1)]; + fp16 const_194_promoted_to_fp16 = const()[name = string("const_194_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4513_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_194_promoted_to_fp16)[name = string("op_4513_cast_fp16")]; + bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; + tensor input_119_cast_fp16 = concat(axis = var_4511, interleave = input_119_interleave_0, values = (hidden_states_cast_fp16, var_4513_cast_fp16))[name = string("input_119_cast_fp16")]; + tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; + fp16 var_4508_to_fp16 = const()[name = string("op_4508_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_4508_to_fp16, x = input_119_cast_fp16)[name = string("normed_69_cast_fp16")]; + tensor normed_begin_0 = const()[name = string("normed_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_end_0 = const()[name = string("normed_end_0"), val = tensor([1, 1, 1536])]; + tensor normed_end_mask_0 = const()[name = string("normed_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_cast_fp16 = slice_by_index(begin = normed_begin_0, end = normed_end_0, end_mask = normed_end_mask_0, x = normed_69_cast_fp16)[name = string("normed_cast_fp16")]; + tensor const_197_promoted_to_fp16 = const()[name = string("const_197_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324683584)))]; + tensor x_141_cast_fp16 = mul(x = normed_cast_fp16, y = const_197_promoted_to_fp16)[name = string("x_141_cast_fp16")]; + tensor var_4538 = const()[name = string("op_4538"), val = tensor([0, 2, 1])]; + tensor input_121_axes_0 = const()[name = string("input_121_axes_0"), val = tensor([2])]; + tensor var_4539 = transpose(perm = var_4538, x = x_141_cast_fp16)[name = string("transpose_1")]; + tensor input_121 = expand_dims(axes = input_121_axes_0, x = var_4539)[name = string("input_121")]; + string input_123_pad_type_0 = const()[name = string("input_123_pad_type_0"), val = string("valid")]; + tensor input_123_strides_0 = const()[name = string("input_123_strides_0"), val = tensor([1, 1])]; + tensor input_123_pad_0 = const()[name = string("input_123_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_123_dilations_0 = const()[name = string("input_123_dilations_0"), val = tensor([1, 1])]; + int32 input_123_groups_0 = const()[name = string("input_123_groups_0"), val = int32(1)]; + tensor input_123 = conv(dilations = input_123_dilations_0, groups = input_123_groups_0, pad = input_123_pad_0, pad_type = input_123_pad_type_0, strides = input_123_strides_0, weight = model_model_layers_18_mlp_gate_proj_weight_palettized, x = input_121)[name = string("input_123")]; + string b_pad_type_0 = const()[name = string("b_pad_type_0"), val = string("valid")]; + tensor b_strides_0 = const()[name = string("b_strides_0"), val = tensor([1, 1])]; + tensor b_pad_0 = const()[name = string("b_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_dilations_0 = const()[name = string("b_dilations_0"), val = tensor([1, 1])]; + int32 b_groups_0 = const()[name = string("b_groups_0"), val = int32(1)]; + tensor b = conv(dilations = b_dilations_0, groups = b_groups_0, pad = b_pad_0, pad_type = b_pad_type_0, strides = b_strides_0, weight = model_model_layers_18_mlp_up_proj_weight_palettized, x = input_121)[name = string("b")]; + tensor c = silu(x = input_123)[name = string("c")]; + tensor input = mul(x = c, y = b)[name = string("input")]; + string e_pad_type_0 = const()[name = string("e_pad_type_0"), val = string("valid")]; + tensor e_strides_0 = const()[name = string("e_strides_0"), val = tensor([1, 1])]; + tensor e_pad_0 = const()[name = string("e_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_dilations_0 = const()[name = string("e_dilations_0"), val = tensor([1, 1])]; + int32 e_groups_0 = const()[name = string("e_groups_0"), val = int32(1)]; + tensor e = conv(dilations = e_dilations_0, groups = e_groups_0, pad = e_pad_0, pad_type = e_pad_type_0, strides = e_strides_0, weight = model_model_layers_18_mlp_down_proj_weight_palettized, x = input)[name = string("e")]; + tensor var_4561_axes_0 = const()[name = string("op_4561_axes_0"), val = tensor([2])]; + tensor var_4561 = squeeze(axes = var_4561_axes_0, x = e)[name = string("op_4561")]; + tensor var_4562 = const()[name = string("op_4562"), val = tensor([0, 2, 1])]; + tensor var_4563 = transpose(perm = var_4562, x = var_4561)[name = string("transpose_0")]; + tensor output_hidden_states = add(x = hidden_states_cast_fp16, y = var_4563)[name = string("op_4565_cast_fp16")]; + tensor position_ids_tmp = identity(x = position_ids)[name = string("position_ids_tmp")]; + } -> (output_hidden_states); + func prefill(tensor causal_mask, tensor current_pos, tensor hidden_states, state> model_model_kv_cache_0, tensor position_ids) { + tensor model_model_layers_10_self_attn_q_proj_bias = const()[name = string("model_model_layers_10_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64)))]; + tensor model_model_layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1772736))))[name = string("model_model_layers_10_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_10_self_attn_k_proj_bias = const()[name = string("model_model_layers_10_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1821952)))]; + tensor model_model_layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1822528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2117504))))[name = string("model_model_layers_10_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_10_self_attn_v_proj_bias = const()[name = string("model_model_layers_10_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2125760)))]; + tensor model_model_layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2126336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2421312))))[name = string("model_model_layers_10_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_10_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2429568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(12751552))))[name = string("model_model_layers_10_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_10_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13038336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23360320))))[name = string("model_model_layers_10_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_10_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23647104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(33969088))))[name = string("model_model_layers_10_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_11_self_attn_q_proj_bias = const()[name = string("model_model_layers_11_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34018304)))]; + tensor model_model_layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34021440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35790976))))[name = string("model_model_layers_11_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_11_self_attn_k_proj_bias = const()[name = string("model_model_layers_11_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35840192)))]; + tensor model_model_layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(35840768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36135744))))[name = string("model_model_layers_11_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_11_self_attn_v_proj_bias = const()[name = string("model_model_layers_11_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36144000)))]; + tensor model_model_layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36144576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36439552))))[name = string("model_model_layers_11_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_11_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(36447808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46769792))))[name = string("model_model_layers_11_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_11_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47056576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57378560))))[name = string("model_model_layers_11_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_11_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57665344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67987328))))[name = string("model_model_layers_11_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_12_self_attn_q_proj_bias = const()[name = string("model_model_layers_12_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68036544)))]; + tensor model_model_layers_12_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68039680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69809216))))[name = string("model_model_layers_12_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_12_self_attn_k_proj_bias = const()[name = string("model_model_layers_12_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69858432)))]; + tensor model_model_layers_12_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(69859008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70153984))))[name = string("model_model_layers_12_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_12_self_attn_v_proj_bias = const()[name = string("model_model_layers_12_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70162240)))]; + tensor model_model_layers_12_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70162816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70457792))))[name = string("model_model_layers_12_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_12_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(70466048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80788032))))[name = string("model_model_layers_12_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_12_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81074816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91396800))))[name = string("model_model_layers_12_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_12_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(91683584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102005568))))[name = string("model_model_layers_12_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_13_self_attn_q_proj_bias = const()[name = string("model_model_layers_13_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102054784)))]; + tensor model_model_layers_13_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(102057920))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103827456))))[name = string("model_model_layers_13_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_13_self_attn_k_proj_bias = const()[name = string("model_model_layers_13_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103876672)))]; + tensor model_model_layers_13_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103877248))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104172224))))[name = string("model_model_layers_13_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_13_self_attn_v_proj_bias = const()[name = string("model_model_layers_13_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104180480)))]; + tensor model_model_layers_13_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104181056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104476032))))[name = string("model_model_layers_13_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_13_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(104484288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(114806272))))[name = string("model_model_layers_13_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_13_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115093056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125415040))))[name = string("model_model_layers_13_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_13_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125701824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(136023808))))[name = string("model_model_layers_13_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_14_self_attn_q_proj_bias = const()[name = string("model_model_layers_14_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(136073024)))]; + tensor model_model_layers_14_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(136076160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(137845696))))[name = string("model_model_layers_14_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_14_self_attn_k_proj_bias = const()[name = string("model_model_layers_14_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(137894912)))]; + tensor model_model_layers_14_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(137895488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138190464))))[name = string("model_model_layers_14_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_14_self_attn_v_proj_bias = const()[name = string("model_model_layers_14_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138198720)))]; + tensor model_model_layers_14_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138199296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138494272))))[name = string("model_model_layers_14_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_14_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(138502528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148824512))))[name = string("model_model_layers_14_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_14_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149111296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159433280))))[name = string("model_model_layers_14_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_14_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159720064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(170042048))))[name = string("model_model_layers_14_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_15_self_attn_q_proj_bias = const()[name = string("model_model_layers_15_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(170091264)))]; + tensor model_model_layers_15_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(170094400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(171863936))))[name = string("model_model_layers_15_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_15_self_attn_k_proj_bias = const()[name = string("model_model_layers_15_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(171913152)))]; + tensor model_model_layers_15_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(171913728))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172208704))))[name = string("model_model_layers_15_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_15_self_attn_v_proj_bias = const()[name = string("model_model_layers_15_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172216960)))]; + tensor model_model_layers_15_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172217536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172512512))))[name = string("model_model_layers_15_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_15_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172520768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(182842752))))[name = string("model_model_layers_15_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_15_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(183129536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193451520))))[name = string("model_model_layers_15_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_15_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193738304))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204060288))))[name = string("model_model_layers_15_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_16_self_attn_q_proj_bias = const()[name = string("model_model_layers_16_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204109504)))]; + tensor model_model_layers_16_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(204112640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205882176))))[name = string("model_model_layers_16_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_16_self_attn_k_proj_bias = const()[name = string("model_model_layers_16_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205931392)))]; + tensor model_model_layers_16_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205931968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206226944))))[name = string("model_model_layers_16_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_16_self_attn_v_proj_bias = const()[name = string("model_model_layers_16_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206235200)))]; + tensor model_model_layers_16_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206235776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206530752))))[name = string("model_model_layers_16_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_16_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(206539008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(216860992))))[name = string("model_model_layers_16_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_16_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(217147776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(227469760))))[name = string("model_model_layers_16_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_16_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(227756544))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(238078528))))[name = string("model_model_layers_16_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_17_self_attn_q_proj_bias = const()[name = string("model_model_layers_17_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(238127744)))]; + tensor model_model_layers_17_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(238130880))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239900416))))[name = string("model_model_layers_17_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_17_self_attn_k_proj_bias = const()[name = string("model_model_layers_17_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239949632)))]; + tensor model_model_layers_17_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(239950208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240245184))))[name = string("model_model_layers_17_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_17_self_attn_v_proj_bias = const()[name = string("model_model_layers_17_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240253440)))]; + tensor model_model_layers_17_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240254016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240548992))))[name = string("model_model_layers_17_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_17_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240557248))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(250879232))))[name = string("model_model_layers_17_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_17_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(251166016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261488000))))[name = string("model_model_layers_17_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_17_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261774784))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272096768))))[name = string("model_model_layers_17_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_18_self_attn_q_proj_bias = const()[name = string("model_model_layers_18_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272145984)))]; + tensor model_model_layers_18_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(272149120))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273918656))))[name = string("model_model_layers_18_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_18_self_attn_k_proj_bias = const()[name = string("model_model_layers_18_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273967872)))]; + tensor model_model_layers_18_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(273968448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274263424))))[name = string("model_model_layers_18_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_18_self_attn_v_proj_bias = const()[name = string("model_model_layers_18_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274271680)))]; + tensor model_model_layers_18_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274272256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274567232))))[name = string("model_model_layers_18_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_18_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(274575488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(284897472))))[name = string("model_model_layers_18_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_18_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(285184256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295506240))))[name = string("model_model_layers_18_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_18_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295793024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306115008))))[name = string("model_model_layers_18_mlp_down_proj_weight_palettized")]; + int32 var_390_batch_dims_0 = const()[name = string("op_390_batch_dims_0"), val = int32(0)]; + bool var_390_validate_indices_0 = const()[name = string("op_390_validate_indices_0"), val = bool(false)]; + tensor var_382_to_fp16 = const()[name = string("op_382_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307212864)))]; + string position_ids_to_int16_dtype_0 = const()[name = string("position_ids_to_int16_dtype_0"), val = string("int16")]; + string cast_78_dtype_0 = const()[name = string("cast_78_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; + tensor position_ids_to_int16 = cast(dtype = position_ids_to_int16_dtype_0, x = position_ids)[name = string("cast_5")]; + tensor cast_78 = cast(dtype = cast_78_dtype_0, x = position_ids_to_int16)[name = string("cast_4")]; + tensor greater_equal_0 = greater_equal(x = cast_78, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; + int32 slice_by_index_72 = const()[name = string("slice_by_index_72"), val = int32(4096)]; + tensor add_0 = add(x = cast_78, y = slice_by_index_72)[name = string("add_0")]; + tensor select_0 = select(a = cast_78, b = add_0, cond = greater_equal_0)[name = string("select_0")]; + string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; + string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; + tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; + tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; + tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; + int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(4096)]; + tensor add_0_1 = add(x = cast_0, y = slice_by_index_0)[name = string("add_0_1")]; + tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; + int32 op_390_cast_fp16_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_390_cast_fp16_cast_uint16_cast_uint16_axis_0"), val = int32(1)]; + tensor op_390_cast_fp16_cast_uint16_cast_uint16 = gather(axis = op_390_cast_fp16_cast_uint16_cast_uint16_axis_0, batch_dims = var_390_batch_dims_0, indices = select_0_1, validate_indices = var_390_validate_indices_0, x = var_382_to_fp16)[name = string("op_390_cast_fp16_cast_uint16_cast_uint16")]; + tensor var_394 = const()[name = string("op_394"), val = tensor([1, 64, 1, 128])]; + tensor cos_1_cast_fp16 = reshape(shape = var_394, x = op_390_cast_fp16_cast_uint16_cast_uint16)[name = string("cos_1_cast_fp16")]; + int32 var_404_axis_0 = const()[name = string("op_404_axis_0"), val = int32(1)]; + int32 var_404_batch_dims_0 = const()[name = string("op_404_batch_dims_0"), val = int32(0)]; + bool var_404_validate_indices_0 = const()[name = string("op_404_validate_indices_0"), val = bool(false)]; + tensor var_396_to_fp16 = const()[name = string("op_396_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(306164224)))]; + string position_ids_to_uint16_dtype_0 = const()[name = string("position_ids_to_uint16_dtype_0"), val = string("uint16")]; + tensor position_ids_to_uint16 = cast(dtype = position_ids_to_uint16_dtype_0, x = position_ids)[name = string("cast_1")]; + tensor var_404_cast_fp16_cast_uint16 = gather(axis = var_404_axis_0, batch_dims = var_404_batch_dims_0, indices = position_ids_to_uint16, validate_indices = var_404_validate_indices_0, x = var_396_to_fp16)[name = string("op_404_cast_fp16_cast_uint16")]; + tensor var_408 = const()[name = string("op_408"), val = tensor([1, 64, 1, 128])]; + tensor sin_1_cast_fp16 = reshape(shape = var_408, x = var_404_cast_fp16_cast_uint16)[name = string("sin_1_cast_fp16")]; + int32 var_429 = const()[name = string("op_429"), val = int32(-1)]; + fp16 const_1_promoted_to_fp16 = const()[name = string("const_1_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_431_cast_fp16 = mul(x = hidden_states, y = const_1_promoted_to_fp16)[name = string("op_431_cast_fp16")]; + bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; + tensor input_1_cast_fp16 = concat(axis = var_429, interleave = input_1_interleave_0, values = (hidden_states, var_431_cast_fp16))[name = string("input_1_cast_fp16")]; + tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; + fp16 var_426_to_fp16 = const()[name = string("op_426_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_426_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; + tensor normed_3_begin_0 = const()[name = string("normed_3_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_3_end_0 = const()[name = string("normed_3_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_3_end_mask_0 = const()[name = string("normed_3_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_3_cast_fp16 = slice_by_index(begin = normed_3_begin_0, end = normed_3_end_0, end_mask = normed_3_end_mask_0, x = normed_1_cast_fp16)[name = string("normed_3_cast_fp16")]; + tensor const_4_promoted_to_fp16 = const()[name = string("const_4_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308261504)))]; + tensor hidden_states_3_cast_fp16 = mul(x = normed_3_cast_fp16, y = const_4_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; + tensor var_454 = const()[name = string("op_454"), val = tensor([0, 2, 1])]; + tensor var_457_axes_0 = const()[name = string("op_457_axes_0"), val = tensor([2])]; + tensor var_455_cast_fp16 = transpose(perm = var_454, x = hidden_states_3_cast_fp16)[name = string("transpose_82")]; + tensor var_457_cast_fp16 = expand_dims(axes = var_457_axes_0, x = var_455_cast_fp16)[name = string("op_457_cast_fp16")]; + string query_states_1_pad_type_0 = const()[name = string("query_states_1_pad_type_0"), val = string("valid")]; + tensor query_states_1_strides_0 = const()[name = string("query_states_1_strides_0"), val = tensor([1, 1])]; + tensor query_states_1_pad_0 = const()[name = string("query_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_1_dilations_0 = const()[name = string("query_states_1_dilations_0"), val = tensor([1, 1])]; + int32 query_states_1_groups_0 = const()[name = string("query_states_1_groups_0"), val = int32(1)]; + tensor query_states_1 = conv(bias = model_model_layers_10_self_attn_q_proj_bias, dilations = query_states_1_dilations_0, groups = query_states_1_groups_0, pad = query_states_1_pad_0, pad_type = query_states_1_pad_type_0, strides = query_states_1_strides_0, weight = model_model_layers_10_self_attn_q_proj_weight_palettized, x = var_457_cast_fp16)[name = string("query_states_1")]; + string key_states_1_pad_type_0 = const()[name = string("key_states_1_pad_type_0"), val = string("valid")]; + tensor key_states_1_strides_0 = const()[name = string("key_states_1_strides_0"), val = tensor([1, 1])]; + tensor key_states_1_pad_0 = const()[name = string("key_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_1_dilations_0 = const()[name = string("key_states_1_dilations_0"), val = tensor([1, 1])]; + int32 key_states_1_groups_0 = const()[name = string("key_states_1_groups_0"), val = int32(1)]; + tensor key_states_1 = conv(bias = model_model_layers_10_self_attn_k_proj_bias, dilations = key_states_1_dilations_0, groups = key_states_1_groups_0, pad = key_states_1_pad_0, pad_type = key_states_1_pad_type_0, strides = key_states_1_strides_0, weight = model_model_layers_10_self_attn_k_proj_weight_palettized, x = var_457_cast_fp16)[name = string("key_states_1")]; + string value_states_1_pad_type_0 = const()[name = string("value_states_1_pad_type_0"), val = string("valid")]; + tensor value_states_1_strides_0 = const()[name = string("value_states_1_strides_0"), val = tensor([1, 1])]; + tensor value_states_1_pad_0 = const()[name = string("value_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_1_dilations_0 = const()[name = string("value_states_1_dilations_0"), val = tensor([1, 1])]; + int32 value_states_1_groups_0 = const()[name = string("value_states_1_groups_0"), val = int32(1)]; + tensor value_states_1 = conv(bias = model_model_layers_10_self_attn_v_proj_bias, dilations = value_states_1_dilations_0, groups = value_states_1_groups_0, pad = value_states_1_pad_0, pad_type = value_states_1_pad_type_0, strides = value_states_1_strides_0, weight = model_model_layers_10_self_attn_v_proj_weight_palettized, x = var_457_cast_fp16)[name = string("value_states_1")]; + tensor var_499 = const()[name = string("op_499"), val = tensor([1, 12, 128, 64])]; + tensor var_500 = reshape(shape = var_499, x = query_states_1)[name = string("op_500")]; + tensor var_505 = const()[name = string("op_505"), val = tensor([0, 1, 3, 2])]; + tensor var_510 = const()[name = string("op_510"), val = tensor([1, 2, 128, 64])]; + tensor var_511 = reshape(shape = var_510, x = key_states_1)[name = string("op_511")]; + tensor var_516 = const()[name = string("op_516"), val = tensor([0, 1, 3, 2])]; + tensor var_521 = const()[name = string("op_521"), val = tensor([1, 2, 128, 64])]; + tensor var_522 = reshape(shape = var_521, x = value_states_1)[name = string("op_522")]; + tensor var_527 = const()[name = string("op_527"), val = tensor([0, 1, 3, 2])]; + tensor var_533 = const()[name = string("op_533"), val = tensor([0, 2, 1, 3])]; + tensor var_539 = const()[name = string("op_539"), val = tensor([0, 2, 1, 3])]; + tensor q_1 = transpose(perm = var_505, x = var_500)[name = string("transpose_80")]; + tensor cos_5 = transpose(perm = var_533, x = cos_1_cast_fp16)[name = string("transpose_81")]; + tensor var_541 = mul(x = q_1, y = cos_5)[name = string("op_541")]; + tensor x1_1_begin_0 = const()[name = string("x1_1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_1_end_0 = const()[name = string("x1_1_end_0"), val = tensor([1, 12, 64, 64])]; + tensor x1_1_end_mask_0 = const()[name = string("x1_1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_1 = slice_by_index(begin = x1_1_begin_0, end = x1_1_end_0, end_mask = x1_1_end_mask_0, x = q_1)[name = string("x1_1")]; + tensor x2_1_begin_0 = const()[name = string("x2_1_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_1_end_0 = const()[name = string("x2_1_end_0"), val = tensor([1, 12, 64, 128])]; + tensor x2_1_end_mask_0 = const()[name = string("x2_1_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_1 = slice_by_index(begin = x2_1_begin_0, end = x2_1_end_0, end_mask = x2_1_end_mask_0, x = q_1)[name = string("x2_1")]; + fp16 const_8_promoted = const()[name = string("const_8_promoted"), val = fp16(-0x1p+0)]; + tensor var_562 = mul(x = x2_1, y = const_8_promoted)[name = string("op_562")]; + int32 var_564 = const()[name = string("op_564"), val = int32(-1)]; + bool var_565_interleave_0 = const()[name = string("op_565_interleave_0"), val = bool(false)]; + tensor var_565 = concat(axis = var_564, interleave = var_565_interleave_0, values = (var_562, x1_1))[name = string("op_565")]; + tensor sin_5 = transpose(perm = var_539, x = sin_1_cast_fp16)[name = string("transpose_79")]; + tensor var_566 = mul(x = var_565, y = sin_5)[name = string("op_566")]; + tensor query_states_3 = add(x = var_541, y = var_566)[name = string("query_states_3")]; + tensor k_1 = transpose(perm = var_516, x = var_511)[name = string("transpose_78")]; + tensor var_569 = mul(x = k_1, y = cos_5)[name = string("op_569")]; + tensor x1_3_begin_0 = const()[name = string("x1_3_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_3_end_0 = const()[name = string("x1_3_end_0"), val = tensor([1, 2, 64, 64])]; + tensor x1_3_end_mask_0 = const()[name = string("x1_3_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_3 = slice_by_index(begin = x1_3_begin_0, end = x1_3_end_0, end_mask = x1_3_end_mask_0, x = k_1)[name = string("x1_3")]; + tensor x2_3_begin_0 = const()[name = string("x2_3_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_3_end_0 = const()[name = string("x2_3_end_0"), val = tensor([1, 2, 64, 128])]; + tensor x2_3_end_mask_0 = const()[name = string("x2_3_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_3 = slice_by_index(begin = x2_3_begin_0, end = x2_3_end_0, end_mask = x2_3_end_mask_0, x = k_1)[name = string("x2_3")]; + fp16 const_11_promoted = const()[name = string("const_11_promoted"), val = fp16(-0x1p+0)]; + tensor var_590 = mul(x = x2_3, y = const_11_promoted)[name = string("op_590")]; + int32 var_592 = const()[name = string("op_592"), val = int32(-1)]; + bool var_593_interleave_0 = const()[name = string("op_593_interleave_0"), val = bool(false)]; + tensor var_593 = concat(axis = var_592, interleave = var_593_interleave_0, values = (var_590, x1_3))[name = string("op_593")]; + tensor var_594 = mul(x = var_593, y = sin_5)[name = string("op_594")]; + tensor key_states_3 = add(x = var_569, y = var_594)[name = string("key_states_3")]; + tensor seq_length_1 = const()[name = string("seq_length_1"), val = tensor([64])]; + tensor var_616 = add(x = current_pos, y = seq_length_1)[name = string("op_616")]; + tensor read_state_0 = read_state(input = model_model_kv_cache_0)[name = string("read_state_0")]; + tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([10])]; + tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; + tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; + tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([11])]; + int32 concat_2_axis_0 = const()[name = string("concat_2_axis_0"), val = int32(0)]; + bool concat_2_interleave_0 = const()[name = string("concat_2_interleave_0"), val = bool(false)]; + tensor concat_2 = concat(axis = concat_2_axis_0, interleave = concat_2_interleave_0, values = (expand_dims_0, expand_dims_1, current_pos, expand_dims_3))[name = string("concat_2")]; + tensor concat_3_values1_0 = const()[name = string("concat_3_values1_0"), val = tensor([0])]; + tensor concat_3_values3_0 = const()[name = string("concat_3_values3_0"), val = tensor([0])]; + int32 concat_3_axis_0 = const()[name = string("concat_3_axis_0"), val = int32(0)]; + bool concat_3_interleave_0 = const()[name = string("concat_3_interleave_0"), val = bool(false)]; + tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (expand_dims_4, concat_3_values1_0, var_616, concat_3_values3_0))[name = string("concat_3")]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_2, begin_mask = model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0, end = concat_3, end_mask = model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_1_stride_0, update = key_states_3, x = read_state_0)[name = string("model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_18_write_state")]; + tensor coreml_update_state_18 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_18")]; + tensor expand_dims_6 = const()[name = string("expand_dims_6"), val = tensor([38])]; + tensor expand_dims_7 = const()[name = string("expand_dims_7"), val = tensor([0])]; + tensor expand_dims_9 = const()[name = string("expand_dims_9"), val = tensor([0])]; + tensor expand_dims_10 = const()[name = string("expand_dims_10"), val = tensor([39])]; + int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; + bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; + tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_6, expand_dims_7, current_pos, expand_dims_9))[name = string("concat_6")]; + tensor concat_7_values1_0 = const()[name = string("concat_7_values1_0"), val = tensor([0])]; + tensor concat_7_values3_0 = const()[name = string("concat_7_values3_0"), val = tensor([0])]; + int32 concat_7_axis_0 = const()[name = string("concat_7_axis_0"), val = int32(0)]; + bool concat_7_interleave_0 = const()[name = string("concat_7_interleave_0"), val = bool(false)]; + tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (expand_dims_10, concat_7_values1_0, var_616, concat_7_values3_0))[name = string("concat_7")]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_3 = transpose(perm = var_527, x = var_522)[name = string("transpose_77")]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_6, begin_mask = model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0, end = concat_7, end_mask = model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_2_stride_0, update = value_states_3, x = coreml_update_state_18)[name = string("model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_19_write_state")]; + tensor coreml_update_state_19 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_19")]; + tensor var_665_begin_0 = const()[name = string("op_665_begin_0"), val = tensor([10, 0, 0, 0])]; + tensor var_665_end_0 = const()[name = string("op_665_end_0"), val = tensor([11, 2, 2048, 128])]; + tensor var_665_end_mask_0 = const()[name = string("op_665_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_665_cast_fp16 = slice_by_index(begin = var_665_begin_0, end = var_665_end_0, end_mask = var_665_end_mask_0, x = coreml_update_state_19)[name = string("op_665_cast_fp16")]; + tensor K_layer_cache_1_axes_0 = const()[name = string("K_layer_cache_1_axes_0"), val = tensor([0])]; + tensor K_layer_cache_1_cast_fp16 = squeeze(axes = K_layer_cache_1_axes_0, x = var_665_cast_fp16)[name = string("K_layer_cache_1_cast_fp16")]; + tensor var_672_begin_0 = const()[name = string("op_672_begin_0"), val = tensor([38, 0, 0, 0])]; + tensor var_672_end_0 = const()[name = string("op_672_end_0"), val = tensor([39, 2, 2048, 128])]; + tensor var_672_end_mask_0 = const()[name = string("op_672_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_672_cast_fp16 = slice_by_index(begin = var_672_begin_0, end = var_672_end_0, end_mask = var_672_end_mask_0, x = coreml_update_state_19)[name = string("op_672_cast_fp16")]; + tensor V_layer_cache_1_axes_0 = const()[name = string("V_layer_cache_1_axes_0"), val = tensor([0])]; + tensor V_layer_cache_1_cast_fp16 = squeeze(axes = V_layer_cache_1_axes_0, x = var_672_cast_fp16)[name = string("V_layer_cache_1_cast_fp16")]; + tensor x_3_axes_0 = const()[name = string("x_3_axes_0"), val = tensor([1])]; + tensor x_3_cast_fp16 = expand_dims(axes = x_3_axes_0, x = K_layer_cache_1_cast_fp16)[name = string("x_3_cast_fp16")]; + tensor var_701 = const()[name = string("op_701"), val = tensor([1, 6, 1, 1])]; + tensor x_5_cast_fp16 = tile(reps = var_701, x = x_3_cast_fp16)[name = string("x_5_cast_fp16")]; + tensor var_713 = const()[name = string("op_713"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_7_cast_fp16 = reshape(shape = var_713, x = x_5_cast_fp16)[name = string("key_states_7_cast_fp16")]; + tensor x_9_axes_0 = const()[name = string("x_9_axes_0"), val = tensor([1])]; + tensor x_9_cast_fp16 = expand_dims(axes = x_9_axes_0, x = V_layer_cache_1_cast_fp16)[name = string("x_9_cast_fp16")]; + tensor var_721 = const()[name = string("op_721"), val = tensor([1, 6, 1, 1])]; + tensor x_11_cast_fp16 = tile(reps = var_721, x = x_9_cast_fp16)[name = string("x_11_cast_fp16")]; + bool var_756_transpose_x_1 = const()[name = string("op_756_transpose_x_1"), val = bool(false)]; + bool var_756_transpose_y_1 = const()[name = string("op_756_transpose_y_1"), val = bool(true)]; + tensor var_756_cast_fp16 = matmul(transpose_x = var_756_transpose_x_1, transpose_y = var_756_transpose_y_1, x = query_states_3, y = key_states_7_cast_fp16)[name = string("op_756_cast_fp16")]; + fp16 var_757_to_fp16 = const()[name = string("op_757_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_1_cast_fp16 = mul(x = var_756_cast_fp16, y = var_757_to_fp16)[name = string("attn_logits_1_cast_fp16")]; + tensor attn_logits_3_cast_fp16 = add(x = attn_logits_1_cast_fp16, y = causal_mask)[name = string("attn_logits_3_cast_fp16")]; + int32 var_784 = const()[name = string("op_784"), val = int32(-1)]; + tensor var_786_cast_fp16 = softmax(axis = var_784, x = attn_logits_3_cast_fp16)[name = string("op_786_cast_fp16")]; + tensor concat_12 = const()[name = string("concat_12"), val = tensor([12, 64, 2048])]; + tensor reshape_0_cast_fp16 = reshape(shape = concat_12, x = var_786_cast_fp16)[name = string("reshape_0_cast_fp16")]; + tensor concat_13 = const()[name = string("concat_13"), val = tensor([12, 2048, 128])]; + tensor reshape_1_cast_fp16 = reshape(shape = concat_13, x = x_11_cast_fp16)[name = string("reshape_1_cast_fp16")]; + bool matmul_0_transpose_x_0 = const()[name = string("matmul_0_transpose_x_0"), val = bool(false)]; + bool matmul_0_transpose_y_0 = const()[name = string("matmul_0_transpose_y_0"), val = bool(false)]; + tensor matmul_0_cast_fp16 = matmul(transpose_x = matmul_0_transpose_x_0, transpose_y = matmul_0_transpose_y_0, x = reshape_0_cast_fp16, y = reshape_1_cast_fp16)[name = string("matmul_0_cast_fp16")]; + tensor concat_17 = const()[name = string("concat_17"), val = tensor([1, 12, 64, 128])]; + tensor reshape_2_cast_fp16 = reshape(shape = concat_17, x = matmul_0_cast_fp16)[name = string("reshape_2_cast_fp16")]; + tensor var_813_perm_0 = const()[name = string("op_813_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_832 = const()[name = string("op_832"), val = tensor([1, 64, 1536])]; + tensor var_813 = transpose(perm = var_813_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_76")]; + tensor attn_output_5 = reshape(shape = var_832, x = var_813)[name = string("attn_output_5")]; + tensor var_837 = const()[name = string("op_837"), val = tensor([0, 2, 1])]; + tensor squeeze_0_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308264640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310034176))))[name = string("squeeze_0_palettized")]; + string var_853_pad_type_0 = const()[name = string("op_853_pad_type_0"), val = string("valid")]; + int32 var_853_groups_0 = const()[name = string("op_853_groups_0"), val = int32(1)]; + tensor var_853_strides_0 = const()[name = string("op_853_strides_0"), val = tensor([1])]; + tensor var_853_pad_0 = const()[name = string("op_853_pad_0"), val = tensor([0, 0])]; + tensor var_853_dilations_0 = const()[name = string("op_853_dilations_0"), val = tensor([1])]; + tensor var_838 = transpose(perm = var_837, x = attn_output_5)[name = string("transpose_75")]; + tensor var_853 = conv(dilations = var_853_dilations_0, groups = var_853_groups_0, pad = var_853_pad_0, pad_type = var_853_pad_type_0, strides = var_853_strides_0, weight = squeeze_0_palettized, x = var_838)[name = string("op_853")]; + tensor var_857 = const()[name = string("op_857"), val = tensor([0, 2, 1])]; + tensor attn_output_9 = transpose(perm = var_857, x = var_853)[name = string("transpose_74")]; + tensor hidden_states_5_cast_fp16 = add(x = hidden_states, y = attn_output_9)[name = string("hidden_states_5_cast_fp16")]; + int32 var_870 = const()[name = string("op_870"), val = int32(-1)]; + fp16 const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_872_cast_fp16 = mul(x = hidden_states_5_cast_fp16, y = const_23_promoted_to_fp16)[name = string("op_872_cast_fp16")]; + bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; + tensor input_7_cast_fp16 = concat(axis = var_870, interleave = input_7_interleave_0, values = (hidden_states_5_cast_fp16, var_872_cast_fp16))[name = string("input_7_cast_fp16")]; + tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; + fp16 var_867_to_fp16 = const()[name = string("op_867_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_867_to_fp16, x = input_7_cast_fp16)[name = string("normed_5_cast_fp16")]; + tensor normed_7_begin_0 = const()[name = string("normed_7_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_7_end_0 = const()[name = string("normed_7_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_7_end_mask_0 = const()[name = string("normed_7_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_7_cast_fp16 = slice_by_index(begin = normed_7_begin_0, end = normed_7_end_0, end_mask = normed_7_end_mask_0, x = normed_5_cast_fp16)[name = string("normed_7_cast_fp16")]; + tensor const_26_promoted_to_fp16 = const()[name = string("const_26_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310083392)))]; + tensor x_13_cast_fp16 = mul(x = normed_7_cast_fp16, y = const_26_promoted_to_fp16)[name = string("x_13_cast_fp16")]; + tensor var_897 = const()[name = string("op_897"), val = tensor([0, 2, 1])]; + tensor input_9_axes_0 = const()[name = string("input_9_axes_0"), val = tensor([2])]; + tensor var_898 = transpose(perm = var_897, x = x_13_cast_fp16)[name = string("transpose_73")]; + tensor input_9 = expand_dims(axes = input_9_axes_0, x = var_898)[name = string("input_9")]; + string input_11_pad_type_0 = const()[name = string("input_11_pad_type_0"), val = string("valid")]; + tensor input_11_strides_0 = const()[name = string("input_11_strides_0"), val = tensor([1, 1])]; + tensor input_11_pad_0 = const()[name = string("input_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_11_dilations_0 = const()[name = string("input_11_dilations_0"), val = tensor([1, 1])]; + int32 input_11_groups_0 = const()[name = string("input_11_groups_0"), val = int32(1)]; + tensor input_11 = conv(dilations = input_11_dilations_0, groups = input_11_groups_0, pad = input_11_pad_0, pad_type = input_11_pad_type_0, strides = input_11_strides_0, weight = model_model_layers_10_mlp_gate_proj_weight_palettized, x = input_9)[name = string("input_11")]; + string b_1_pad_type_0 = const()[name = string("b_1_pad_type_0"), val = string("valid")]; + tensor b_1_strides_0 = const()[name = string("b_1_strides_0"), val = tensor([1, 1])]; + tensor b_1_pad_0 = const()[name = string("b_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_1_dilations_0 = const()[name = string("b_1_dilations_0"), val = tensor([1, 1])]; + int32 b_1_groups_0 = const()[name = string("b_1_groups_0"), val = int32(1)]; + tensor b_1 = conv(dilations = b_1_dilations_0, groups = b_1_groups_0, pad = b_1_pad_0, pad_type = b_1_pad_type_0, strides = b_1_strides_0, weight = model_model_layers_10_mlp_up_proj_weight_palettized, x = input_9)[name = string("b_1")]; + tensor c_1 = silu(x = input_11)[name = string("c_1")]; + tensor input_13 = mul(x = c_1, y = b_1)[name = string("input_13")]; + string e_1_pad_type_0 = const()[name = string("e_1_pad_type_0"), val = string("valid")]; + tensor e_1_strides_0 = const()[name = string("e_1_strides_0"), val = tensor([1, 1])]; + tensor e_1_pad_0 = const()[name = string("e_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_1_dilations_0 = const()[name = string("e_1_dilations_0"), val = tensor([1, 1])]; + int32 e_1_groups_0 = const()[name = string("e_1_groups_0"), val = int32(1)]; + tensor e_1 = conv(dilations = e_1_dilations_0, groups = e_1_groups_0, pad = e_1_pad_0, pad_type = e_1_pad_type_0, strides = e_1_strides_0, weight = model_model_layers_10_mlp_down_proj_weight_palettized, x = input_13)[name = string("e_1")]; + tensor var_920_axes_0 = const()[name = string("op_920_axes_0"), val = tensor([2])]; + tensor var_920 = squeeze(axes = var_920_axes_0, x = e_1)[name = string("op_920")]; + tensor var_921 = const()[name = string("op_921"), val = tensor([0, 2, 1])]; + tensor var_922 = transpose(perm = var_921, x = var_920)[name = string("transpose_72")]; + tensor hidden_states_7_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = var_922)[name = string("hidden_states_7_cast_fp16")]; + int32 var_934 = const()[name = string("op_934"), val = int32(-1)]; + fp16 const_27_promoted_to_fp16 = const()[name = string("const_27_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_936_cast_fp16 = mul(x = hidden_states_7_cast_fp16, y = const_27_promoted_to_fp16)[name = string("op_936_cast_fp16")]; + bool input_15_interleave_0 = const()[name = string("input_15_interleave_0"), val = bool(false)]; + tensor input_15_cast_fp16 = concat(axis = var_934, interleave = input_15_interleave_0, values = (hidden_states_7_cast_fp16, var_936_cast_fp16))[name = string("input_15_cast_fp16")]; + tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; + fp16 var_931_to_fp16 = const()[name = string("op_931_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_931_to_fp16, x = input_15_cast_fp16)[name = string("normed_9_cast_fp16")]; + tensor normed_11_begin_0 = const()[name = string("normed_11_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_11_end_0 = const()[name = string("normed_11_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_11_end_mask_0 = const()[name = string("normed_11_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_11_cast_fp16 = slice_by_index(begin = normed_11_begin_0, end = normed_11_end_0, end_mask = normed_11_end_mask_0, x = normed_9_cast_fp16)[name = string("normed_11_cast_fp16")]; + tensor const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310086528)))]; + tensor hidden_states_9_cast_fp16 = mul(x = normed_11_cast_fp16, y = const_30_promoted_to_fp16)[name = string("hidden_states_9_cast_fp16")]; + tensor var_959 = const()[name = string("op_959"), val = tensor([0, 2, 1])]; + tensor var_962_axes_0 = const()[name = string("op_962_axes_0"), val = tensor([2])]; + tensor var_960_cast_fp16 = transpose(perm = var_959, x = hidden_states_9_cast_fp16)[name = string("transpose_71")]; + tensor var_962_cast_fp16 = expand_dims(axes = var_962_axes_0, x = var_960_cast_fp16)[name = string("op_962_cast_fp16")]; + string query_states_7_pad_type_0 = const()[name = string("query_states_7_pad_type_0"), val = string("valid")]; + tensor query_states_7_strides_0 = const()[name = string("query_states_7_strides_0"), val = tensor([1, 1])]; + tensor query_states_7_pad_0 = const()[name = string("query_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_7_dilations_0 = const()[name = string("query_states_7_dilations_0"), val = tensor([1, 1])]; + int32 query_states_7_groups_0 = const()[name = string("query_states_7_groups_0"), val = int32(1)]; + tensor query_states_7 = conv(bias = model_model_layers_11_self_attn_q_proj_bias, dilations = query_states_7_dilations_0, groups = query_states_7_groups_0, pad = query_states_7_pad_0, pad_type = query_states_7_pad_type_0, strides = query_states_7_strides_0, weight = model_model_layers_11_self_attn_q_proj_weight_palettized, x = var_962_cast_fp16)[name = string("query_states_7")]; + string key_states_9_pad_type_0 = const()[name = string("key_states_9_pad_type_0"), val = string("valid")]; + tensor key_states_9_strides_0 = const()[name = string("key_states_9_strides_0"), val = tensor([1, 1])]; + tensor key_states_9_pad_0 = const()[name = string("key_states_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_9_dilations_0 = const()[name = string("key_states_9_dilations_0"), val = tensor([1, 1])]; + int32 key_states_9_groups_0 = const()[name = string("key_states_9_groups_0"), val = int32(1)]; + tensor key_states_9 = conv(bias = model_model_layers_11_self_attn_k_proj_bias, dilations = key_states_9_dilations_0, groups = key_states_9_groups_0, pad = key_states_9_pad_0, pad_type = key_states_9_pad_type_0, strides = key_states_9_strides_0, weight = model_model_layers_11_self_attn_k_proj_weight_palettized, x = var_962_cast_fp16)[name = string("key_states_9")]; + string value_states_9_pad_type_0 = const()[name = string("value_states_9_pad_type_0"), val = string("valid")]; + tensor value_states_9_strides_0 = const()[name = string("value_states_9_strides_0"), val = tensor([1, 1])]; + tensor value_states_9_pad_0 = const()[name = string("value_states_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_9_dilations_0 = const()[name = string("value_states_9_dilations_0"), val = tensor([1, 1])]; + int32 value_states_9_groups_0 = const()[name = string("value_states_9_groups_0"), val = int32(1)]; + tensor value_states_9 = conv(bias = model_model_layers_11_self_attn_v_proj_bias, dilations = value_states_9_dilations_0, groups = value_states_9_groups_0, pad = value_states_9_pad_0, pad_type = value_states_9_pad_type_0, strides = value_states_9_strides_0, weight = model_model_layers_11_self_attn_v_proj_weight_palettized, x = var_962_cast_fp16)[name = string("value_states_9")]; + tensor var_1004 = const()[name = string("op_1004"), val = tensor([1, 12, 128, 64])]; + tensor var_1005 = reshape(shape = var_1004, x = query_states_7)[name = string("op_1005")]; + tensor var_1010 = const()[name = string("op_1010"), val = tensor([0, 1, 3, 2])]; + tensor var_1015 = const()[name = string("op_1015"), val = tensor([1, 2, 128, 64])]; + tensor var_1016 = reshape(shape = var_1015, x = key_states_9)[name = string("op_1016")]; + tensor var_1021 = const()[name = string("op_1021"), val = tensor([0, 1, 3, 2])]; + tensor var_1026 = const()[name = string("op_1026"), val = tensor([1, 2, 128, 64])]; + tensor var_1027 = reshape(shape = var_1026, x = value_states_9)[name = string("op_1027")]; + tensor var_1032 = const()[name = string("op_1032"), val = tensor([0, 1, 3, 2])]; + tensor q_5 = transpose(perm = var_1010, x = var_1005)[name = string("transpose_70")]; + tensor var_1046 = mul(x = q_5, y = cos_5)[name = string("op_1046")]; + tensor x1_5_begin_0 = const()[name = string("x1_5_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_5_end_0 = const()[name = string("x1_5_end_0"), val = tensor([1, 12, 64, 64])]; + tensor x1_5_end_mask_0 = const()[name = string("x1_5_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_5 = slice_by_index(begin = x1_5_begin_0, end = x1_5_end_0, end_mask = x1_5_end_mask_0, x = q_5)[name = string("x1_5")]; + tensor x2_5_begin_0 = const()[name = string("x2_5_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_5_end_0 = const()[name = string("x2_5_end_0"), val = tensor([1, 12, 64, 128])]; + tensor x2_5_end_mask_0 = const()[name = string("x2_5_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_5 = slice_by_index(begin = x2_5_begin_0, end = x2_5_end_0, end_mask = x2_5_end_mask_0, x = q_5)[name = string("x2_5")]; + fp16 const_34_promoted = const()[name = string("const_34_promoted"), val = fp16(-0x1p+0)]; + tensor var_1067 = mul(x = x2_5, y = const_34_promoted)[name = string("op_1067")]; + int32 var_1069 = const()[name = string("op_1069"), val = int32(-1)]; + bool var_1070_interleave_0 = const()[name = string("op_1070_interleave_0"), val = bool(false)]; + tensor var_1070 = concat(axis = var_1069, interleave = var_1070_interleave_0, values = (var_1067, x1_5))[name = string("op_1070")]; + tensor var_1071 = mul(x = var_1070, y = sin_5)[name = string("op_1071")]; + tensor query_states_9 = add(x = var_1046, y = var_1071)[name = string("query_states_9")]; + tensor k_5 = transpose(perm = var_1021, x = var_1016)[name = string("transpose_69")]; + tensor var_1074 = mul(x = k_5, y = cos_5)[name = string("op_1074")]; + tensor x1_7_begin_0 = const()[name = string("x1_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_7_end_0 = const()[name = string("x1_7_end_0"), val = tensor([1, 2, 64, 64])]; + tensor x1_7_end_mask_0 = const()[name = string("x1_7_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_7 = slice_by_index(begin = x1_7_begin_0, end = x1_7_end_0, end_mask = x1_7_end_mask_0, x = k_5)[name = string("x1_7")]; + tensor x2_7_begin_0 = const()[name = string("x2_7_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_7_end_0 = const()[name = string("x2_7_end_0"), val = tensor([1, 2, 64, 128])]; + tensor x2_7_end_mask_0 = const()[name = string("x2_7_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_7 = slice_by_index(begin = x2_7_begin_0, end = x2_7_end_0, end_mask = x2_7_end_mask_0, x = k_5)[name = string("x2_7")]; + fp16 const_37_promoted = const()[name = string("const_37_promoted"), val = fp16(-0x1p+0)]; + tensor var_1095 = mul(x = x2_7, y = const_37_promoted)[name = string("op_1095")]; + int32 var_1097 = const()[name = string("op_1097"), val = int32(-1)]; + bool var_1098_interleave_0 = const()[name = string("op_1098_interleave_0"), val = bool(false)]; + tensor var_1098 = concat(axis = var_1097, interleave = var_1098_interleave_0, values = (var_1095, x1_7))[name = string("op_1098")]; + tensor var_1099 = mul(x = var_1098, y = sin_5)[name = string("op_1099")]; + tensor key_states_11 = add(x = var_1074, y = var_1099)[name = string("key_states_11")]; + tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([11])]; + tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; + tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; + tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([12])]; + int32 concat_20_axis_0 = const()[name = string("concat_20_axis_0"), val = int32(0)]; + bool concat_20_interleave_0 = const()[name = string("concat_20_interleave_0"), val = bool(false)]; + tensor concat_20 = concat(axis = concat_20_axis_0, interleave = concat_20_interleave_0, values = (expand_dims_12, expand_dims_13, current_pos, expand_dims_15))[name = string("concat_20")]; + tensor concat_21_values1_0 = const()[name = string("concat_21_values1_0"), val = tensor([0])]; + tensor concat_21_values3_0 = const()[name = string("concat_21_values3_0"), val = tensor([0])]; + int32 concat_21_axis_0 = const()[name = string("concat_21_axis_0"), val = int32(0)]; + bool concat_21_interleave_0 = const()[name = string("concat_21_interleave_0"), val = bool(false)]; + tensor concat_21 = concat(axis = concat_21_axis_0, interleave = concat_21_interleave_0, values = (expand_dims_16, concat_21_values1_0, var_616, concat_21_values3_0))[name = string("concat_21")]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_20, begin_mask = model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0, end = concat_21, end_mask = model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_3_stride_0, update = key_states_11, x = coreml_update_state_19)[name = string("model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_20_write_state")]; + tensor coreml_update_state_20 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_20")]; + tensor expand_dims_18 = const()[name = string("expand_dims_18"), val = tensor([39])]; + tensor expand_dims_19 = const()[name = string("expand_dims_19"), val = tensor([0])]; + tensor expand_dims_21 = const()[name = string("expand_dims_21"), val = tensor([0])]; + tensor expand_dims_22 = const()[name = string("expand_dims_22"), val = tensor([40])]; + int32 concat_24_axis_0 = const()[name = string("concat_24_axis_0"), val = int32(0)]; + bool concat_24_interleave_0 = const()[name = string("concat_24_interleave_0"), val = bool(false)]; + tensor concat_24 = concat(axis = concat_24_axis_0, interleave = concat_24_interleave_0, values = (expand_dims_18, expand_dims_19, current_pos, expand_dims_21))[name = string("concat_24")]; + tensor concat_25_values1_0 = const()[name = string("concat_25_values1_0"), val = tensor([0])]; + tensor concat_25_values3_0 = const()[name = string("concat_25_values3_0"), val = tensor([0])]; + int32 concat_25_axis_0 = const()[name = string("concat_25_axis_0"), val = int32(0)]; + bool concat_25_interleave_0 = const()[name = string("concat_25_interleave_0"), val = bool(false)]; + tensor concat_25 = concat(axis = concat_25_axis_0, interleave = concat_25_interleave_0, values = (expand_dims_22, concat_25_values1_0, var_616, concat_25_values3_0))[name = string("concat_25")]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_11 = transpose(perm = var_1032, x = var_1027)[name = string("transpose_68")]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_24, begin_mask = model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0, end = concat_25, end_mask = model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_4_stride_0, update = value_states_11, x = coreml_update_state_20)[name = string("model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_21_write_state")]; + tensor coreml_update_state_21 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_21")]; + tensor var_1170_begin_0 = const()[name = string("op_1170_begin_0"), val = tensor([11, 0, 0, 0])]; + tensor var_1170_end_0 = const()[name = string("op_1170_end_0"), val = tensor([12, 2, 2048, 128])]; + tensor var_1170_end_mask_0 = const()[name = string("op_1170_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1170_cast_fp16 = slice_by_index(begin = var_1170_begin_0, end = var_1170_end_0, end_mask = var_1170_end_mask_0, x = coreml_update_state_21)[name = string("op_1170_cast_fp16")]; + tensor K_layer_cache_3_axes_0 = const()[name = string("K_layer_cache_3_axes_0"), val = tensor([0])]; + tensor K_layer_cache_3_cast_fp16 = squeeze(axes = K_layer_cache_3_axes_0, x = var_1170_cast_fp16)[name = string("K_layer_cache_3_cast_fp16")]; + tensor var_1177_begin_0 = const()[name = string("op_1177_begin_0"), val = tensor([39, 0, 0, 0])]; + tensor var_1177_end_0 = const()[name = string("op_1177_end_0"), val = tensor([40, 2, 2048, 128])]; + tensor var_1177_end_mask_0 = const()[name = string("op_1177_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1177_cast_fp16 = slice_by_index(begin = var_1177_begin_0, end = var_1177_end_0, end_mask = var_1177_end_mask_0, x = coreml_update_state_21)[name = string("op_1177_cast_fp16")]; + tensor V_layer_cache_3_axes_0 = const()[name = string("V_layer_cache_3_axes_0"), val = tensor([0])]; + tensor V_layer_cache_3_cast_fp16 = squeeze(axes = V_layer_cache_3_axes_0, x = var_1177_cast_fp16)[name = string("V_layer_cache_3_cast_fp16")]; + tensor x_19_axes_0 = const()[name = string("x_19_axes_0"), val = tensor([1])]; + tensor x_19_cast_fp16 = expand_dims(axes = x_19_axes_0, x = K_layer_cache_3_cast_fp16)[name = string("x_19_cast_fp16")]; + tensor var_1206 = const()[name = string("op_1206"), val = tensor([1, 6, 1, 1])]; + tensor x_21_cast_fp16 = tile(reps = var_1206, x = x_19_cast_fp16)[name = string("x_21_cast_fp16")]; + tensor var_1218 = const()[name = string("op_1218"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_15_cast_fp16 = reshape(shape = var_1218, x = x_21_cast_fp16)[name = string("key_states_15_cast_fp16")]; + tensor x_25_axes_0 = const()[name = string("x_25_axes_0"), val = tensor([1])]; + tensor x_25_cast_fp16 = expand_dims(axes = x_25_axes_0, x = V_layer_cache_3_cast_fp16)[name = string("x_25_cast_fp16")]; + tensor var_1226 = const()[name = string("op_1226"), val = tensor([1, 6, 1, 1])]; + tensor x_27_cast_fp16 = tile(reps = var_1226, x = x_25_cast_fp16)[name = string("x_27_cast_fp16")]; + bool var_1261_transpose_x_1 = const()[name = string("op_1261_transpose_x_1"), val = bool(false)]; + bool var_1261_transpose_y_1 = const()[name = string("op_1261_transpose_y_1"), val = bool(true)]; + tensor var_1261_cast_fp16 = matmul(transpose_x = var_1261_transpose_x_1, transpose_y = var_1261_transpose_y_1, x = query_states_9, y = key_states_15_cast_fp16)[name = string("op_1261_cast_fp16")]; + fp16 var_1262_to_fp16 = const()[name = string("op_1262_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_5_cast_fp16 = mul(x = var_1261_cast_fp16, y = var_1262_to_fp16)[name = string("attn_logits_5_cast_fp16")]; + tensor attn_logits_7_cast_fp16 = add(x = attn_logits_5_cast_fp16, y = causal_mask)[name = string("attn_logits_7_cast_fp16")]; + int32 var_1289 = const()[name = string("op_1289"), val = int32(-1)]; + tensor var_1291_cast_fp16 = softmax(axis = var_1289, x = attn_logits_7_cast_fp16)[name = string("op_1291_cast_fp16")]; + tensor concat_30 = const()[name = string("concat_30"), val = tensor([12, 64, 2048])]; + tensor reshape_3_cast_fp16 = reshape(shape = concat_30, x = var_1291_cast_fp16)[name = string("reshape_3_cast_fp16")]; + tensor concat_31 = const()[name = string("concat_31"), val = tensor([12, 2048, 128])]; + tensor reshape_4_cast_fp16 = reshape(shape = concat_31, x = x_27_cast_fp16)[name = string("reshape_4_cast_fp16")]; + bool matmul_1_transpose_x_0 = const()[name = string("matmul_1_transpose_x_0"), val = bool(false)]; + bool matmul_1_transpose_y_0 = const()[name = string("matmul_1_transpose_y_0"), val = bool(false)]; + tensor matmul_1_cast_fp16 = matmul(transpose_x = matmul_1_transpose_x_0, transpose_y = matmul_1_transpose_y_0, x = reshape_3_cast_fp16, y = reshape_4_cast_fp16)[name = string("matmul_1_cast_fp16")]; + tensor concat_35 = const()[name = string("concat_35"), val = tensor([1, 12, 64, 128])]; + tensor reshape_5_cast_fp16 = reshape(shape = concat_35, x = matmul_1_cast_fp16)[name = string("reshape_5_cast_fp16")]; + tensor var_1318_perm_0 = const()[name = string("op_1318_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1337 = const()[name = string("op_1337"), val = tensor([1, 64, 1536])]; + tensor var_1318 = transpose(perm = var_1318_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_67")]; + tensor attn_output_15 = reshape(shape = var_1337, x = var_1318)[name = string("attn_output_15")]; + tensor var_1342 = const()[name = string("op_1342"), val = tensor([0, 2, 1])]; + tensor squeeze_1_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(310089664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311859200))))[name = string("squeeze_1_palettized")]; + string var_1358_pad_type_0 = const()[name = string("op_1358_pad_type_0"), val = string("valid")]; + int32 var_1358_groups_0 = const()[name = string("op_1358_groups_0"), val = int32(1)]; + tensor var_1358_strides_0 = const()[name = string("op_1358_strides_0"), val = tensor([1])]; + tensor var_1358_pad_0 = const()[name = string("op_1358_pad_0"), val = tensor([0, 0])]; + tensor var_1358_dilations_0 = const()[name = string("op_1358_dilations_0"), val = tensor([1])]; + tensor var_1343 = transpose(perm = var_1342, x = attn_output_15)[name = string("transpose_66")]; + tensor var_1358 = conv(dilations = var_1358_dilations_0, groups = var_1358_groups_0, pad = var_1358_pad_0, pad_type = var_1358_pad_type_0, strides = var_1358_strides_0, weight = squeeze_1_palettized, x = var_1343)[name = string("op_1358")]; + tensor var_1362 = const()[name = string("op_1362"), val = tensor([0, 2, 1])]; + tensor attn_output_19 = transpose(perm = var_1362, x = var_1358)[name = string("transpose_65")]; + tensor hidden_states_11_cast_fp16 = add(x = hidden_states_7_cast_fp16, y = attn_output_19)[name = string("hidden_states_11_cast_fp16")]; + int32 var_1375 = const()[name = string("op_1375"), val = int32(-1)]; + fp16 const_49_promoted_to_fp16 = const()[name = string("const_49_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1377_cast_fp16 = mul(x = hidden_states_11_cast_fp16, y = const_49_promoted_to_fp16)[name = string("op_1377_cast_fp16")]; + bool input_21_interleave_0 = const()[name = string("input_21_interleave_0"), val = bool(false)]; + tensor input_21_cast_fp16 = concat(axis = var_1375, interleave = input_21_interleave_0, values = (hidden_states_11_cast_fp16, var_1377_cast_fp16))[name = string("input_21_cast_fp16")]; + tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; + fp16 var_1372_to_fp16 = const()[name = string("op_1372_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_1372_to_fp16, x = input_21_cast_fp16)[name = string("normed_13_cast_fp16")]; + tensor normed_15_begin_0 = const()[name = string("normed_15_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_15_end_0 = const()[name = string("normed_15_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_15_end_mask_0 = const()[name = string("normed_15_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_15_cast_fp16 = slice_by_index(begin = normed_15_begin_0, end = normed_15_end_0, end_mask = normed_15_end_mask_0, x = normed_13_cast_fp16)[name = string("normed_15_cast_fp16")]; + tensor const_52_promoted_to_fp16 = const()[name = string("const_52_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311908416)))]; + tensor x_29_cast_fp16 = mul(x = normed_15_cast_fp16, y = const_52_promoted_to_fp16)[name = string("x_29_cast_fp16")]; + tensor var_1402 = const()[name = string("op_1402"), val = tensor([0, 2, 1])]; + tensor input_23_axes_0 = const()[name = string("input_23_axes_0"), val = tensor([2])]; + tensor var_1403 = transpose(perm = var_1402, x = x_29_cast_fp16)[name = string("transpose_64")]; + tensor input_23 = expand_dims(axes = input_23_axes_0, x = var_1403)[name = string("input_23")]; + string input_25_pad_type_0 = const()[name = string("input_25_pad_type_0"), val = string("valid")]; + tensor input_25_strides_0 = const()[name = string("input_25_strides_0"), val = tensor([1, 1])]; + tensor input_25_pad_0 = const()[name = string("input_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_25_dilations_0 = const()[name = string("input_25_dilations_0"), val = tensor([1, 1])]; + int32 input_25_groups_0 = const()[name = string("input_25_groups_0"), val = int32(1)]; + tensor input_25 = conv(dilations = input_25_dilations_0, groups = input_25_groups_0, pad = input_25_pad_0, pad_type = input_25_pad_type_0, strides = input_25_strides_0, weight = model_model_layers_11_mlp_gate_proj_weight_palettized, x = input_23)[name = string("input_25")]; + string b_3_pad_type_0 = const()[name = string("b_3_pad_type_0"), val = string("valid")]; + tensor b_3_strides_0 = const()[name = string("b_3_strides_0"), val = tensor([1, 1])]; + tensor b_3_pad_0 = const()[name = string("b_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_3_dilations_0 = const()[name = string("b_3_dilations_0"), val = tensor([1, 1])]; + int32 b_3_groups_0 = const()[name = string("b_3_groups_0"), val = int32(1)]; + tensor b_3 = conv(dilations = b_3_dilations_0, groups = b_3_groups_0, pad = b_3_pad_0, pad_type = b_3_pad_type_0, strides = b_3_strides_0, weight = model_model_layers_11_mlp_up_proj_weight_palettized, x = input_23)[name = string("b_3")]; + tensor c_3 = silu(x = input_25)[name = string("c_3")]; + tensor input_27 = mul(x = c_3, y = b_3)[name = string("input_27")]; + string e_3_pad_type_0 = const()[name = string("e_3_pad_type_0"), val = string("valid")]; + tensor e_3_strides_0 = const()[name = string("e_3_strides_0"), val = tensor([1, 1])]; + tensor e_3_pad_0 = const()[name = string("e_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_3_dilations_0 = const()[name = string("e_3_dilations_0"), val = tensor([1, 1])]; + int32 e_3_groups_0 = const()[name = string("e_3_groups_0"), val = int32(1)]; + tensor e_3 = conv(dilations = e_3_dilations_0, groups = e_3_groups_0, pad = e_3_pad_0, pad_type = e_3_pad_type_0, strides = e_3_strides_0, weight = model_model_layers_11_mlp_down_proj_weight_palettized, x = input_27)[name = string("e_3")]; + tensor var_1425_axes_0 = const()[name = string("op_1425_axes_0"), val = tensor([2])]; + tensor var_1425 = squeeze(axes = var_1425_axes_0, x = e_3)[name = string("op_1425")]; + tensor var_1426 = const()[name = string("op_1426"), val = tensor([0, 2, 1])]; + tensor var_1427 = transpose(perm = var_1426, x = var_1425)[name = string("transpose_63")]; + tensor hidden_states_13_cast_fp16 = add(x = hidden_states_11_cast_fp16, y = var_1427)[name = string("hidden_states_13_cast_fp16")]; + int32 var_1439 = const()[name = string("op_1439"), val = int32(-1)]; + fp16 const_53_promoted_to_fp16 = const()[name = string("const_53_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1441_cast_fp16 = mul(x = hidden_states_13_cast_fp16, y = const_53_promoted_to_fp16)[name = string("op_1441_cast_fp16")]; + bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; + tensor input_29_cast_fp16 = concat(axis = var_1439, interleave = input_29_interleave_0, values = (hidden_states_13_cast_fp16, var_1441_cast_fp16))[name = string("input_29_cast_fp16")]; + tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; + fp16 var_1436_to_fp16 = const()[name = string("op_1436_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_1436_to_fp16, x = input_29_cast_fp16)[name = string("normed_17_cast_fp16")]; + tensor normed_19_begin_0 = const()[name = string("normed_19_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_19_end_0 = const()[name = string("normed_19_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_19_end_mask_0 = const()[name = string("normed_19_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_19_cast_fp16 = slice_by_index(begin = normed_19_begin_0, end = normed_19_end_0, end_mask = normed_19_end_mask_0, x = normed_17_cast_fp16)[name = string("normed_19_cast_fp16")]; + tensor const_56_promoted_to_fp16 = const()[name = string("const_56_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311911552)))]; + tensor hidden_states_15_cast_fp16 = mul(x = normed_19_cast_fp16, y = const_56_promoted_to_fp16)[name = string("hidden_states_15_cast_fp16")]; + tensor var_1464 = const()[name = string("op_1464"), val = tensor([0, 2, 1])]; + tensor var_1467_axes_0 = const()[name = string("op_1467_axes_0"), val = tensor([2])]; + tensor var_1465_cast_fp16 = transpose(perm = var_1464, x = hidden_states_15_cast_fp16)[name = string("transpose_62")]; + tensor var_1467_cast_fp16 = expand_dims(axes = var_1467_axes_0, x = var_1465_cast_fp16)[name = string("op_1467_cast_fp16")]; + string query_states_13_pad_type_0 = const()[name = string("query_states_13_pad_type_0"), val = string("valid")]; + tensor query_states_13_strides_0 = const()[name = string("query_states_13_strides_0"), val = tensor([1, 1])]; + tensor query_states_13_pad_0 = const()[name = string("query_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_13_dilations_0 = const()[name = string("query_states_13_dilations_0"), val = tensor([1, 1])]; + int32 query_states_13_groups_0 = const()[name = string("query_states_13_groups_0"), val = int32(1)]; + tensor query_states_13 = conv(bias = model_model_layers_12_self_attn_q_proj_bias, dilations = query_states_13_dilations_0, groups = query_states_13_groups_0, pad = query_states_13_pad_0, pad_type = query_states_13_pad_type_0, strides = query_states_13_strides_0, weight = model_model_layers_12_self_attn_q_proj_weight_palettized, x = var_1467_cast_fp16)[name = string("query_states_13")]; + string key_states_17_pad_type_0 = const()[name = string("key_states_17_pad_type_0"), val = string("valid")]; + tensor key_states_17_strides_0 = const()[name = string("key_states_17_strides_0"), val = tensor([1, 1])]; + tensor key_states_17_pad_0 = const()[name = string("key_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_17_dilations_0 = const()[name = string("key_states_17_dilations_0"), val = tensor([1, 1])]; + int32 key_states_17_groups_0 = const()[name = string("key_states_17_groups_0"), val = int32(1)]; + tensor key_states_17 = conv(bias = model_model_layers_12_self_attn_k_proj_bias, dilations = key_states_17_dilations_0, groups = key_states_17_groups_0, pad = key_states_17_pad_0, pad_type = key_states_17_pad_type_0, strides = key_states_17_strides_0, weight = model_model_layers_12_self_attn_k_proj_weight_palettized, x = var_1467_cast_fp16)[name = string("key_states_17")]; + string value_states_17_pad_type_0 = const()[name = string("value_states_17_pad_type_0"), val = string("valid")]; + tensor value_states_17_strides_0 = const()[name = string("value_states_17_strides_0"), val = tensor([1, 1])]; + tensor value_states_17_pad_0 = const()[name = string("value_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_17_dilations_0 = const()[name = string("value_states_17_dilations_0"), val = tensor([1, 1])]; + int32 value_states_17_groups_0 = const()[name = string("value_states_17_groups_0"), val = int32(1)]; + tensor value_states_17 = conv(bias = model_model_layers_12_self_attn_v_proj_bias, dilations = value_states_17_dilations_0, groups = value_states_17_groups_0, pad = value_states_17_pad_0, pad_type = value_states_17_pad_type_0, strides = value_states_17_strides_0, weight = model_model_layers_12_self_attn_v_proj_weight_palettized, x = var_1467_cast_fp16)[name = string("value_states_17")]; + tensor var_1509 = const()[name = string("op_1509"), val = tensor([1, 12, 128, 64])]; + tensor var_1510 = reshape(shape = var_1509, x = query_states_13)[name = string("op_1510")]; + tensor var_1515 = const()[name = string("op_1515"), val = tensor([0, 1, 3, 2])]; + tensor var_1520 = const()[name = string("op_1520"), val = tensor([1, 2, 128, 64])]; + tensor var_1521 = reshape(shape = var_1520, x = key_states_17)[name = string("op_1521")]; + tensor var_1526 = const()[name = string("op_1526"), val = tensor([0, 1, 3, 2])]; + tensor var_1531 = const()[name = string("op_1531"), val = tensor([1, 2, 128, 64])]; + tensor var_1532 = reshape(shape = var_1531, x = value_states_17)[name = string("op_1532")]; + tensor var_1537 = const()[name = string("op_1537"), val = tensor([0, 1, 3, 2])]; + tensor q_9 = transpose(perm = var_1515, x = var_1510)[name = string("transpose_61")]; + tensor var_1551 = mul(x = q_9, y = cos_5)[name = string("op_1551")]; + tensor x1_9_begin_0 = const()[name = string("x1_9_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_9_end_0 = const()[name = string("x1_9_end_0"), val = tensor([1, 12, 64, 64])]; + tensor x1_9_end_mask_0 = const()[name = string("x1_9_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_9 = slice_by_index(begin = x1_9_begin_0, end = x1_9_end_0, end_mask = x1_9_end_mask_0, x = q_9)[name = string("x1_9")]; + tensor x2_9_begin_0 = const()[name = string("x2_9_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_9_end_0 = const()[name = string("x2_9_end_0"), val = tensor([1, 12, 64, 128])]; + tensor x2_9_end_mask_0 = const()[name = string("x2_9_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_9 = slice_by_index(begin = x2_9_begin_0, end = x2_9_end_0, end_mask = x2_9_end_mask_0, x = q_9)[name = string("x2_9")]; + fp16 const_60_promoted = const()[name = string("const_60_promoted"), val = fp16(-0x1p+0)]; + tensor var_1572 = mul(x = x2_9, y = const_60_promoted)[name = string("op_1572")]; + int32 var_1574 = const()[name = string("op_1574"), val = int32(-1)]; + bool var_1575_interleave_0 = const()[name = string("op_1575_interleave_0"), val = bool(false)]; + tensor var_1575 = concat(axis = var_1574, interleave = var_1575_interleave_0, values = (var_1572, x1_9))[name = string("op_1575")]; + tensor var_1576 = mul(x = var_1575, y = sin_5)[name = string("op_1576")]; + tensor query_states_15 = add(x = var_1551, y = var_1576)[name = string("query_states_15")]; + tensor k_9 = transpose(perm = var_1526, x = var_1521)[name = string("transpose_60")]; + tensor var_1579 = mul(x = k_9, y = cos_5)[name = string("op_1579")]; + tensor x1_11_begin_0 = const()[name = string("x1_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_11_end_0 = const()[name = string("x1_11_end_0"), val = tensor([1, 2, 64, 64])]; + tensor x1_11_end_mask_0 = const()[name = string("x1_11_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_11 = slice_by_index(begin = x1_11_begin_0, end = x1_11_end_0, end_mask = x1_11_end_mask_0, x = k_9)[name = string("x1_11")]; + tensor x2_11_begin_0 = const()[name = string("x2_11_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_11_end_0 = const()[name = string("x2_11_end_0"), val = tensor([1, 2, 64, 128])]; + tensor x2_11_end_mask_0 = const()[name = string("x2_11_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_11 = slice_by_index(begin = x2_11_begin_0, end = x2_11_end_0, end_mask = x2_11_end_mask_0, x = k_9)[name = string("x2_11")]; + fp16 const_63_promoted = const()[name = string("const_63_promoted"), val = fp16(-0x1p+0)]; + tensor var_1600 = mul(x = x2_11, y = const_63_promoted)[name = string("op_1600")]; + int32 var_1602 = const()[name = string("op_1602"), val = int32(-1)]; + bool var_1603_interleave_0 = const()[name = string("op_1603_interleave_0"), val = bool(false)]; + tensor var_1603 = concat(axis = var_1602, interleave = var_1603_interleave_0, values = (var_1600, x1_11))[name = string("op_1603")]; + tensor var_1604 = mul(x = var_1603, y = sin_5)[name = string("op_1604")]; + tensor key_states_19 = add(x = var_1579, y = var_1604)[name = string("key_states_19")]; + tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([12])]; + tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; + tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; + tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([13])]; + int32 concat_38_axis_0 = const()[name = string("concat_38_axis_0"), val = int32(0)]; + bool concat_38_interleave_0 = const()[name = string("concat_38_interleave_0"), val = bool(false)]; + tensor concat_38 = concat(axis = concat_38_axis_0, interleave = concat_38_interleave_0, values = (expand_dims_24, expand_dims_25, current_pos, expand_dims_27))[name = string("concat_38")]; + tensor concat_39_values1_0 = const()[name = string("concat_39_values1_0"), val = tensor([0])]; + tensor concat_39_values3_0 = const()[name = string("concat_39_values3_0"), val = tensor([0])]; + int32 concat_39_axis_0 = const()[name = string("concat_39_axis_0"), val = int32(0)]; + bool concat_39_interleave_0 = const()[name = string("concat_39_interleave_0"), val = bool(false)]; + tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (expand_dims_28, concat_39_values1_0, var_616, concat_39_values3_0))[name = string("concat_39")]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_38, begin_mask = model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0, end = concat_39, end_mask = model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_5_stride_0, update = key_states_19, x = coreml_update_state_21)[name = string("model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_22_write_state")]; + tensor coreml_update_state_22 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_22")]; + tensor expand_dims_30 = const()[name = string("expand_dims_30"), val = tensor([40])]; + tensor expand_dims_31 = const()[name = string("expand_dims_31"), val = tensor([0])]; + tensor expand_dims_33 = const()[name = string("expand_dims_33"), val = tensor([0])]; + tensor expand_dims_34 = const()[name = string("expand_dims_34"), val = tensor([41])]; + int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; + bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; + tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_30, expand_dims_31, current_pos, expand_dims_33))[name = string("concat_42")]; + tensor concat_43_values1_0 = const()[name = string("concat_43_values1_0"), val = tensor([0])]; + tensor concat_43_values3_0 = const()[name = string("concat_43_values3_0"), val = tensor([0])]; + int32 concat_43_axis_0 = const()[name = string("concat_43_axis_0"), val = int32(0)]; + bool concat_43_interleave_0 = const()[name = string("concat_43_interleave_0"), val = bool(false)]; + tensor concat_43 = concat(axis = concat_43_axis_0, interleave = concat_43_interleave_0, values = (expand_dims_34, concat_43_values1_0, var_616, concat_43_values3_0))[name = string("concat_43")]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_19 = transpose(perm = var_1537, x = var_1532)[name = string("transpose_59")]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_42, begin_mask = model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0, end = concat_43, end_mask = model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_6_stride_0, update = value_states_19, x = coreml_update_state_22)[name = string("model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_23_write_state")]; + tensor coreml_update_state_23 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_23")]; + tensor var_1675_begin_0 = const()[name = string("op_1675_begin_0"), val = tensor([12, 0, 0, 0])]; + tensor var_1675_end_0 = const()[name = string("op_1675_end_0"), val = tensor([13, 2, 2048, 128])]; + tensor var_1675_end_mask_0 = const()[name = string("op_1675_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1675_cast_fp16 = slice_by_index(begin = var_1675_begin_0, end = var_1675_end_0, end_mask = var_1675_end_mask_0, x = coreml_update_state_23)[name = string("op_1675_cast_fp16")]; + tensor K_layer_cache_5_axes_0 = const()[name = string("K_layer_cache_5_axes_0"), val = tensor([0])]; + tensor K_layer_cache_5_cast_fp16 = squeeze(axes = K_layer_cache_5_axes_0, x = var_1675_cast_fp16)[name = string("K_layer_cache_5_cast_fp16")]; + tensor var_1682_begin_0 = const()[name = string("op_1682_begin_0"), val = tensor([40, 0, 0, 0])]; + tensor var_1682_end_0 = const()[name = string("op_1682_end_0"), val = tensor([41, 2, 2048, 128])]; + tensor var_1682_end_mask_0 = const()[name = string("op_1682_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1682_cast_fp16 = slice_by_index(begin = var_1682_begin_0, end = var_1682_end_0, end_mask = var_1682_end_mask_0, x = coreml_update_state_23)[name = string("op_1682_cast_fp16")]; + tensor V_layer_cache_5_axes_0 = const()[name = string("V_layer_cache_5_axes_0"), val = tensor([0])]; + tensor V_layer_cache_5_cast_fp16 = squeeze(axes = V_layer_cache_5_axes_0, x = var_1682_cast_fp16)[name = string("V_layer_cache_5_cast_fp16")]; + tensor x_35_axes_0 = const()[name = string("x_35_axes_0"), val = tensor([1])]; + tensor x_35_cast_fp16 = expand_dims(axes = x_35_axes_0, x = K_layer_cache_5_cast_fp16)[name = string("x_35_cast_fp16")]; + tensor var_1711 = const()[name = string("op_1711"), val = tensor([1, 6, 1, 1])]; + tensor x_37_cast_fp16 = tile(reps = var_1711, x = x_35_cast_fp16)[name = string("x_37_cast_fp16")]; + tensor var_1723 = const()[name = string("op_1723"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_23_cast_fp16 = reshape(shape = var_1723, x = x_37_cast_fp16)[name = string("key_states_23_cast_fp16")]; + tensor x_41_axes_0 = const()[name = string("x_41_axes_0"), val = tensor([1])]; + tensor x_41_cast_fp16 = expand_dims(axes = x_41_axes_0, x = V_layer_cache_5_cast_fp16)[name = string("x_41_cast_fp16")]; + tensor var_1731 = const()[name = string("op_1731"), val = tensor([1, 6, 1, 1])]; + tensor x_43_cast_fp16 = tile(reps = var_1731, x = x_41_cast_fp16)[name = string("x_43_cast_fp16")]; + bool var_1766_transpose_x_1 = const()[name = string("op_1766_transpose_x_1"), val = bool(false)]; + bool var_1766_transpose_y_1 = const()[name = string("op_1766_transpose_y_1"), val = bool(true)]; + tensor var_1766_cast_fp16 = matmul(transpose_x = var_1766_transpose_x_1, transpose_y = var_1766_transpose_y_1, x = query_states_15, y = key_states_23_cast_fp16)[name = string("op_1766_cast_fp16")]; + fp16 var_1767_to_fp16 = const()[name = string("op_1767_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_9_cast_fp16 = mul(x = var_1766_cast_fp16, y = var_1767_to_fp16)[name = string("attn_logits_9_cast_fp16")]; + tensor attn_logits_11_cast_fp16 = add(x = attn_logits_9_cast_fp16, y = causal_mask)[name = string("attn_logits_11_cast_fp16")]; + int32 var_1794 = const()[name = string("op_1794"), val = int32(-1)]; + tensor var_1796_cast_fp16 = softmax(axis = var_1794, x = attn_logits_11_cast_fp16)[name = string("op_1796_cast_fp16")]; + tensor concat_48 = const()[name = string("concat_48"), val = tensor([12, 64, 2048])]; + tensor reshape_6_cast_fp16 = reshape(shape = concat_48, x = var_1796_cast_fp16)[name = string("reshape_6_cast_fp16")]; + tensor concat_49 = const()[name = string("concat_49"), val = tensor([12, 2048, 128])]; + tensor reshape_7_cast_fp16 = reshape(shape = concat_49, x = x_43_cast_fp16)[name = string("reshape_7_cast_fp16")]; + bool matmul_2_transpose_x_0 = const()[name = string("matmul_2_transpose_x_0"), val = bool(false)]; + bool matmul_2_transpose_y_0 = const()[name = string("matmul_2_transpose_y_0"), val = bool(false)]; + tensor matmul_2_cast_fp16 = matmul(transpose_x = matmul_2_transpose_x_0, transpose_y = matmul_2_transpose_y_0, x = reshape_6_cast_fp16, y = reshape_7_cast_fp16)[name = string("matmul_2_cast_fp16")]; + tensor concat_53 = const()[name = string("concat_53"), val = tensor([1, 12, 64, 128])]; + tensor reshape_8_cast_fp16 = reshape(shape = concat_53, x = matmul_2_cast_fp16)[name = string("reshape_8_cast_fp16")]; + tensor var_1823_perm_0 = const()[name = string("op_1823_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1842 = const()[name = string("op_1842"), val = tensor([1, 64, 1536])]; + tensor var_1823 = transpose(perm = var_1823_perm_0, x = reshape_8_cast_fp16)[name = string("transpose_58")]; + tensor attn_output_25 = reshape(shape = var_1842, x = var_1823)[name = string("attn_output_25")]; + tensor var_1847 = const()[name = string("op_1847"), val = tensor([0, 2, 1])]; + tensor squeeze_2_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(311914688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313684224))))[name = string("squeeze_2_palettized")]; + string var_1863_pad_type_0 = const()[name = string("op_1863_pad_type_0"), val = string("valid")]; + int32 var_1863_groups_0 = const()[name = string("op_1863_groups_0"), val = int32(1)]; + tensor var_1863_strides_0 = const()[name = string("op_1863_strides_0"), val = tensor([1])]; + tensor var_1863_pad_0 = const()[name = string("op_1863_pad_0"), val = tensor([0, 0])]; + tensor var_1863_dilations_0 = const()[name = string("op_1863_dilations_0"), val = tensor([1])]; + tensor var_1848 = transpose(perm = var_1847, x = attn_output_25)[name = string("transpose_57")]; + tensor var_1863 = conv(dilations = var_1863_dilations_0, groups = var_1863_groups_0, pad = var_1863_pad_0, pad_type = var_1863_pad_type_0, strides = var_1863_strides_0, weight = squeeze_2_palettized, x = var_1848)[name = string("op_1863")]; + tensor var_1867 = const()[name = string("op_1867"), val = tensor([0, 2, 1])]; + tensor attn_output_29 = transpose(perm = var_1867, x = var_1863)[name = string("transpose_56")]; + tensor hidden_states_17_cast_fp16 = add(x = hidden_states_13_cast_fp16, y = attn_output_29)[name = string("hidden_states_17_cast_fp16")]; + int32 var_1880 = const()[name = string("op_1880"), val = int32(-1)]; + fp16 const_75_promoted_to_fp16 = const()[name = string("const_75_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1882_cast_fp16 = mul(x = hidden_states_17_cast_fp16, y = const_75_promoted_to_fp16)[name = string("op_1882_cast_fp16")]; + bool input_35_interleave_0 = const()[name = string("input_35_interleave_0"), val = bool(false)]; + tensor input_35_cast_fp16 = concat(axis = var_1880, interleave = input_35_interleave_0, values = (hidden_states_17_cast_fp16, var_1882_cast_fp16))[name = string("input_35_cast_fp16")]; + tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; + fp16 var_1877_to_fp16 = const()[name = string("op_1877_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_1877_to_fp16, x = input_35_cast_fp16)[name = string("normed_21_cast_fp16")]; + tensor normed_23_begin_0 = const()[name = string("normed_23_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_23_end_0 = const()[name = string("normed_23_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_23_end_mask_0 = const()[name = string("normed_23_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_23_cast_fp16 = slice_by_index(begin = normed_23_begin_0, end = normed_23_end_0, end_mask = normed_23_end_mask_0, x = normed_21_cast_fp16)[name = string("normed_23_cast_fp16")]; + tensor const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313733440)))]; + tensor x_45_cast_fp16 = mul(x = normed_23_cast_fp16, y = const_78_promoted_to_fp16)[name = string("x_45_cast_fp16")]; + tensor var_1907 = const()[name = string("op_1907"), val = tensor([0, 2, 1])]; + tensor input_37_axes_0 = const()[name = string("input_37_axes_0"), val = tensor([2])]; + tensor var_1908 = transpose(perm = var_1907, x = x_45_cast_fp16)[name = string("transpose_55")]; + tensor input_37 = expand_dims(axes = input_37_axes_0, x = var_1908)[name = string("input_37")]; + string input_39_pad_type_0 = const()[name = string("input_39_pad_type_0"), val = string("valid")]; + tensor input_39_strides_0 = const()[name = string("input_39_strides_0"), val = tensor([1, 1])]; + tensor input_39_pad_0 = const()[name = string("input_39_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_39_dilations_0 = const()[name = string("input_39_dilations_0"), val = tensor([1, 1])]; + int32 input_39_groups_0 = const()[name = string("input_39_groups_0"), val = int32(1)]; + tensor input_39 = conv(dilations = input_39_dilations_0, groups = input_39_groups_0, pad = input_39_pad_0, pad_type = input_39_pad_type_0, strides = input_39_strides_0, weight = model_model_layers_12_mlp_gate_proj_weight_palettized, x = input_37)[name = string("input_39")]; + string b_5_pad_type_0 = const()[name = string("b_5_pad_type_0"), val = string("valid")]; + tensor b_5_strides_0 = const()[name = string("b_5_strides_0"), val = tensor([1, 1])]; + tensor b_5_pad_0 = const()[name = string("b_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_5_dilations_0 = const()[name = string("b_5_dilations_0"), val = tensor([1, 1])]; + int32 b_5_groups_0 = const()[name = string("b_5_groups_0"), val = int32(1)]; + tensor b_5 = conv(dilations = b_5_dilations_0, groups = b_5_groups_0, pad = b_5_pad_0, pad_type = b_5_pad_type_0, strides = b_5_strides_0, weight = model_model_layers_12_mlp_up_proj_weight_palettized, x = input_37)[name = string("b_5")]; + tensor c_5 = silu(x = input_39)[name = string("c_5")]; + tensor input_41 = mul(x = c_5, y = b_5)[name = string("input_41")]; + string e_5_pad_type_0 = const()[name = string("e_5_pad_type_0"), val = string("valid")]; + tensor e_5_strides_0 = const()[name = string("e_5_strides_0"), val = tensor([1, 1])]; + tensor e_5_pad_0 = const()[name = string("e_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_5_dilations_0 = const()[name = string("e_5_dilations_0"), val = tensor([1, 1])]; + int32 e_5_groups_0 = const()[name = string("e_5_groups_0"), val = int32(1)]; + tensor e_5 = conv(dilations = e_5_dilations_0, groups = e_5_groups_0, pad = e_5_pad_0, pad_type = e_5_pad_type_0, strides = e_5_strides_0, weight = model_model_layers_12_mlp_down_proj_weight_palettized, x = input_41)[name = string("e_5")]; + tensor var_1930_axes_0 = const()[name = string("op_1930_axes_0"), val = tensor([2])]; + tensor var_1930 = squeeze(axes = var_1930_axes_0, x = e_5)[name = string("op_1930")]; + tensor var_1931 = const()[name = string("op_1931"), val = tensor([0, 2, 1])]; + tensor var_1932 = transpose(perm = var_1931, x = var_1930)[name = string("transpose_54")]; + tensor hidden_states_19_cast_fp16 = add(x = hidden_states_17_cast_fp16, y = var_1932)[name = string("hidden_states_19_cast_fp16")]; + int32 var_1944 = const()[name = string("op_1944"), val = int32(-1)]; + fp16 const_79_promoted_to_fp16 = const()[name = string("const_79_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1946_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_79_promoted_to_fp16)[name = string("op_1946_cast_fp16")]; + bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; + tensor input_43_cast_fp16 = concat(axis = var_1944, interleave = input_43_interleave_0, values = (hidden_states_19_cast_fp16, var_1946_cast_fp16))[name = string("input_43_cast_fp16")]; + tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; + fp16 var_1941_to_fp16 = const()[name = string("op_1941_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_1941_to_fp16, x = input_43_cast_fp16)[name = string("normed_25_cast_fp16")]; + tensor normed_27_begin_0 = const()[name = string("normed_27_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_27_end_0 = const()[name = string("normed_27_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_27_end_mask_0 = const()[name = string("normed_27_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_27_cast_fp16 = slice_by_index(begin = normed_27_begin_0, end = normed_27_end_0, end_mask = normed_27_end_mask_0, x = normed_25_cast_fp16)[name = string("normed_27_cast_fp16")]; + tensor const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313736576)))]; + tensor hidden_states_21_cast_fp16 = mul(x = normed_27_cast_fp16, y = const_82_promoted_to_fp16)[name = string("hidden_states_21_cast_fp16")]; + tensor var_1969 = const()[name = string("op_1969"), val = tensor([0, 2, 1])]; + tensor var_1972_axes_0 = const()[name = string("op_1972_axes_0"), val = tensor([2])]; + tensor var_1970_cast_fp16 = transpose(perm = var_1969, x = hidden_states_21_cast_fp16)[name = string("transpose_53")]; + tensor var_1972_cast_fp16 = expand_dims(axes = var_1972_axes_0, x = var_1970_cast_fp16)[name = string("op_1972_cast_fp16")]; + string query_states_19_pad_type_0 = const()[name = string("query_states_19_pad_type_0"), val = string("valid")]; + tensor query_states_19_strides_0 = const()[name = string("query_states_19_strides_0"), val = tensor([1, 1])]; + tensor query_states_19_pad_0 = const()[name = string("query_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_19_dilations_0 = const()[name = string("query_states_19_dilations_0"), val = tensor([1, 1])]; + int32 query_states_19_groups_0 = const()[name = string("query_states_19_groups_0"), val = int32(1)]; + tensor query_states_19 = conv(bias = model_model_layers_13_self_attn_q_proj_bias, dilations = query_states_19_dilations_0, groups = query_states_19_groups_0, pad = query_states_19_pad_0, pad_type = query_states_19_pad_type_0, strides = query_states_19_strides_0, weight = model_model_layers_13_self_attn_q_proj_weight_palettized, x = var_1972_cast_fp16)[name = string("query_states_19")]; + string key_states_25_pad_type_0 = const()[name = string("key_states_25_pad_type_0"), val = string("valid")]; + tensor key_states_25_strides_0 = const()[name = string("key_states_25_strides_0"), val = tensor([1, 1])]; + tensor key_states_25_pad_0 = const()[name = string("key_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_25_dilations_0 = const()[name = string("key_states_25_dilations_0"), val = tensor([1, 1])]; + int32 key_states_25_groups_0 = const()[name = string("key_states_25_groups_0"), val = int32(1)]; + tensor key_states_25 = conv(bias = model_model_layers_13_self_attn_k_proj_bias, dilations = key_states_25_dilations_0, groups = key_states_25_groups_0, pad = key_states_25_pad_0, pad_type = key_states_25_pad_type_0, strides = key_states_25_strides_0, weight = model_model_layers_13_self_attn_k_proj_weight_palettized, x = var_1972_cast_fp16)[name = string("key_states_25")]; + string value_states_25_pad_type_0 = const()[name = string("value_states_25_pad_type_0"), val = string("valid")]; + tensor value_states_25_strides_0 = const()[name = string("value_states_25_strides_0"), val = tensor([1, 1])]; + tensor value_states_25_pad_0 = const()[name = string("value_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_25_dilations_0 = const()[name = string("value_states_25_dilations_0"), val = tensor([1, 1])]; + int32 value_states_25_groups_0 = const()[name = string("value_states_25_groups_0"), val = int32(1)]; + tensor value_states_25 = conv(bias = model_model_layers_13_self_attn_v_proj_bias, dilations = value_states_25_dilations_0, groups = value_states_25_groups_0, pad = value_states_25_pad_0, pad_type = value_states_25_pad_type_0, strides = value_states_25_strides_0, weight = model_model_layers_13_self_attn_v_proj_weight_palettized, x = var_1972_cast_fp16)[name = string("value_states_25")]; + tensor var_2014 = const()[name = string("op_2014"), val = tensor([1, 12, 128, 64])]; + tensor var_2015 = reshape(shape = var_2014, x = query_states_19)[name = string("op_2015")]; + tensor var_2020 = const()[name = string("op_2020"), val = tensor([0, 1, 3, 2])]; + tensor var_2025 = const()[name = string("op_2025"), val = tensor([1, 2, 128, 64])]; + tensor var_2026 = reshape(shape = var_2025, x = key_states_25)[name = string("op_2026")]; + tensor var_2031 = const()[name = string("op_2031"), val = tensor([0, 1, 3, 2])]; + tensor var_2036 = const()[name = string("op_2036"), val = tensor([1, 2, 128, 64])]; + tensor var_2037 = reshape(shape = var_2036, x = value_states_25)[name = string("op_2037")]; + tensor var_2042 = const()[name = string("op_2042"), val = tensor([0, 1, 3, 2])]; + tensor q_13 = transpose(perm = var_2020, x = var_2015)[name = string("transpose_52")]; + tensor var_2056 = mul(x = q_13, y = cos_5)[name = string("op_2056")]; + tensor x1_13_begin_0 = const()[name = string("x1_13_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_13_end_0 = const()[name = string("x1_13_end_0"), val = tensor([1, 12, 64, 64])]; + tensor x1_13_end_mask_0 = const()[name = string("x1_13_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_13 = slice_by_index(begin = x1_13_begin_0, end = x1_13_end_0, end_mask = x1_13_end_mask_0, x = q_13)[name = string("x1_13")]; + tensor x2_13_begin_0 = const()[name = string("x2_13_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_13_end_0 = const()[name = string("x2_13_end_0"), val = tensor([1, 12, 64, 128])]; + tensor x2_13_end_mask_0 = const()[name = string("x2_13_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_13 = slice_by_index(begin = x2_13_begin_0, end = x2_13_end_0, end_mask = x2_13_end_mask_0, x = q_13)[name = string("x2_13")]; + fp16 const_86_promoted = const()[name = string("const_86_promoted"), val = fp16(-0x1p+0)]; + tensor var_2077 = mul(x = x2_13, y = const_86_promoted)[name = string("op_2077")]; + int32 var_2079 = const()[name = string("op_2079"), val = int32(-1)]; + bool var_2080_interleave_0 = const()[name = string("op_2080_interleave_0"), val = bool(false)]; + tensor var_2080 = concat(axis = var_2079, interleave = var_2080_interleave_0, values = (var_2077, x1_13))[name = string("op_2080")]; + tensor var_2081 = mul(x = var_2080, y = sin_5)[name = string("op_2081")]; + tensor query_states_21 = add(x = var_2056, y = var_2081)[name = string("query_states_21")]; + tensor k_13 = transpose(perm = var_2031, x = var_2026)[name = string("transpose_51")]; + tensor var_2084 = mul(x = k_13, y = cos_5)[name = string("op_2084")]; + tensor x1_15_begin_0 = const()[name = string("x1_15_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_15_end_0 = const()[name = string("x1_15_end_0"), val = tensor([1, 2, 64, 64])]; + tensor x1_15_end_mask_0 = const()[name = string("x1_15_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_15 = slice_by_index(begin = x1_15_begin_0, end = x1_15_end_0, end_mask = x1_15_end_mask_0, x = k_13)[name = string("x1_15")]; + tensor x2_15_begin_0 = const()[name = string("x2_15_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_15_end_0 = const()[name = string("x2_15_end_0"), val = tensor([1, 2, 64, 128])]; + tensor x2_15_end_mask_0 = const()[name = string("x2_15_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_15 = slice_by_index(begin = x2_15_begin_0, end = x2_15_end_0, end_mask = x2_15_end_mask_0, x = k_13)[name = string("x2_15")]; + fp16 const_89_promoted = const()[name = string("const_89_promoted"), val = fp16(-0x1p+0)]; + tensor var_2105 = mul(x = x2_15, y = const_89_promoted)[name = string("op_2105")]; + int32 var_2107 = const()[name = string("op_2107"), val = int32(-1)]; + bool var_2108_interleave_0 = const()[name = string("op_2108_interleave_0"), val = bool(false)]; + tensor var_2108 = concat(axis = var_2107, interleave = var_2108_interleave_0, values = (var_2105, x1_15))[name = string("op_2108")]; + tensor var_2109 = mul(x = var_2108, y = sin_5)[name = string("op_2109")]; + tensor key_states_27 = add(x = var_2084, y = var_2109)[name = string("key_states_27")]; + tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([13])]; + tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; + tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; + tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([14])]; + int32 concat_56_axis_0 = const()[name = string("concat_56_axis_0"), val = int32(0)]; + bool concat_56_interleave_0 = const()[name = string("concat_56_interleave_0"), val = bool(false)]; + tensor concat_56 = concat(axis = concat_56_axis_0, interleave = concat_56_interleave_0, values = (expand_dims_36, expand_dims_37, current_pos, expand_dims_39))[name = string("concat_56")]; + tensor concat_57_values1_0 = const()[name = string("concat_57_values1_0"), val = tensor([0])]; + tensor concat_57_values3_0 = const()[name = string("concat_57_values3_0"), val = tensor([0])]; + int32 concat_57_axis_0 = const()[name = string("concat_57_axis_0"), val = int32(0)]; + bool concat_57_interleave_0 = const()[name = string("concat_57_interleave_0"), val = bool(false)]; + tensor concat_57 = concat(axis = concat_57_axis_0, interleave = concat_57_interleave_0, values = (expand_dims_40, concat_57_values1_0, var_616, concat_57_values3_0))[name = string("concat_57")]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_56, begin_mask = model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0, end = concat_57, end_mask = model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_7_stride_0, update = key_states_27, x = coreml_update_state_23)[name = string("model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_24_write_state")]; + tensor coreml_update_state_24 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_24")]; + tensor expand_dims_42 = const()[name = string("expand_dims_42"), val = tensor([41])]; + tensor expand_dims_43 = const()[name = string("expand_dims_43"), val = tensor([0])]; + tensor expand_dims_45 = const()[name = string("expand_dims_45"), val = tensor([0])]; + tensor expand_dims_46 = const()[name = string("expand_dims_46"), val = tensor([42])]; + int32 concat_60_axis_0 = const()[name = string("concat_60_axis_0"), val = int32(0)]; + bool concat_60_interleave_0 = const()[name = string("concat_60_interleave_0"), val = bool(false)]; + tensor concat_60 = concat(axis = concat_60_axis_0, interleave = concat_60_interleave_0, values = (expand_dims_42, expand_dims_43, current_pos, expand_dims_45))[name = string("concat_60")]; + tensor concat_61_values1_0 = const()[name = string("concat_61_values1_0"), val = tensor([0])]; + tensor concat_61_values3_0 = const()[name = string("concat_61_values3_0"), val = tensor([0])]; + int32 concat_61_axis_0 = const()[name = string("concat_61_axis_0"), val = int32(0)]; + bool concat_61_interleave_0 = const()[name = string("concat_61_interleave_0"), val = bool(false)]; + tensor concat_61 = concat(axis = concat_61_axis_0, interleave = concat_61_interleave_0, values = (expand_dims_46, concat_61_values1_0, var_616, concat_61_values3_0))[name = string("concat_61")]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_27 = transpose(perm = var_2042, x = var_2037)[name = string("transpose_50")]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_60, begin_mask = model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0, end = concat_61, end_mask = model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_8_stride_0, update = value_states_27, x = coreml_update_state_24)[name = string("model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_25_write_state")]; + tensor coreml_update_state_25 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_25")]; + tensor var_2180_begin_0 = const()[name = string("op_2180_begin_0"), val = tensor([13, 0, 0, 0])]; + tensor var_2180_end_0 = const()[name = string("op_2180_end_0"), val = tensor([14, 2, 2048, 128])]; + tensor var_2180_end_mask_0 = const()[name = string("op_2180_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2180_cast_fp16 = slice_by_index(begin = var_2180_begin_0, end = var_2180_end_0, end_mask = var_2180_end_mask_0, x = coreml_update_state_25)[name = string("op_2180_cast_fp16")]; + tensor K_layer_cache_7_axes_0 = const()[name = string("K_layer_cache_7_axes_0"), val = tensor([0])]; + tensor K_layer_cache_7_cast_fp16 = squeeze(axes = K_layer_cache_7_axes_0, x = var_2180_cast_fp16)[name = string("K_layer_cache_7_cast_fp16")]; + tensor var_2187_begin_0 = const()[name = string("op_2187_begin_0"), val = tensor([41, 0, 0, 0])]; + tensor var_2187_end_0 = const()[name = string("op_2187_end_0"), val = tensor([42, 2, 2048, 128])]; + tensor var_2187_end_mask_0 = const()[name = string("op_2187_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2187_cast_fp16 = slice_by_index(begin = var_2187_begin_0, end = var_2187_end_0, end_mask = var_2187_end_mask_0, x = coreml_update_state_25)[name = string("op_2187_cast_fp16")]; + tensor V_layer_cache_7_axes_0 = const()[name = string("V_layer_cache_7_axes_0"), val = tensor([0])]; + tensor V_layer_cache_7_cast_fp16 = squeeze(axes = V_layer_cache_7_axes_0, x = var_2187_cast_fp16)[name = string("V_layer_cache_7_cast_fp16")]; + tensor x_51_axes_0 = const()[name = string("x_51_axes_0"), val = tensor([1])]; + tensor x_51_cast_fp16 = expand_dims(axes = x_51_axes_0, x = K_layer_cache_7_cast_fp16)[name = string("x_51_cast_fp16")]; + tensor var_2216 = const()[name = string("op_2216"), val = tensor([1, 6, 1, 1])]; + tensor x_53_cast_fp16 = tile(reps = var_2216, x = x_51_cast_fp16)[name = string("x_53_cast_fp16")]; + tensor var_2228 = const()[name = string("op_2228"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_31_cast_fp16 = reshape(shape = var_2228, x = x_53_cast_fp16)[name = string("key_states_31_cast_fp16")]; + tensor x_57_axes_0 = const()[name = string("x_57_axes_0"), val = tensor([1])]; + tensor x_57_cast_fp16 = expand_dims(axes = x_57_axes_0, x = V_layer_cache_7_cast_fp16)[name = string("x_57_cast_fp16")]; + tensor var_2236 = const()[name = string("op_2236"), val = tensor([1, 6, 1, 1])]; + tensor x_59_cast_fp16 = tile(reps = var_2236, x = x_57_cast_fp16)[name = string("x_59_cast_fp16")]; + bool var_2271_transpose_x_1 = const()[name = string("op_2271_transpose_x_1"), val = bool(false)]; + bool var_2271_transpose_y_1 = const()[name = string("op_2271_transpose_y_1"), val = bool(true)]; + tensor var_2271_cast_fp16 = matmul(transpose_x = var_2271_transpose_x_1, transpose_y = var_2271_transpose_y_1, x = query_states_21, y = key_states_31_cast_fp16)[name = string("op_2271_cast_fp16")]; + fp16 var_2272_to_fp16 = const()[name = string("op_2272_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_13_cast_fp16 = mul(x = var_2271_cast_fp16, y = var_2272_to_fp16)[name = string("attn_logits_13_cast_fp16")]; + tensor attn_logits_15_cast_fp16 = add(x = attn_logits_13_cast_fp16, y = causal_mask)[name = string("attn_logits_15_cast_fp16")]; + int32 var_2299 = const()[name = string("op_2299"), val = int32(-1)]; + tensor var_2301_cast_fp16 = softmax(axis = var_2299, x = attn_logits_15_cast_fp16)[name = string("op_2301_cast_fp16")]; + tensor concat_66 = const()[name = string("concat_66"), val = tensor([12, 64, 2048])]; + tensor reshape_9_cast_fp16 = reshape(shape = concat_66, x = var_2301_cast_fp16)[name = string("reshape_9_cast_fp16")]; + tensor concat_67 = const()[name = string("concat_67"), val = tensor([12, 2048, 128])]; + tensor reshape_10_cast_fp16 = reshape(shape = concat_67, x = x_59_cast_fp16)[name = string("reshape_10_cast_fp16")]; + bool matmul_3_transpose_x_0 = const()[name = string("matmul_3_transpose_x_0"), val = bool(false)]; + bool matmul_3_transpose_y_0 = const()[name = string("matmul_3_transpose_y_0"), val = bool(false)]; + tensor matmul_3_cast_fp16 = matmul(transpose_x = matmul_3_transpose_x_0, transpose_y = matmul_3_transpose_y_0, x = reshape_9_cast_fp16, y = reshape_10_cast_fp16)[name = string("matmul_3_cast_fp16")]; + tensor concat_71 = const()[name = string("concat_71"), val = tensor([1, 12, 64, 128])]; + tensor reshape_11_cast_fp16 = reshape(shape = concat_71, x = matmul_3_cast_fp16)[name = string("reshape_11_cast_fp16")]; + tensor var_2328_perm_0 = const()[name = string("op_2328_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2347 = const()[name = string("op_2347"), val = tensor([1, 64, 1536])]; + tensor var_2328 = transpose(perm = var_2328_perm_0, x = reshape_11_cast_fp16)[name = string("transpose_49")]; + tensor attn_output_35 = reshape(shape = var_2347, x = var_2328)[name = string("attn_output_35")]; + tensor var_2352 = const()[name = string("op_2352"), val = tensor([0, 2, 1])]; + tensor squeeze_3_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(313739712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315509248))))[name = string("squeeze_3_palettized")]; + string var_2368_pad_type_0 = const()[name = string("op_2368_pad_type_0"), val = string("valid")]; + int32 var_2368_groups_0 = const()[name = string("op_2368_groups_0"), val = int32(1)]; + tensor var_2368_strides_0 = const()[name = string("op_2368_strides_0"), val = tensor([1])]; + tensor var_2368_pad_0 = const()[name = string("op_2368_pad_0"), val = tensor([0, 0])]; + tensor var_2368_dilations_0 = const()[name = string("op_2368_dilations_0"), val = tensor([1])]; + tensor var_2353 = transpose(perm = var_2352, x = attn_output_35)[name = string("transpose_48")]; + tensor var_2368 = conv(dilations = var_2368_dilations_0, groups = var_2368_groups_0, pad = var_2368_pad_0, pad_type = var_2368_pad_type_0, strides = var_2368_strides_0, weight = squeeze_3_palettized, x = var_2353)[name = string("op_2368")]; + tensor var_2372 = const()[name = string("op_2372"), val = tensor([0, 2, 1])]; + tensor attn_output_39 = transpose(perm = var_2372, x = var_2368)[name = string("transpose_47")]; + tensor hidden_states_23_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = attn_output_39)[name = string("hidden_states_23_cast_fp16")]; + int32 var_2385 = const()[name = string("op_2385"), val = int32(-1)]; + fp16 const_101_promoted_to_fp16 = const()[name = string("const_101_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2387_cast_fp16 = mul(x = hidden_states_23_cast_fp16, y = const_101_promoted_to_fp16)[name = string("op_2387_cast_fp16")]; + bool input_49_interleave_0 = const()[name = string("input_49_interleave_0"), val = bool(false)]; + tensor input_49_cast_fp16 = concat(axis = var_2385, interleave = input_49_interleave_0, values = (hidden_states_23_cast_fp16, var_2387_cast_fp16))[name = string("input_49_cast_fp16")]; + tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; + fp16 var_2382_to_fp16 = const()[name = string("op_2382_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_2382_to_fp16, x = input_49_cast_fp16)[name = string("normed_29_cast_fp16")]; + tensor normed_31_begin_0 = const()[name = string("normed_31_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_31_end_0 = const()[name = string("normed_31_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_31_end_mask_0 = const()[name = string("normed_31_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_31_cast_fp16 = slice_by_index(begin = normed_31_begin_0, end = normed_31_end_0, end_mask = normed_31_end_mask_0, x = normed_29_cast_fp16)[name = string("normed_31_cast_fp16")]; + tensor const_104_promoted_to_fp16 = const()[name = string("const_104_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315558464)))]; + tensor x_61_cast_fp16 = mul(x = normed_31_cast_fp16, y = const_104_promoted_to_fp16)[name = string("x_61_cast_fp16")]; + tensor var_2412 = const()[name = string("op_2412"), val = tensor([0, 2, 1])]; + tensor input_51_axes_0 = const()[name = string("input_51_axes_0"), val = tensor([2])]; + tensor var_2413 = transpose(perm = var_2412, x = x_61_cast_fp16)[name = string("transpose_46")]; + tensor input_51 = expand_dims(axes = input_51_axes_0, x = var_2413)[name = string("input_51")]; + string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; + tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; + tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; + int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; + tensor input_53 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = model_model_layers_13_mlp_gate_proj_weight_palettized, x = input_51)[name = string("input_53")]; + string b_7_pad_type_0 = const()[name = string("b_7_pad_type_0"), val = string("valid")]; + tensor b_7_strides_0 = const()[name = string("b_7_strides_0"), val = tensor([1, 1])]; + tensor b_7_pad_0 = const()[name = string("b_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_7_dilations_0 = const()[name = string("b_7_dilations_0"), val = tensor([1, 1])]; + int32 b_7_groups_0 = const()[name = string("b_7_groups_0"), val = int32(1)]; + tensor b_7 = conv(dilations = b_7_dilations_0, groups = b_7_groups_0, pad = b_7_pad_0, pad_type = b_7_pad_type_0, strides = b_7_strides_0, weight = model_model_layers_13_mlp_up_proj_weight_palettized, x = input_51)[name = string("b_7")]; + tensor c_7 = silu(x = input_53)[name = string("c_7")]; + tensor input_55 = mul(x = c_7, y = b_7)[name = string("input_55")]; + string e_7_pad_type_0 = const()[name = string("e_7_pad_type_0"), val = string("valid")]; + tensor e_7_strides_0 = const()[name = string("e_7_strides_0"), val = tensor([1, 1])]; + tensor e_7_pad_0 = const()[name = string("e_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_7_dilations_0 = const()[name = string("e_7_dilations_0"), val = tensor([1, 1])]; + int32 e_7_groups_0 = const()[name = string("e_7_groups_0"), val = int32(1)]; + tensor e_7 = conv(dilations = e_7_dilations_0, groups = e_7_groups_0, pad = e_7_pad_0, pad_type = e_7_pad_type_0, strides = e_7_strides_0, weight = model_model_layers_13_mlp_down_proj_weight_palettized, x = input_55)[name = string("e_7")]; + tensor var_2435_axes_0 = const()[name = string("op_2435_axes_0"), val = tensor([2])]; + tensor var_2435 = squeeze(axes = var_2435_axes_0, x = e_7)[name = string("op_2435")]; + tensor var_2436 = const()[name = string("op_2436"), val = tensor([0, 2, 1])]; + tensor var_2437 = transpose(perm = var_2436, x = var_2435)[name = string("transpose_45")]; + tensor hidden_states_25_cast_fp16 = add(x = hidden_states_23_cast_fp16, y = var_2437)[name = string("hidden_states_25_cast_fp16")]; + int32 var_2449 = const()[name = string("op_2449"), val = int32(-1)]; + fp16 const_105_promoted_to_fp16 = const()[name = string("const_105_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2451_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_105_promoted_to_fp16)[name = string("op_2451_cast_fp16")]; + bool input_57_interleave_0 = const()[name = string("input_57_interleave_0"), val = bool(false)]; + tensor input_57_cast_fp16 = concat(axis = var_2449, interleave = input_57_interleave_0, values = (hidden_states_25_cast_fp16, var_2451_cast_fp16))[name = string("input_57_cast_fp16")]; + tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; + fp16 var_2446_to_fp16 = const()[name = string("op_2446_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_2446_to_fp16, x = input_57_cast_fp16)[name = string("normed_33_cast_fp16")]; + tensor normed_35_begin_0 = const()[name = string("normed_35_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_35_end_0 = const()[name = string("normed_35_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_35_end_mask_0 = const()[name = string("normed_35_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_35_cast_fp16 = slice_by_index(begin = normed_35_begin_0, end = normed_35_end_0, end_mask = normed_35_end_mask_0, x = normed_33_cast_fp16)[name = string("normed_35_cast_fp16")]; + tensor const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315561600)))]; + tensor hidden_states_27_cast_fp16 = mul(x = normed_35_cast_fp16, y = const_108_promoted_to_fp16)[name = string("hidden_states_27_cast_fp16")]; + tensor var_2474 = const()[name = string("op_2474"), val = tensor([0, 2, 1])]; + tensor var_2477_axes_0 = const()[name = string("op_2477_axes_0"), val = tensor([2])]; + tensor var_2475_cast_fp16 = transpose(perm = var_2474, x = hidden_states_27_cast_fp16)[name = string("transpose_44")]; + tensor var_2477_cast_fp16 = expand_dims(axes = var_2477_axes_0, x = var_2475_cast_fp16)[name = string("op_2477_cast_fp16")]; + string query_states_25_pad_type_0 = const()[name = string("query_states_25_pad_type_0"), val = string("valid")]; + tensor query_states_25_strides_0 = const()[name = string("query_states_25_strides_0"), val = tensor([1, 1])]; + tensor query_states_25_pad_0 = const()[name = string("query_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_25_dilations_0 = const()[name = string("query_states_25_dilations_0"), val = tensor([1, 1])]; + int32 query_states_25_groups_0 = const()[name = string("query_states_25_groups_0"), val = int32(1)]; + tensor query_states_25 = conv(bias = model_model_layers_14_self_attn_q_proj_bias, dilations = query_states_25_dilations_0, groups = query_states_25_groups_0, pad = query_states_25_pad_0, pad_type = query_states_25_pad_type_0, strides = query_states_25_strides_0, weight = model_model_layers_14_self_attn_q_proj_weight_palettized, x = var_2477_cast_fp16)[name = string("query_states_25")]; + string key_states_33_pad_type_0 = const()[name = string("key_states_33_pad_type_0"), val = string("valid")]; + tensor key_states_33_strides_0 = const()[name = string("key_states_33_strides_0"), val = tensor([1, 1])]; + tensor key_states_33_pad_0 = const()[name = string("key_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_33_dilations_0 = const()[name = string("key_states_33_dilations_0"), val = tensor([1, 1])]; + int32 key_states_33_groups_0 = const()[name = string("key_states_33_groups_0"), val = int32(1)]; + tensor key_states_33 = conv(bias = model_model_layers_14_self_attn_k_proj_bias, dilations = key_states_33_dilations_0, groups = key_states_33_groups_0, pad = key_states_33_pad_0, pad_type = key_states_33_pad_type_0, strides = key_states_33_strides_0, weight = model_model_layers_14_self_attn_k_proj_weight_palettized, x = var_2477_cast_fp16)[name = string("key_states_33")]; + string value_states_33_pad_type_0 = const()[name = string("value_states_33_pad_type_0"), val = string("valid")]; + tensor value_states_33_strides_0 = const()[name = string("value_states_33_strides_0"), val = tensor([1, 1])]; + tensor value_states_33_pad_0 = const()[name = string("value_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_33_dilations_0 = const()[name = string("value_states_33_dilations_0"), val = tensor([1, 1])]; + int32 value_states_33_groups_0 = const()[name = string("value_states_33_groups_0"), val = int32(1)]; + tensor value_states_33 = conv(bias = model_model_layers_14_self_attn_v_proj_bias, dilations = value_states_33_dilations_0, groups = value_states_33_groups_0, pad = value_states_33_pad_0, pad_type = value_states_33_pad_type_0, strides = value_states_33_strides_0, weight = model_model_layers_14_self_attn_v_proj_weight_palettized, x = var_2477_cast_fp16)[name = string("value_states_33")]; + tensor var_2519 = const()[name = string("op_2519"), val = tensor([1, 12, 128, 64])]; + tensor var_2520 = reshape(shape = var_2519, x = query_states_25)[name = string("op_2520")]; + tensor var_2525 = const()[name = string("op_2525"), val = tensor([0, 1, 3, 2])]; + tensor var_2530 = const()[name = string("op_2530"), val = tensor([1, 2, 128, 64])]; + tensor var_2531 = reshape(shape = var_2530, x = key_states_33)[name = string("op_2531")]; + tensor var_2536 = const()[name = string("op_2536"), val = tensor([0, 1, 3, 2])]; + tensor var_2541 = const()[name = string("op_2541"), val = tensor([1, 2, 128, 64])]; + tensor var_2542 = reshape(shape = var_2541, x = value_states_33)[name = string("op_2542")]; + tensor var_2547 = const()[name = string("op_2547"), val = tensor([0, 1, 3, 2])]; + tensor q_17 = transpose(perm = var_2525, x = var_2520)[name = string("transpose_43")]; + tensor var_2561 = mul(x = q_17, y = cos_5)[name = string("op_2561")]; + tensor x1_17_begin_0 = const()[name = string("x1_17_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_17_end_0 = const()[name = string("x1_17_end_0"), val = tensor([1, 12, 64, 64])]; + tensor x1_17_end_mask_0 = const()[name = string("x1_17_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_17 = slice_by_index(begin = x1_17_begin_0, end = x1_17_end_0, end_mask = x1_17_end_mask_0, x = q_17)[name = string("x1_17")]; + tensor x2_17_begin_0 = const()[name = string("x2_17_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_17_end_0 = const()[name = string("x2_17_end_0"), val = tensor([1, 12, 64, 128])]; + tensor x2_17_end_mask_0 = const()[name = string("x2_17_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_17 = slice_by_index(begin = x2_17_begin_0, end = x2_17_end_0, end_mask = x2_17_end_mask_0, x = q_17)[name = string("x2_17")]; + fp16 const_112_promoted = const()[name = string("const_112_promoted"), val = fp16(-0x1p+0)]; + tensor var_2582 = mul(x = x2_17, y = const_112_promoted)[name = string("op_2582")]; + int32 var_2584 = const()[name = string("op_2584"), val = int32(-1)]; + bool var_2585_interleave_0 = const()[name = string("op_2585_interleave_0"), val = bool(false)]; + tensor var_2585 = concat(axis = var_2584, interleave = var_2585_interleave_0, values = (var_2582, x1_17))[name = string("op_2585")]; + tensor var_2586 = mul(x = var_2585, y = sin_5)[name = string("op_2586")]; + tensor query_states_27 = add(x = var_2561, y = var_2586)[name = string("query_states_27")]; + tensor k_17 = transpose(perm = var_2536, x = var_2531)[name = string("transpose_42")]; + tensor var_2589 = mul(x = k_17, y = cos_5)[name = string("op_2589")]; + tensor x1_19_begin_0 = const()[name = string("x1_19_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_19_end_0 = const()[name = string("x1_19_end_0"), val = tensor([1, 2, 64, 64])]; + tensor x1_19_end_mask_0 = const()[name = string("x1_19_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_19 = slice_by_index(begin = x1_19_begin_0, end = x1_19_end_0, end_mask = x1_19_end_mask_0, x = k_17)[name = string("x1_19")]; + tensor x2_19_begin_0 = const()[name = string("x2_19_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_19_end_0 = const()[name = string("x2_19_end_0"), val = tensor([1, 2, 64, 128])]; + tensor x2_19_end_mask_0 = const()[name = string("x2_19_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_19 = slice_by_index(begin = x2_19_begin_0, end = x2_19_end_0, end_mask = x2_19_end_mask_0, x = k_17)[name = string("x2_19")]; + fp16 const_115_promoted = const()[name = string("const_115_promoted"), val = fp16(-0x1p+0)]; + tensor var_2610 = mul(x = x2_19, y = const_115_promoted)[name = string("op_2610")]; + int32 var_2612 = const()[name = string("op_2612"), val = int32(-1)]; + bool var_2613_interleave_0 = const()[name = string("op_2613_interleave_0"), val = bool(false)]; + tensor var_2613 = concat(axis = var_2612, interleave = var_2613_interleave_0, values = (var_2610, x1_19))[name = string("op_2613")]; + tensor var_2614 = mul(x = var_2613, y = sin_5)[name = string("op_2614")]; + tensor key_states_35 = add(x = var_2589, y = var_2614)[name = string("key_states_35")]; + tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([14])]; + tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; + tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; + tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([15])]; + int32 concat_74_axis_0 = const()[name = string("concat_74_axis_0"), val = int32(0)]; + bool concat_74_interleave_0 = const()[name = string("concat_74_interleave_0"), val = bool(false)]; + tensor concat_74 = concat(axis = concat_74_axis_0, interleave = concat_74_interleave_0, values = (expand_dims_48, expand_dims_49, current_pos, expand_dims_51))[name = string("concat_74")]; + tensor concat_75_values1_0 = const()[name = string("concat_75_values1_0"), val = tensor([0])]; + tensor concat_75_values3_0 = const()[name = string("concat_75_values3_0"), val = tensor([0])]; + int32 concat_75_axis_0 = const()[name = string("concat_75_axis_0"), val = int32(0)]; + bool concat_75_interleave_0 = const()[name = string("concat_75_interleave_0"), val = bool(false)]; + tensor concat_75 = concat(axis = concat_75_axis_0, interleave = concat_75_interleave_0, values = (expand_dims_52, concat_75_values1_0, var_616, concat_75_values3_0))[name = string("concat_75")]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_74, begin_mask = model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0, end = concat_75, end_mask = model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_9_stride_0, update = key_states_35, x = coreml_update_state_25)[name = string("model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_26_write_state")]; + tensor coreml_update_state_26 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_26")]; + tensor expand_dims_54 = const()[name = string("expand_dims_54"), val = tensor([42])]; + tensor expand_dims_55 = const()[name = string("expand_dims_55"), val = tensor([0])]; + tensor expand_dims_57 = const()[name = string("expand_dims_57"), val = tensor([0])]; + tensor expand_dims_58 = const()[name = string("expand_dims_58"), val = tensor([43])]; + int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; + bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; + tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_54, expand_dims_55, current_pos, expand_dims_57))[name = string("concat_78")]; + tensor concat_79_values1_0 = const()[name = string("concat_79_values1_0"), val = tensor([0])]; + tensor concat_79_values3_0 = const()[name = string("concat_79_values3_0"), val = tensor([0])]; + int32 concat_79_axis_0 = const()[name = string("concat_79_axis_0"), val = int32(0)]; + bool concat_79_interleave_0 = const()[name = string("concat_79_interleave_0"), val = bool(false)]; + tensor concat_79 = concat(axis = concat_79_axis_0, interleave = concat_79_interleave_0, values = (expand_dims_58, concat_79_values1_0, var_616, concat_79_values3_0))[name = string("concat_79")]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_35 = transpose(perm = var_2547, x = var_2542)[name = string("transpose_41")]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_78, begin_mask = model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0, end = concat_79, end_mask = model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_10_stride_0, update = value_states_35, x = coreml_update_state_26)[name = string("model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_27_write_state")]; + tensor coreml_update_state_27 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_27")]; + tensor var_2685_begin_0 = const()[name = string("op_2685_begin_0"), val = tensor([14, 0, 0, 0])]; + tensor var_2685_end_0 = const()[name = string("op_2685_end_0"), val = tensor([15, 2, 2048, 128])]; + tensor var_2685_end_mask_0 = const()[name = string("op_2685_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2685_cast_fp16 = slice_by_index(begin = var_2685_begin_0, end = var_2685_end_0, end_mask = var_2685_end_mask_0, x = coreml_update_state_27)[name = string("op_2685_cast_fp16")]; + tensor K_layer_cache_9_axes_0 = const()[name = string("K_layer_cache_9_axes_0"), val = tensor([0])]; + tensor K_layer_cache_9_cast_fp16 = squeeze(axes = K_layer_cache_9_axes_0, x = var_2685_cast_fp16)[name = string("K_layer_cache_9_cast_fp16")]; + tensor var_2692_begin_0 = const()[name = string("op_2692_begin_0"), val = tensor([42, 0, 0, 0])]; + tensor var_2692_end_0 = const()[name = string("op_2692_end_0"), val = tensor([43, 2, 2048, 128])]; + tensor var_2692_end_mask_0 = const()[name = string("op_2692_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2692_cast_fp16 = slice_by_index(begin = var_2692_begin_0, end = var_2692_end_0, end_mask = var_2692_end_mask_0, x = coreml_update_state_27)[name = string("op_2692_cast_fp16")]; + tensor V_layer_cache_9_axes_0 = const()[name = string("V_layer_cache_9_axes_0"), val = tensor([0])]; + tensor V_layer_cache_9_cast_fp16 = squeeze(axes = V_layer_cache_9_axes_0, x = var_2692_cast_fp16)[name = string("V_layer_cache_9_cast_fp16")]; + tensor x_67_axes_0 = const()[name = string("x_67_axes_0"), val = tensor([1])]; + tensor x_67_cast_fp16 = expand_dims(axes = x_67_axes_0, x = K_layer_cache_9_cast_fp16)[name = string("x_67_cast_fp16")]; + tensor var_2721 = const()[name = string("op_2721"), val = tensor([1, 6, 1, 1])]; + tensor x_69_cast_fp16 = tile(reps = var_2721, x = x_67_cast_fp16)[name = string("x_69_cast_fp16")]; + tensor var_2733 = const()[name = string("op_2733"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_39_cast_fp16 = reshape(shape = var_2733, x = x_69_cast_fp16)[name = string("key_states_39_cast_fp16")]; + tensor x_73_axes_0 = const()[name = string("x_73_axes_0"), val = tensor([1])]; + tensor x_73_cast_fp16 = expand_dims(axes = x_73_axes_0, x = V_layer_cache_9_cast_fp16)[name = string("x_73_cast_fp16")]; + tensor var_2741 = const()[name = string("op_2741"), val = tensor([1, 6, 1, 1])]; + tensor x_75_cast_fp16 = tile(reps = var_2741, x = x_73_cast_fp16)[name = string("x_75_cast_fp16")]; + bool var_2776_transpose_x_1 = const()[name = string("op_2776_transpose_x_1"), val = bool(false)]; + bool var_2776_transpose_y_1 = const()[name = string("op_2776_transpose_y_1"), val = bool(true)]; + tensor var_2776_cast_fp16 = matmul(transpose_x = var_2776_transpose_x_1, transpose_y = var_2776_transpose_y_1, x = query_states_27, y = key_states_39_cast_fp16)[name = string("op_2776_cast_fp16")]; + fp16 var_2777_to_fp16 = const()[name = string("op_2777_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_17_cast_fp16 = mul(x = var_2776_cast_fp16, y = var_2777_to_fp16)[name = string("attn_logits_17_cast_fp16")]; + tensor attn_logits_19_cast_fp16 = add(x = attn_logits_17_cast_fp16, y = causal_mask)[name = string("attn_logits_19_cast_fp16")]; + int32 var_2804 = const()[name = string("op_2804"), val = int32(-1)]; + tensor var_2806_cast_fp16 = softmax(axis = var_2804, x = attn_logits_19_cast_fp16)[name = string("op_2806_cast_fp16")]; + tensor concat_84 = const()[name = string("concat_84"), val = tensor([12, 64, 2048])]; + tensor reshape_12_cast_fp16 = reshape(shape = concat_84, x = var_2806_cast_fp16)[name = string("reshape_12_cast_fp16")]; + tensor concat_85 = const()[name = string("concat_85"), val = tensor([12, 2048, 128])]; + tensor reshape_13_cast_fp16 = reshape(shape = concat_85, x = x_75_cast_fp16)[name = string("reshape_13_cast_fp16")]; + bool matmul_4_transpose_x_0 = const()[name = string("matmul_4_transpose_x_0"), val = bool(false)]; + bool matmul_4_transpose_y_0 = const()[name = string("matmul_4_transpose_y_0"), val = bool(false)]; + tensor matmul_4_cast_fp16 = matmul(transpose_x = matmul_4_transpose_x_0, transpose_y = matmul_4_transpose_y_0, x = reshape_12_cast_fp16, y = reshape_13_cast_fp16)[name = string("matmul_4_cast_fp16")]; + tensor concat_89 = const()[name = string("concat_89"), val = tensor([1, 12, 64, 128])]; + tensor reshape_14_cast_fp16 = reshape(shape = concat_89, x = matmul_4_cast_fp16)[name = string("reshape_14_cast_fp16")]; + tensor var_2833_perm_0 = const()[name = string("op_2833_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2852 = const()[name = string("op_2852"), val = tensor([1, 64, 1536])]; + tensor var_2833 = transpose(perm = var_2833_perm_0, x = reshape_14_cast_fp16)[name = string("transpose_40")]; + tensor attn_output_45 = reshape(shape = var_2852, x = var_2833)[name = string("attn_output_45")]; + tensor var_2857 = const()[name = string("op_2857"), val = tensor([0, 2, 1])]; + tensor squeeze_4_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315564736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(317334272))))[name = string("squeeze_4_palettized")]; + string var_2873_pad_type_0 = const()[name = string("op_2873_pad_type_0"), val = string("valid")]; + int32 var_2873_groups_0 = const()[name = string("op_2873_groups_0"), val = int32(1)]; + tensor var_2873_strides_0 = const()[name = string("op_2873_strides_0"), val = tensor([1])]; + tensor var_2873_pad_0 = const()[name = string("op_2873_pad_0"), val = tensor([0, 0])]; + tensor var_2873_dilations_0 = const()[name = string("op_2873_dilations_0"), val = tensor([1])]; + tensor var_2858 = transpose(perm = var_2857, x = attn_output_45)[name = string("transpose_39")]; + tensor var_2873 = conv(dilations = var_2873_dilations_0, groups = var_2873_groups_0, pad = var_2873_pad_0, pad_type = var_2873_pad_type_0, strides = var_2873_strides_0, weight = squeeze_4_palettized, x = var_2858)[name = string("op_2873")]; + tensor var_2877 = const()[name = string("op_2877"), val = tensor([0, 2, 1])]; + tensor attn_output_49 = transpose(perm = var_2877, x = var_2873)[name = string("transpose_38")]; + tensor hidden_states_29_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = attn_output_49)[name = string("hidden_states_29_cast_fp16")]; + int32 var_2890 = const()[name = string("op_2890"), val = int32(-1)]; + fp16 const_127_promoted_to_fp16 = const()[name = string("const_127_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2892_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_127_promoted_to_fp16)[name = string("op_2892_cast_fp16")]; + bool input_63_interleave_0 = const()[name = string("input_63_interleave_0"), val = bool(false)]; + tensor input_63_cast_fp16 = concat(axis = var_2890, interleave = input_63_interleave_0, values = (hidden_states_29_cast_fp16, var_2892_cast_fp16))[name = string("input_63_cast_fp16")]; + tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; + fp16 var_2887_to_fp16 = const()[name = string("op_2887_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_2887_to_fp16, x = input_63_cast_fp16)[name = string("normed_37_cast_fp16")]; + tensor normed_39_begin_0 = const()[name = string("normed_39_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_39_end_0 = const()[name = string("normed_39_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_39_end_mask_0 = const()[name = string("normed_39_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_39_cast_fp16 = slice_by_index(begin = normed_39_begin_0, end = normed_39_end_0, end_mask = normed_39_end_mask_0, x = normed_37_cast_fp16)[name = string("normed_39_cast_fp16")]; + tensor const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(317383488)))]; + tensor x_77_cast_fp16 = mul(x = normed_39_cast_fp16, y = const_130_promoted_to_fp16)[name = string("x_77_cast_fp16")]; + tensor var_2917 = const()[name = string("op_2917"), val = tensor([0, 2, 1])]; + tensor input_65_axes_0 = const()[name = string("input_65_axes_0"), val = tensor([2])]; + tensor var_2918 = transpose(perm = var_2917, x = x_77_cast_fp16)[name = string("transpose_37")]; + tensor input_65 = expand_dims(axes = input_65_axes_0, x = var_2918)[name = string("input_65")]; + string input_67_pad_type_0 = const()[name = string("input_67_pad_type_0"), val = string("valid")]; + tensor input_67_strides_0 = const()[name = string("input_67_strides_0"), val = tensor([1, 1])]; + tensor input_67_pad_0 = const()[name = string("input_67_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_67_dilations_0 = const()[name = string("input_67_dilations_0"), val = tensor([1, 1])]; + int32 input_67_groups_0 = const()[name = string("input_67_groups_0"), val = int32(1)]; + tensor input_67 = conv(dilations = input_67_dilations_0, groups = input_67_groups_0, pad = input_67_pad_0, pad_type = input_67_pad_type_0, strides = input_67_strides_0, weight = model_model_layers_14_mlp_gate_proj_weight_palettized, x = input_65)[name = string("input_67")]; + string b_9_pad_type_0 = const()[name = string("b_9_pad_type_0"), val = string("valid")]; + tensor b_9_strides_0 = const()[name = string("b_9_strides_0"), val = tensor([1, 1])]; + tensor b_9_pad_0 = const()[name = string("b_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_9_dilations_0 = const()[name = string("b_9_dilations_0"), val = tensor([1, 1])]; + int32 b_9_groups_0 = const()[name = string("b_9_groups_0"), val = int32(1)]; + tensor b_9 = conv(dilations = b_9_dilations_0, groups = b_9_groups_0, pad = b_9_pad_0, pad_type = b_9_pad_type_0, strides = b_9_strides_0, weight = model_model_layers_14_mlp_up_proj_weight_palettized, x = input_65)[name = string("b_9")]; + tensor c_9 = silu(x = input_67)[name = string("c_9")]; + tensor input_69 = mul(x = c_9, y = b_9)[name = string("input_69")]; + string e_9_pad_type_0 = const()[name = string("e_9_pad_type_0"), val = string("valid")]; + tensor e_9_strides_0 = const()[name = string("e_9_strides_0"), val = tensor([1, 1])]; + tensor e_9_pad_0 = const()[name = string("e_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_9_dilations_0 = const()[name = string("e_9_dilations_0"), val = tensor([1, 1])]; + int32 e_9_groups_0 = const()[name = string("e_9_groups_0"), val = int32(1)]; + tensor e_9 = conv(dilations = e_9_dilations_0, groups = e_9_groups_0, pad = e_9_pad_0, pad_type = e_9_pad_type_0, strides = e_9_strides_0, weight = model_model_layers_14_mlp_down_proj_weight_palettized, x = input_69)[name = string("e_9")]; + tensor var_2940_axes_0 = const()[name = string("op_2940_axes_0"), val = tensor([2])]; + tensor var_2940 = squeeze(axes = var_2940_axes_0, x = e_9)[name = string("op_2940")]; + tensor var_2941 = const()[name = string("op_2941"), val = tensor([0, 2, 1])]; + tensor var_2942 = transpose(perm = var_2941, x = var_2940)[name = string("transpose_36")]; + tensor hidden_states_31_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = var_2942)[name = string("hidden_states_31_cast_fp16")]; + int32 var_2954 = const()[name = string("op_2954"), val = int32(-1)]; + fp16 const_131_promoted_to_fp16 = const()[name = string("const_131_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2956_cast_fp16 = mul(x = hidden_states_31_cast_fp16, y = const_131_promoted_to_fp16)[name = string("op_2956_cast_fp16")]; + bool input_71_interleave_0 = const()[name = string("input_71_interleave_0"), val = bool(false)]; + tensor input_71_cast_fp16 = concat(axis = var_2954, interleave = input_71_interleave_0, values = (hidden_states_31_cast_fp16, var_2956_cast_fp16))[name = string("input_71_cast_fp16")]; + tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; + fp16 var_2951_to_fp16 = const()[name = string("op_2951_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_2951_to_fp16, x = input_71_cast_fp16)[name = string("normed_41_cast_fp16")]; + tensor normed_43_begin_0 = const()[name = string("normed_43_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_43_end_0 = const()[name = string("normed_43_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_43_end_mask_0 = const()[name = string("normed_43_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_43_cast_fp16 = slice_by_index(begin = normed_43_begin_0, end = normed_43_end_0, end_mask = normed_43_end_mask_0, x = normed_41_cast_fp16)[name = string("normed_43_cast_fp16")]; + tensor const_134_promoted_to_fp16 = const()[name = string("const_134_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(317386624)))]; + tensor hidden_states_33_cast_fp16 = mul(x = normed_43_cast_fp16, y = const_134_promoted_to_fp16)[name = string("hidden_states_33_cast_fp16")]; + tensor var_2979 = const()[name = string("op_2979"), val = tensor([0, 2, 1])]; + tensor var_2982_axes_0 = const()[name = string("op_2982_axes_0"), val = tensor([2])]; + tensor var_2980_cast_fp16 = transpose(perm = var_2979, x = hidden_states_33_cast_fp16)[name = string("transpose_35")]; + tensor var_2982_cast_fp16 = expand_dims(axes = var_2982_axes_0, x = var_2980_cast_fp16)[name = string("op_2982_cast_fp16")]; + string query_states_31_pad_type_0 = const()[name = string("query_states_31_pad_type_0"), val = string("valid")]; + tensor query_states_31_strides_0 = const()[name = string("query_states_31_strides_0"), val = tensor([1, 1])]; + tensor query_states_31_pad_0 = const()[name = string("query_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_31_dilations_0 = const()[name = string("query_states_31_dilations_0"), val = tensor([1, 1])]; + int32 query_states_31_groups_0 = const()[name = string("query_states_31_groups_0"), val = int32(1)]; + tensor query_states_31 = conv(bias = model_model_layers_15_self_attn_q_proj_bias, dilations = query_states_31_dilations_0, groups = query_states_31_groups_0, pad = query_states_31_pad_0, pad_type = query_states_31_pad_type_0, strides = query_states_31_strides_0, weight = model_model_layers_15_self_attn_q_proj_weight_palettized, x = var_2982_cast_fp16)[name = string("query_states_31")]; + string key_states_41_pad_type_0 = const()[name = string("key_states_41_pad_type_0"), val = string("valid")]; + tensor key_states_41_strides_0 = const()[name = string("key_states_41_strides_0"), val = tensor([1, 1])]; + tensor key_states_41_pad_0 = const()[name = string("key_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_41_dilations_0 = const()[name = string("key_states_41_dilations_0"), val = tensor([1, 1])]; + int32 key_states_41_groups_0 = const()[name = string("key_states_41_groups_0"), val = int32(1)]; + tensor key_states_41 = conv(bias = model_model_layers_15_self_attn_k_proj_bias, dilations = key_states_41_dilations_0, groups = key_states_41_groups_0, pad = key_states_41_pad_0, pad_type = key_states_41_pad_type_0, strides = key_states_41_strides_0, weight = model_model_layers_15_self_attn_k_proj_weight_palettized, x = var_2982_cast_fp16)[name = string("key_states_41")]; + string value_states_41_pad_type_0 = const()[name = string("value_states_41_pad_type_0"), val = string("valid")]; + tensor value_states_41_strides_0 = const()[name = string("value_states_41_strides_0"), val = tensor([1, 1])]; + tensor value_states_41_pad_0 = const()[name = string("value_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_41_dilations_0 = const()[name = string("value_states_41_dilations_0"), val = tensor([1, 1])]; + int32 value_states_41_groups_0 = const()[name = string("value_states_41_groups_0"), val = int32(1)]; + tensor value_states_41 = conv(bias = model_model_layers_15_self_attn_v_proj_bias, dilations = value_states_41_dilations_0, groups = value_states_41_groups_0, pad = value_states_41_pad_0, pad_type = value_states_41_pad_type_0, strides = value_states_41_strides_0, weight = model_model_layers_15_self_attn_v_proj_weight_palettized, x = var_2982_cast_fp16)[name = string("value_states_41")]; + tensor var_3024 = const()[name = string("op_3024"), val = tensor([1, 12, 128, 64])]; + tensor var_3025 = reshape(shape = var_3024, x = query_states_31)[name = string("op_3025")]; + tensor var_3030 = const()[name = string("op_3030"), val = tensor([0, 1, 3, 2])]; + tensor var_3035 = const()[name = string("op_3035"), val = tensor([1, 2, 128, 64])]; + tensor var_3036 = reshape(shape = var_3035, x = key_states_41)[name = string("op_3036")]; + tensor var_3041 = const()[name = string("op_3041"), val = tensor([0, 1, 3, 2])]; + tensor var_3046 = const()[name = string("op_3046"), val = tensor([1, 2, 128, 64])]; + tensor var_3047 = reshape(shape = var_3046, x = value_states_41)[name = string("op_3047")]; + tensor var_3052 = const()[name = string("op_3052"), val = tensor([0, 1, 3, 2])]; + tensor q_21 = transpose(perm = var_3030, x = var_3025)[name = string("transpose_34")]; + tensor var_3066 = mul(x = q_21, y = cos_5)[name = string("op_3066")]; + tensor x1_21_begin_0 = const()[name = string("x1_21_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_21_end_0 = const()[name = string("x1_21_end_0"), val = tensor([1, 12, 64, 64])]; + tensor x1_21_end_mask_0 = const()[name = string("x1_21_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_21 = slice_by_index(begin = x1_21_begin_0, end = x1_21_end_0, end_mask = x1_21_end_mask_0, x = q_21)[name = string("x1_21")]; + tensor x2_21_begin_0 = const()[name = string("x2_21_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_21_end_0 = const()[name = string("x2_21_end_0"), val = tensor([1, 12, 64, 128])]; + tensor x2_21_end_mask_0 = const()[name = string("x2_21_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_21 = slice_by_index(begin = x2_21_begin_0, end = x2_21_end_0, end_mask = x2_21_end_mask_0, x = q_21)[name = string("x2_21")]; + fp16 const_138_promoted = const()[name = string("const_138_promoted"), val = fp16(-0x1p+0)]; + tensor var_3087 = mul(x = x2_21, y = const_138_promoted)[name = string("op_3087")]; + int32 var_3089 = const()[name = string("op_3089"), val = int32(-1)]; + bool var_3090_interleave_0 = const()[name = string("op_3090_interleave_0"), val = bool(false)]; + tensor var_3090 = concat(axis = var_3089, interleave = var_3090_interleave_0, values = (var_3087, x1_21))[name = string("op_3090")]; + tensor var_3091 = mul(x = var_3090, y = sin_5)[name = string("op_3091")]; + tensor query_states_33 = add(x = var_3066, y = var_3091)[name = string("query_states_33")]; + tensor k_21 = transpose(perm = var_3041, x = var_3036)[name = string("transpose_33")]; + tensor var_3094 = mul(x = k_21, y = cos_5)[name = string("op_3094")]; + tensor x1_23_begin_0 = const()[name = string("x1_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_23_end_0 = const()[name = string("x1_23_end_0"), val = tensor([1, 2, 64, 64])]; + tensor x1_23_end_mask_0 = const()[name = string("x1_23_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_23 = slice_by_index(begin = x1_23_begin_0, end = x1_23_end_0, end_mask = x1_23_end_mask_0, x = k_21)[name = string("x1_23")]; + tensor x2_23_begin_0 = const()[name = string("x2_23_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_23_end_0 = const()[name = string("x2_23_end_0"), val = tensor([1, 2, 64, 128])]; + tensor x2_23_end_mask_0 = const()[name = string("x2_23_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_23 = slice_by_index(begin = x2_23_begin_0, end = x2_23_end_0, end_mask = x2_23_end_mask_0, x = k_21)[name = string("x2_23")]; + fp16 const_141_promoted = const()[name = string("const_141_promoted"), val = fp16(-0x1p+0)]; + tensor var_3115 = mul(x = x2_23, y = const_141_promoted)[name = string("op_3115")]; + int32 var_3117 = const()[name = string("op_3117"), val = int32(-1)]; + bool var_3118_interleave_0 = const()[name = string("op_3118_interleave_0"), val = bool(false)]; + tensor var_3118 = concat(axis = var_3117, interleave = var_3118_interleave_0, values = (var_3115, x1_23))[name = string("op_3118")]; + tensor var_3119 = mul(x = var_3118, y = sin_5)[name = string("op_3119")]; + tensor key_states_43 = add(x = var_3094, y = var_3119)[name = string("key_states_43")]; + tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([15])]; + tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; + tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; + tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([16])]; + int32 concat_92_axis_0 = const()[name = string("concat_92_axis_0"), val = int32(0)]; + bool concat_92_interleave_0 = const()[name = string("concat_92_interleave_0"), val = bool(false)]; + tensor concat_92 = concat(axis = concat_92_axis_0, interleave = concat_92_interleave_0, values = (expand_dims_60, expand_dims_61, current_pos, expand_dims_63))[name = string("concat_92")]; + tensor concat_93_values1_0 = const()[name = string("concat_93_values1_0"), val = tensor([0])]; + tensor concat_93_values3_0 = const()[name = string("concat_93_values3_0"), val = tensor([0])]; + int32 concat_93_axis_0 = const()[name = string("concat_93_axis_0"), val = int32(0)]; + bool concat_93_interleave_0 = const()[name = string("concat_93_interleave_0"), val = bool(false)]; + tensor concat_93 = concat(axis = concat_93_axis_0, interleave = concat_93_interleave_0, values = (expand_dims_64, concat_93_values1_0, var_616, concat_93_values3_0))[name = string("concat_93")]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_92, begin_mask = model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0, end = concat_93, end_mask = model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_11_stride_0, update = key_states_43, x = coreml_update_state_27)[name = string("model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_28_write_state")]; + tensor coreml_update_state_28 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_28")]; + tensor expand_dims_66 = const()[name = string("expand_dims_66"), val = tensor([43])]; + tensor expand_dims_67 = const()[name = string("expand_dims_67"), val = tensor([0])]; + tensor expand_dims_69 = const()[name = string("expand_dims_69"), val = tensor([0])]; + tensor expand_dims_70 = const()[name = string("expand_dims_70"), val = tensor([44])]; + int32 concat_96_axis_0 = const()[name = string("concat_96_axis_0"), val = int32(0)]; + bool concat_96_interleave_0 = const()[name = string("concat_96_interleave_0"), val = bool(false)]; + tensor concat_96 = concat(axis = concat_96_axis_0, interleave = concat_96_interleave_0, values = (expand_dims_66, expand_dims_67, current_pos, expand_dims_69))[name = string("concat_96")]; + tensor concat_97_values1_0 = const()[name = string("concat_97_values1_0"), val = tensor([0])]; + tensor concat_97_values3_0 = const()[name = string("concat_97_values3_0"), val = tensor([0])]; + int32 concat_97_axis_0 = const()[name = string("concat_97_axis_0"), val = int32(0)]; + bool concat_97_interleave_0 = const()[name = string("concat_97_interleave_0"), val = bool(false)]; + tensor concat_97 = concat(axis = concat_97_axis_0, interleave = concat_97_interleave_0, values = (expand_dims_70, concat_97_values1_0, var_616, concat_97_values3_0))[name = string("concat_97")]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_43 = transpose(perm = var_3052, x = var_3047)[name = string("transpose_32")]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_96, begin_mask = model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0, end = concat_97, end_mask = model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_12_stride_0, update = value_states_43, x = coreml_update_state_28)[name = string("model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_29_write_state")]; + tensor coreml_update_state_29 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_29")]; + tensor var_3190_begin_0 = const()[name = string("op_3190_begin_0"), val = tensor([15, 0, 0, 0])]; + tensor var_3190_end_0 = const()[name = string("op_3190_end_0"), val = tensor([16, 2, 2048, 128])]; + tensor var_3190_end_mask_0 = const()[name = string("op_3190_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3190_cast_fp16 = slice_by_index(begin = var_3190_begin_0, end = var_3190_end_0, end_mask = var_3190_end_mask_0, x = coreml_update_state_29)[name = string("op_3190_cast_fp16")]; + tensor K_layer_cache_11_axes_0 = const()[name = string("K_layer_cache_11_axes_0"), val = tensor([0])]; + tensor K_layer_cache_11_cast_fp16 = squeeze(axes = K_layer_cache_11_axes_0, x = var_3190_cast_fp16)[name = string("K_layer_cache_11_cast_fp16")]; + tensor var_3197_begin_0 = const()[name = string("op_3197_begin_0"), val = tensor([43, 0, 0, 0])]; + tensor var_3197_end_0 = const()[name = string("op_3197_end_0"), val = tensor([44, 2, 2048, 128])]; + tensor var_3197_end_mask_0 = const()[name = string("op_3197_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3197_cast_fp16 = slice_by_index(begin = var_3197_begin_0, end = var_3197_end_0, end_mask = var_3197_end_mask_0, x = coreml_update_state_29)[name = string("op_3197_cast_fp16")]; + tensor V_layer_cache_11_axes_0 = const()[name = string("V_layer_cache_11_axes_0"), val = tensor([0])]; + tensor V_layer_cache_11_cast_fp16 = squeeze(axes = V_layer_cache_11_axes_0, x = var_3197_cast_fp16)[name = string("V_layer_cache_11_cast_fp16")]; + tensor x_83_axes_0 = const()[name = string("x_83_axes_0"), val = tensor([1])]; + tensor x_83_cast_fp16 = expand_dims(axes = x_83_axes_0, x = K_layer_cache_11_cast_fp16)[name = string("x_83_cast_fp16")]; + tensor var_3226 = const()[name = string("op_3226"), val = tensor([1, 6, 1, 1])]; + tensor x_85_cast_fp16 = tile(reps = var_3226, x = x_83_cast_fp16)[name = string("x_85_cast_fp16")]; + tensor var_3238 = const()[name = string("op_3238"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_47_cast_fp16 = reshape(shape = var_3238, x = x_85_cast_fp16)[name = string("key_states_47_cast_fp16")]; + tensor x_89_axes_0 = const()[name = string("x_89_axes_0"), val = tensor([1])]; + tensor x_89_cast_fp16 = expand_dims(axes = x_89_axes_0, x = V_layer_cache_11_cast_fp16)[name = string("x_89_cast_fp16")]; + tensor var_3246 = const()[name = string("op_3246"), val = tensor([1, 6, 1, 1])]; + tensor x_91_cast_fp16 = tile(reps = var_3246, x = x_89_cast_fp16)[name = string("x_91_cast_fp16")]; + bool var_3281_transpose_x_1 = const()[name = string("op_3281_transpose_x_1"), val = bool(false)]; + bool var_3281_transpose_y_1 = const()[name = string("op_3281_transpose_y_1"), val = bool(true)]; + tensor var_3281_cast_fp16 = matmul(transpose_x = var_3281_transpose_x_1, transpose_y = var_3281_transpose_y_1, x = query_states_33, y = key_states_47_cast_fp16)[name = string("op_3281_cast_fp16")]; + fp16 var_3282_to_fp16 = const()[name = string("op_3282_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_21_cast_fp16 = mul(x = var_3281_cast_fp16, y = var_3282_to_fp16)[name = string("attn_logits_21_cast_fp16")]; + tensor attn_logits_23_cast_fp16 = add(x = attn_logits_21_cast_fp16, y = causal_mask)[name = string("attn_logits_23_cast_fp16")]; + int32 var_3309 = const()[name = string("op_3309"), val = int32(-1)]; + tensor var_3311_cast_fp16 = softmax(axis = var_3309, x = attn_logits_23_cast_fp16)[name = string("op_3311_cast_fp16")]; + tensor concat_102 = const()[name = string("concat_102"), val = tensor([12, 64, 2048])]; + tensor reshape_15_cast_fp16 = reshape(shape = concat_102, x = var_3311_cast_fp16)[name = string("reshape_15_cast_fp16")]; + tensor concat_103 = const()[name = string("concat_103"), val = tensor([12, 2048, 128])]; + tensor reshape_16_cast_fp16 = reshape(shape = concat_103, x = x_91_cast_fp16)[name = string("reshape_16_cast_fp16")]; + bool matmul_5_transpose_x_0 = const()[name = string("matmul_5_transpose_x_0"), val = bool(false)]; + bool matmul_5_transpose_y_0 = const()[name = string("matmul_5_transpose_y_0"), val = bool(false)]; + tensor matmul_5_cast_fp16 = matmul(transpose_x = matmul_5_transpose_x_0, transpose_y = matmul_5_transpose_y_0, x = reshape_15_cast_fp16, y = reshape_16_cast_fp16)[name = string("matmul_5_cast_fp16")]; + tensor concat_107 = const()[name = string("concat_107"), val = tensor([1, 12, 64, 128])]; + tensor reshape_17_cast_fp16 = reshape(shape = concat_107, x = matmul_5_cast_fp16)[name = string("reshape_17_cast_fp16")]; + tensor var_3338_perm_0 = const()[name = string("op_3338_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_3357 = const()[name = string("op_3357"), val = tensor([1, 64, 1536])]; + tensor var_3338 = transpose(perm = var_3338_perm_0, x = reshape_17_cast_fp16)[name = string("transpose_31")]; + tensor attn_output_55 = reshape(shape = var_3357, x = var_3338)[name = string("attn_output_55")]; + tensor var_3362 = const()[name = string("op_3362"), val = tensor([0, 2, 1])]; + tensor squeeze_5_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(317389760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(319159296))))[name = string("squeeze_5_palettized")]; + string var_3378_pad_type_0 = const()[name = string("op_3378_pad_type_0"), val = string("valid")]; + int32 var_3378_groups_0 = const()[name = string("op_3378_groups_0"), val = int32(1)]; + tensor var_3378_strides_0 = const()[name = string("op_3378_strides_0"), val = tensor([1])]; + tensor var_3378_pad_0 = const()[name = string("op_3378_pad_0"), val = tensor([0, 0])]; + tensor var_3378_dilations_0 = const()[name = string("op_3378_dilations_0"), val = tensor([1])]; + tensor var_3363 = transpose(perm = var_3362, x = attn_output_55)[name = string("transpose_30")]; + tensor var_3378 = conv(dilations = var_3378_dilations_0, groups = var_3378_groups_0, pad = var_3378_pad_0, pad_type = var_3378_pad_type_0, strides = var_3378_strides_0, weight = squeeze_5_palettized, x = var_3363)[name = string("op_3378")]; + tensor var_3382 = const()[name = string("op_3382"), val = tensor([0, 2, 1])]; + tensor attn_output_59 = transpose(perm = var_3382, x = var_3378)[name = string("transpose_29")]; + tensor hidden_states_35_cast_fp16 = add(x = hidden_states_31_cast_fp16, y = attn_output_59)[name = string("hidden_states_35_cast_fp16")]; + int32 var_3395 = const()[name = string("op_3395"), val = int32(-1)]; + fp16 const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3397_cast_fp16 = mul(x = hidden_states_35_cast_fp16, y = const_153_promoted_to_fp16)[name = string("op_3397_cast_fp16")]; + bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; + tensor input_77_cast_fp16 = concat(axis = var_3395, interleave = input_77_interleave_0, values = (hidden_states_35_cast_fp16, var_3397_cast_fp16))[name = string("input_77_cast_fp16")]; + tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; + fp16 var_3392_to_fp16 = const()[name = string("op_3392_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_3392_to_fp16, x = input_77_cast_fp16)[name = string("normed_45_cast_fp16")]; + tensor normed_47_begin_0 = const()[name = string("normed_47_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_47_end_0 = const()[name = string("normed_47_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_47_end_mask_0 = const()[name = string("normed_47_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_47_cast_fp16 = slice_by_index(begin = normed_47_begin_0, end = normed_47_end_0, end_mask = normed_47_end_mask_0, x = normed_45_cast_fp16)[name = string("normed_47_cast_fp16")]; + tensor const_156_promoted_to_fp16 = const()[name = string("const_156_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(319208512)))]; + tensor x_93_cast_fp16 = mul(x = normed_47_cast_fp16, y = const_156_promoted_to_fp16)[name = string("x_93_cast_fp16")]; + tensor var_3422 = const()[name = string("op_3422"), val = tensor([0, 2, 1])]; + tensor input_79_axes_0 = const()[name = string("input_79_axes_0"), val = tensor([2])]; + tensor var_3423 = transpose(perm = var_3422, x = x_93_cast_fp16)[name = string("transpose_28")]; + tensor input_79 = expand_dims(axes = input_79_axes_0, x = var_3423)[name = string("input_79")]; + string input_81_pad_type_0 = const()[name = string("input_81_pad_type_0"), val = string("valid")]; + tensor input_81_strides_0 = const()[name = string("input_81_strides_0"), val = tensor([1, 1])]; + tensor input_81_pad_0 = const()[name = string("input_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_81_dilations_0 = const()[name = string("input_81_dilations_0"), val = tensor([1, 1])]; + int32 input_81_groups_0 = const()[name = string("input_81_groups_0"), val = int32(1)]; + tensor input_81 = conv(dilations = input_81_dilations_0, groups = input_81_groups_0, pad = input_81_pad_0, pad_type = input_81_pad_type_0, strides = input_81_strides_0, weight = model_model_layers_15_mlp_gate_proj_weight_palettized, x = input_79)[name = string("input_81")]; + string b_11_pad_type_0 = const()[name = string("b_11_pad_type_0"), val = string("valid")]; + tensor b_11_strides_0 = const()[name = string("b_11_strides_0"), val = tensor([1, 1])]; + tensor b_11_pad_0 = const()[name = string("b_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_11_dilations_0 = const()[name = string("b_11_dilations_0"), val = tensor([1, 1])]; + int32 b_11_groups_0 = const()[name = string("b_11_groups_0"), val = int32(1)]; + tensor b_11 = conv(dilations = b_11_dilations_0, groups = b_11_groups_0, pad = b_11_pad_0, pad_type = b_11_pad_type_0, strides = b_11_strides_0, weight = model_model_layers_15_mlp_up_proj_weight_palettized, x = input_79)[name = string("b_11")]; + tensor c_11 = silu(x = input_81)[name = string("c_11")]; + tensor input_83 = mul(x = c_11, y = b_11)[name = string("input_83")]; + string e_11_pad_type_0 = const()[name = string("e_11_pad_type_0"), val = string("valid")]; + tensor e_11_strides_0 = const()[name = string("e_11_strides_0"), val = tensor([1, 1])]; + tensor e_11_pad_0 = const()[name = string("e_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_11_dilations_0 = const()[name = string("e_11_dilations_0"), val = tensor([1, 1])]; + int32 e_11_groups_0 = const()[name = string("e_11_groups_0"), val = int32(1)]; + tensor e_11 = conv(dilations = e_11_dilations_0, groups = e_11_groups_0, pad = e_11_pad_0, pad_type = e_11_pad_type_0, strides = e_11_strides_0, weight = model_model_layers_15_mlp_down_proj_weight_palettized, x = input_83)[name = string("e_11")]; + tensor var_3445_axes_0 = const()[name = string("op_3445_axes_0"), val = tensor([2])]; + tensor var_3445 = squeeze(axes = var_3445_axes_0, x = e_11)[name = string("op_3445")]; + tensor var_3446 = const()[name = string("op_3446"), val = tensor([0, 2, 1])]; + tensor var_3447 = transpose(perm = var_3446, x = var_3445)[name = string("transpose_27")]; + tensor hidden_states_37_cast_fp16 = add(x = hidden_states_35_cast_fp16, y = var_3447)[name = string("hidden_states_37_cast_fp16")]; + int32 var_3459 = const()[name = string("op_3459"), val = int32(-1)]; + fp16 const_157_promoted_to_fp16 = const()[name = string("const_157_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3461_cast_fp16 = mul(x = hidden_states_37_cast_fp16, y = const_157_promoted_to_fp16)[name = string("op_3461_cast_fp16")]; + bool input_85_interleave_0 = const()[name = string("input_85_interleave_0"), val = bool(false)]; + tensor input_85_cast_fp16 = concat(axis = var_3459, interleave = input_85_interleave_0, values = (hidden_states_37_cast_fp16, var_3461_cast_fp16))[name = string("input_85_cast_fp16")]; + tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; + fp16 var_3456_to_fp16 = const()[name = string("op_3456_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_3456_to_fp16, x = input_85_cast_fp16)[name = string("normed_49_cast_fp16")]; + tensor normed_51_begin_0 = const()[name = string("normed_51_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_51_end_0 = const()[name = string("normed_51_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_51_end_mask_0 = const()[name = string("normed_51_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_51_cast_fp16 = slice_by_index(begin = normed_51_begin_0, end = normed_51_end_0, end_mask = normed_51_end_mask_0, x = normed_49_cast_fp16)[name = string("normed_51_cast_fp16")]; + tensor const_160_promoted_to_fp16 = const()[name = string("const_160_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(319211648)))]; + tensor hidden_states_39_cast_fp16 = mul(x = normed_51_cast_fp16, y = const_160_promoted_to_fp16)[name = string("hidden_states_39_cast_fp16")]; + tensor var_3484 = const()[name = string("op_3484"), val = tensor([0, 2, 1])]; + tensor var_3487_axes_0 = const()[name = string("op_3487_axes_0"), val = tensor([2])]; + tensor var_3485_cast_fp16 = transpose(perm = var_3484, x = hidden_states_39_cast_fp16)[name = string("transpose_26")]; + tensor var_3487_cast_fp16 = expand_dims(axes = var_3487_axes_0, x = var_3485_cast_fp16)[name = string("op_3487_cast_fp16")]; + string query_states_37_pad_type_0 = const()[name = string("query_states_37_pad_type_0"), val = string("valid")]; + tensor query_states_37_strides_0 = const()[name = string("query_states_37_strides_0"), val = tensor([1, 1])]; + tensor query_states_37_pad_0 = const()[name = string("query_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_37_dilations_0 = const()[name = string("query_states_37_dilations_0"), val = tensor([1, 1])]; + int32 query_states_37_groups_0 = const()[name = string("query_states_37_groups_0"), val = int32(1)]; + tensor query_states_37 = conv(bias = model_model_layers_16_self_attn_q_proj_bias, dilations = query_states_37_dilations_0, groups = query_states_37_groups_0, pad = query_states_37_pad_0, pad_type = query_states_37_pad_type_0, strides = query_states_37_strides_0, weight = model_model_layers_16_self_attn_q_proj_weight_palettized, x = var_3487_cast_fp16)[name = string("query_states_37")]; + string key_states_49_pad_type_0 = const()[name = string("key_states_49_pad_type_0"), val = string("valid")]; + tensor key_states_49_strides_0 = const()[name = string("key_states_49_strides_0"), val = tensor([1, 1])]; + tensor key_states_49_pad_0 = const()[name = string("key_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_49_dilations_0 = const()[name = string("key_states_49_dilations_0"), val = tensor([1, 1])]; + int32 key_states_49_groups_0 = const()[name = string("key_states_49_groups_0"), val = int32(1)]; + tensor key_states_49 = conv(bias = model_model_layers_16_self_attn_k_proj_bias, dilations = key_states_49_dilations_0, groups = key_states_49_groups_0, pad = key_states_49_pad_0, pad_type = key_states_49_pad_type_0, strides = key_states_49_strides_0, weight = model_model_layers_16_self_attn_k_proj_weight_palettized, x = var_3487_cast_fp16)[name = string("key_states_49")]; + string value_states_49_pad_type_0 = const()[name = string("value_states_49_pad_type_0"), val = string("valid")]; + tensor value_states_49_strides_0 = const()[name = string("value_states_49_strides_0"), val = tensor([1, 1])]; + tensor value_states_49_pad_0 = const()[name = string("value_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_49_dilations_0 = const()[name = string("value_states_49_dilations_0"), val = tensor([1, 1])]; + int32 value_states_49_groups_0 = const()[name = string("value_states_49_groups_0"), val = int32(1)]; + tensor value_states_49 = conv(bias = model_model_layers_16_self_attn_v_proj_bias, dilations = value_states_49_dilations_0, groups = value_states_49_groups_0, pad = value_states_49_pad_0, pad_type = value_states_49_pad_type_0, strides = value_states_49_strides_0, weight = model_model_layers_16_self_attn_v_proj_weight_palettized, x = var_3487_cast_fp16)[name = string("value_states_49")]; + tensor var_3529 = const()[name = string("op_3529"), val = tensor([1, 12, 128, 64])]; + tensor var_3530 = reshape(shape = var_3529, x = query_states_37)[name = string("op_3530")]; + tensor var_3535 = const()[name = string("op_3535"), val = tensor([0, 1, 3, 2])]; + tensor var_3540 = const()[name = string("op_3540"), val = tensor([1, 2, 128, 64])]; + tensor var_3541 = reshape(shape = var_3540, x = key_states_49)[name = string("op_3541")]; + tensor var_3546 = const()[name = string("op_3546"), val = tensor([0, 1, 3, 2])]; + tensor var_3551 = const()[name = string("op_3551"), val = tensor([1, 2, 128, 64])]; + tensor var_3552 = reshape(shape = var_3551, x = value_states_49)[name = string("op_3552")]; + tensor var_3557 = const()[name = string("op_3557"), val = tensor([0, 1, 3, 2])]; + tensor q_25 = transpose(perm = var_3535, x = var_3530)[name = string("transpose_25")]; + tensor var_3571 = mul(x = q_25, y = cos_5)[name = string("op_3571")]; + tensor x1_25_begin_0 = const()[name = string("x1_25_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_25_end_0 = const()[name = string("x1_25_end_0"), val = tensor([1, 12, 64, 64])]; + tensor x1_25_end_mask_0 = const()[name = string("x1_25_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_25 = slice_by_index(begin = x1_25_begin_0, end = x1_25_end_0, end_mask = x1_25_end_mask_0, x = q_25)[name = string("x1_25")]; + tensor x2_25_begin_0 = const()[name = string("x2_25_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_25_end_0 = const()[name = string("x2_25_end_0"), val = tensor([1, 12, 64, 128])]; + tensor x2_25_end_mask_0 = const()[name = string("x2_25_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_25 = slice_by_index(begin = x2_25_begin_0, end = x2_25_end_0, end_mask = x2_25_end_mask_0, x = q_25)[name = string("x2_25")]; + fp16 const_164_promoted = const()[name = string("const_164_promoted"), val = fp16(-0x1p+0)]; + tensor var_3592 = mul(x = x2_25, y = const_164_promoted)[name = string("op_3592")]; + int32 var_3594 = const()[name = string("op_3594"), val = int32(-1)]; + bool var_3595_interleave_0 = const()[name = string("op_3595_interleave_0"), val = bool(false)]; + tensor var_3595 = concat(axis = var_3594, interleave = var_3595_interleave_0, values = (var_3592, x1_25))[name = string("op_3595")]; + tensor var_3596 = mul(x = var_3595, y = sin_5)[name = string("op_3596")]; + tensor query_states_39 = add(x = var_3571, y = var_3596)[name = string("query_states_39")]; + tensor k_25 = transpose(perm = var_3546, x = var_3541)[name = string("transpose_24")]; + tensor var_3599 = mul(x = k_25, y = cos_5)[name = string("op_3599")]; + tensor x1_27_begin_0 = const()[name = string("x1_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_27_end_0 = const()[name = string("x1_27_end_0"), val = tensor([1, 2, 64, 64])]; + tensor x1_27_end_mask_0 = const()[name = string("x1_27_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_27 = slice_by_index(begin = x1_27_begin_0, end = x1_27_end_0, end_mask = x1_27_end_mask_0, x = k_25)[name = string("x1_27")]; + tensor x2_27_begin_0 = const()[name = string("x2_27_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_27_end_0 = const()[name = string("x2_27_end_0"), val = tensor([1, 2, 64, 128])]; + tensor x2_27_end_mask_0 = const()[name = string("x2_27_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_27 = slice_by_index(begin = x2_27_begin_0, end = x2_27_end_0, end_mask = x2_27_end_mask_0, x = k_25)[name = string("x2_27")]; + fp16 const_167_promoted = const()[name = string("const_167_promoted"), val = fp16(-0x1p+0)]; + tensor var_3620 = mul(x = x2_27, y = const_167_promoted)[name = string("op_3620")]; + int32 var_3622 = const()[name = string("op_3622"), val = int32(-1)]; + bool var_3623_interleave_0 = const()[name = string("op_3623_interleave_0"), val = bool(false)]; + tensor var_3623 = concat(axis = var_3622, interleave = var_3623_interleave_0, values = (var_3620, x1_27))[name = string("op_3623")]; + tensor var_3624 = mul(x = var_3623, y = sin_5)[name = string("op_3624")]; + tensor key_states_51 = add(x = var_3599, y = var_3624)[name = string("key_states_51")]; + tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([16])]; + tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; + tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; + tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([17])]; + int32 concat_110_axis_0 = const()[name = string("concat_110_axis_0"), val = int32(0)]; + bool concat_110_interleave_0 = const()[name = string("concat_110_interleave_0"), val = bool(false)]; + tensor concat_110 = concat(axis = concat_110_axis_0, interleave = concat_110_interleave_0, values = (expand_dims_72, expand_dims_73, current_pos, expand_dims_75))[name = string("concat_110")]; + tensor concat_111_values1_0 = const()[name = string("concat_111_values1_0"), val = tensor([0])]; + tensor concat_111_values3_0 = const()[name = string("concat_111_values3_0"), val = tensor([0])]; + int32 concat_111_axis_0 = const()[name = string("concat_111_axis_0"), val = int32(0)]; + bool concat_111_interleave_0 = const()[name = string("concat_111_interleave_0"), val = bool(false)]; + tensor concat_111 = concat(axis = concat_111_axis_0, interleave = concat_111_interleave_0, values = (expand_dims_76, concat_111_values1_0, var_616, concat_111_values3_0))[name = string("concat_111")]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_110, begin_mask = model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0, end = concat_111, end_mask = model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_13_stride_0, update = key_states_51, x = coreml_update_state_29)[name = string("model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_30_write_state")]; + tensor coreml_update_state_30 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_30")]; + tensor expand_dims_78 = const()[name = string("expand_dims_78"), val = tensor([44])]; + tensor expand_dims_79 = const()[name = string("expand_dims_79"), val = tensor([0])]; + tensor expand_dims_81 = const()[name = string("expand_dims_81"), val = tensor([0])]; + tensor expand_dims_82 = const()[name = string("expand_dims_82"), val = tensor([45])]; + int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; + bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; + tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_78, expand_dims_79, current_pos, expand_dims_81))[name = string("concat_114")]; + tensor concat_115_values1_0 = const()[name = string("concat_115_values1_0"), val = tensor([0])]; + tensor concat_115_values3_0 = const()[name = string("concat_115_values3_0"), val = tensor([0])]; + int32 concat_115_axis_0 = const()[name = string("concat_115_axis_0"), val = int32(0)]; + bool concat_115_interleave_0 = const()[name = string("concat_115_interleave_0"), val = bool(false)]; + tensor concat_115 = concat(axis = concat_115_axis_0, interleave = concat_115_interleave_0, values = (expand_dims_82, concat_115_values1_0, var_616, concat_115_values3_0))[name = string("concat_115")]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_51 = transpose(perm = var_3557, x = var_3552)[name = string("transpose_23")]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_114, begin_mask = model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0, end = concat_115, end_mask = model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_14_stride_0, update = value_states_51, x = coreml_update_state_30)[name = string("model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_31_write_state")]; + tensor coreml_update_state_31 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_31")]; + tensor var_3695_begin_0 = const()[name = string("op_3695_begin_0"), val = tensor([16, 0, 0, 0])]; + tensor var_3695_end_0 = const()[name = string("op_3695_end_0"), val = tensor([17, 2, 2048, 128])]; + tensor var_3695_end_mask_0 = const()[name = string("op_3695_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3695_cast_fp16 = slice_by_index(begin = var_3695_begin_0, end = var_3695_end_0, end_mask = var_3695_end_mask_0, x = coreml_update_state_31)[name = string("op_3695_cast_fp16")]; + tensor K_layer_cache_13_axes_0 = const()[name = string("K_layer_cache_13_axes_0"), val = tensor([0])]; + tensor K_layer_cache_13_cast_fp16 = squeeze(axes = K_layer_cache_13_axes_0, x = var_3695_cast_fp16)[name = string("K_layer_cache_13_cast_fp16")]; + tensor var_3702_begin_0 = const()[name = string("op_3702_begin_0"), val = tensor([44, 0, 0, 0])]; + tensor var_3702_end_0 = const()[name = string("op_3702_end_0"), val = tensor([45, 2, 2048, 128])]; + tensor var_3702_end_mask_0 = const()[name = string("op_3702_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3702_cast_fp16 = slice_by_index(begin = var_3702_begin_0, end = var_3702_end_0, end_mask = var_3702_end_mask_0, x = coreml_update_state_31)[name = string("op_3702_cast_fp16")]; + tensor V_layer_cache_13_axes_0 = const()[name = string("V_layer_cache_13_axes_0"), val = tensor([0])]; + tensor V_layer_cache_13_cast_fp16 = squeeze(axes = V_layer_cache_13_axes_0, x = var_3702_cast_fp16)[name = string("V_layer_cache_13_cast_fp16")]; + tensor x_99_axes_0 = const()[name = string("x_99_axes_0"), val = tensor([1])]; + tensor x_99_cast_fp16 = expand_dims(axes = x_99_axes_0, x = K_layer_cache_13_cast_fp16)[name = string("x_99_cast_fp16")]; + tensor var_3731 = const()[name = string("op_3731"), val = tensor([1, 6, 1, 1])]; + tensor x_101_cast_fp16 = tile(reps = var_3731, x = x_99_cast_fp16)[name = string("x_101_cast_fp16")]; + tensor var_3743 = const()[name = string("op_3743"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_55_cast_fp16 = reshape(shape = var_3743, x = x_101_cast_fp16)[name = string("key_states_55_cast_fp16")]; + tensor x_105_axes_0 = const()[name = string("x_105_axes_0"), val = tensor([1])]; + tensor x_105_cast_fp16 = expand_dims(axes = x_105_axes_0, x = V_layer_cache_13_cast_fp16)[name = string("x_105_cast_fp16")]; + tensor var_3751 = const()[name = string("op_3751"), val = tensor([1, 6, 1, 1])]; + tensor x_107_cast_fp16 = tile(reps = var_3751, x = x_105_cast_fp16)[name = string("x_107_cast_fp16")]; + bool var_3786_transpose_x_1 = const()[name = string("op_3786_transpose_x_1"), val = bool(false)]; + bool var_3786_transpose_y_1 = const()[name = string("op_3786_transpose_y_1"), val = bool(true)]; + tensor var_3786_cast_fp16 = matmul(transpose_x = var_3786_transpose_x_1, transpose_y = var_3786_transpose_y_1, x = query_states_39, y = key_states_55_cast_fp16)[name = string("op_3786_cast_fp16")]; + fp16 var_3787_to_fp16 = const()[name = string("op_3787_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_25_cast_fp16 = mul(x = var_3786_cast_fp16, y = var_3787_to_fp16)[name = string("attn_logits_25_cast_fp16")]; + tensor attn_logits_27_cast_fp16 = add(x = attn_logits_25_cast_fp16, y = causal_mask)[name = string("attn_logits_27_cast_fp16")]; + int32 var_3814 = const()[name = string("op_3814"), val = int32(-1)]; + tensor var_3816_cast_fp16 = softmax(axis = var_3814, x = attn_logits_27_cast_fp16)[name = string("op_3816_cast_fp16")]; + tensor concat_120 = const()[name = string("concat_120"), val = tensor([12, 64, 2048])]; + tensor reshape_18_cast_fp16 = reshape(shape = concat_120, x = var_3816_cast_fp16)[name = string("reshape_18_cast_fp16")]; + tensor concat_121 = const()[name = string("concat_121"), val = tensor([12, 2048, 128])]; + tensor reshape_19_cast_fp16 = reshape(shape = concat_121, x = x_107_cast_fp16)[name = string("reshape_19_cast_fp16")]; + bool matmul_6_transpose_x_0 = const()[name = string("matmul_6_transpose_x_0"), val = bool(false)]; + bool matmul_6_transpose_y_0 = const()[name = string("matmul_6_transpose_y_0"), val = bool(false)]; + tensor matmul_6_cast_fp16 = matmul(transpose_x = matmul_6_transpose_x_0, transpose_y = matmul_6_transpose_y_0, x = reshape_18_cast_fp16, y = reshape_19_cast_fp16)[name = string("matmul_6_cast_fp16")]; + tensor concat_125 = const()[name = string("concat_125"), val = tensor([1, 12, 64, 128])]; + tensor reshape_20_cast_fp16 = reshape(shape = concat_125, x = matmul_6_cast_fp16)[name = string("reshape_20_cast_fp16")]; + tensor var_3843_perm_0 = const()[name = string("op_3843_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_3862 = const()[name = string("op_3862"), val = tensor([1, 64, 1536])]; + tensor var_3843 = transpose(perm = var_3843_perm_0, x = reshape_20_cast_fp16)[name = string("transpose_22")]; + tensor attn_output_65 = reshape(shape = var_3862, x = var_3843)[name = string("attn_output_65")]; + tensor var_3867 = const()[name = string("op_3867"), val = tensor([0, 2, 1])]; + tensor squeeze_6_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(319214784))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(320984320))))[name = string("squeeze_6_palettized")]; + string var_3883_pad_type_0 = const()[name = string("op_3883_pad_type_0"), val = string("valid")]; + int32 var_3883_groups_0 = const()[name = string("op_3883_groups_0"), val = int32(1)]; + tensor var_3883_strides_0 = const()[name = string("op_3883_strides_0"), val = tensor([1])]; + tensor var_3883_pad_0 = const()[name = string("op_3883_pad_0"), val = tensor([0, 0])]; + tensor var_3883_dilations_0 = const()[name = string("op_3883_dilations_0"), val = tensor([1])]; + tensor var_3868 = transpose(perm = var_3867, x = attn_output_65)[name = string("transpose_21")]; + tensor var_3883 = conv(dilations = var_3883_dilations_0, groups = var_3883_groups_0, pad = var_3883_pad_0, pad_type = var_3883_pad_type_0, strides = var_3883_strides_0, weight = squeeze_6_palettized, x = var_3868)[name = string("op_3883")]; + tensor var_3887 = const()[name = string("op_3887"), val = tensor([0, 2, 1])]; + tensor attn_output_69 = transpose(perm = var_3887, x = var_3883)[name = string("transpose_20")]; + tensor hidden_states_41_cast_fp16 = add(x = hidden_states_37_cast_fp16, y = attn_output_69)[name = string("hidden_states_41_cast_fp16")]; + int32 var_3900 = const()[name = string("op_3900"), val = int32(-1)]; + fp16 const_179_promoted_to_fp16 = const()[name = string("const_179_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3902_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_179_promoted_to_fp16)[name = string("op_3902_cast_fp16")]; + bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; + tensor input_91_cast_fp16 = concat(axis = var_3900, interleave = input_91_interleave_0, values = (hidden_states_41_cast_fp16, var_3902_cast_fp16))[name = string("input_91_cast_fp16")]; + tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; + fp16 var_3897_to_fp16 = const()[name = string("op_3897_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_3897_to_fp16, x = input_91_cast_fp16)[name = string("normed_53_cast_fp16")]; + tensor normed_55_begin_0 = const()[name = string("normed_55_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_55_end_0 = const()[name = string("normed_55_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_55_end_mask_0 = const()[name = string("normed_55_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_55_cast_fp16 = slice_by_index(begin = normed_55_begin_0, end = normed_55_end_0, end_mask = normed_55_end_mask_0, x = normed_53_cast_fp16)[name = string("normed_55_cast_fp16")]; + tensor const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321033536)))]; + tensor x_109_cast_fp16 = mul(x = normed_55_cast_fp16, y = const_182_promoted_to_fp16)[name = string("x_109_cast_fp16")]; + tensor var_3927 = const()[name = string("op_3927"), val = tensor([0, 2, 1])]; + tensor input_93_axes_0 = const()[name = string("input_93_axes_0"), val = tensor([2])]; + tensor var_3928 = transpose(perm = var_3927, x = x_109_cast_fp16)[name = string("transpose_19")]; + tensor input_93 = expand_dims(axes = input_93_axes_0, x = var_3928)[name = string("input_93")]; + string input_95_pad_type_0 = const()[name = string("input_95_pad_type_0"), val = string("valid")]; + tensor input_95_strides_0 = const()[name = string("input_95_strides_0"), val = tensor([1, 1])]; + tensor input_95_pad_0 = const()[name = string("input_95_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_95_dilations_0 = const()[name = string("input_95_dilations_0"), val = tensor([1, 1])]; + int32 input_95_groups_0 = const()[name = string("input_95_groups_0"), val = int32(1)]; + tensor input_95 = conv(dilations = input_95_dilations_0, groups = input_95_groups_0, pad = input_95_pad_0, pad_type = input_95_pad_type_0, strides = input_95_strides_0, weight = model_model_layers_16_mlp_gate_proj_weight_palettized, x = input_93)[name = string("input_95")]; + string b_13_pad_type_0 = const()[name = string("b_13_pad_type_0"), val = string("valid")]; + tensor b_13_strides_0 = const()[name = string("b_13_strides_0"), val = tensor([1, 1])]; + tensor b_13_pad_0 = const()[name = string("b_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_13_dilations_0 = const()[name = string("b_13_dilations_0"), val = tensor([1, 1])]; + int32 b_13_groups_0 = const()[name = string("b_13_groups_0"), val = int32(1)]; + tensor b_13 = conv(dilations = b_13_dilations_0, groups = b_13_groups_0, pad = b_13_pad_0, pad_type = b_13_pad_type_0, strides = b_13_strides_0, weight = model_model_layers_16_mlp_up_proj_weight_palettized, x = input_93)[name = string("b_13")]; + tensor c_13 = silu(x = input_95)[name = string("c_13")]; + tensor input_97 = mul(x = c_13, y = b_13)[name = string("input_97")]; + string e_13_pad_type_0 = const()[name = string("e_13_pad_type_0"), val = string("valid")]; + tensor e_13_strides_0 = const()[name = string("e_13_strides_0"), val = tensor([1, 1])]; + tensor e_13_pad_0 = const()[name = string("e_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_13_dilations_0 = const()[name = string("e_13_dilations_0"), val = tensor([1, 1])]; + int32 e_13_groups_0 = const()[name = string("e_13_groups_0"), val = int32(1)]; + tensor e_13 = conv(dilations = e_13_dilations_0, groups = e_13_groups_0, pad = e_13_pad_0, pad_type = e_13_pad_type_0, strides = e_13_strides_0, weight = model_model_layers_16_mlp_down_proj_weight_palettized, x = input_97)[name = string("e_13")]; + tensor var_3950_axes_0 = const()[name = string("op_3950_axes_0"), val = tensor([2])]; + tensor var_3950 = squeeze(axes = var_3950_axes_0, x = e_13)[name = string("op_3950")]; + tensor var_3951 = const()[name = string("op_3951"), val = tensor([0, 2, 1])]; + tensor var_3952 = transpose(perm = var_3951, x = var_3950)[name = string("transpose_18")]; + tensor hidden_states_43_cast_fp16 = add(x = hidden_states_41_cast_fp16, y = var_3952)[name = string("hidden_states_43_cast_fp16")]; + int32 var_3964 = const()[name = string("op_3964"), val = int32(-1)]; + fp16 const_183_promoted_to_fp16 = const()[name = string("const_183_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3966_cast_fp16 = mul(x = hidden_states_43_cast_fp16, y = const_183_promoted_to_fp16)[name = string("op_3966_cast_fp16")]; + bool input_99_interleave_0 = const()[name = string("input_99_interleave_0"), val = bool(false)]; + tensor input_99_cast_fp16 = concat(axis = var_3964, interleave = input_99_interleave_0, values = (hidden_states_43_cast_fp16, var_3966_cast_fp16))[name = string("input_99_cast_fp16")]; + tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; + fp16 var_3961_to_fp16 = const()[name = string("op_3961_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_3961_to_fp16, x = input_99_cast_fp16)[name = string("normed_57_cast_fp16")]; + tensor normed_59_begin_0 = const()[name = string("normed_59_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_59_end_0 = const()[name = string("normed_59_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_59_end_mask_0 = const()[name = string("normed_59_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_59_cast_fp16 = slice_by_index(begin = normed_59_begin_0, end = normed_59_end_0, end_mask = normed_59_end_mask_0, x = normed_57_cast_fp16)[name = string("normed_59_cast_fp16")]; + tensor const_186_promoted_to_fp16 = const()[name = string("const_186_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321036672)))]; + tensor hidden_states_45_cast_fp16 = mul(x = normed_59_cast_fp16, y = const_186_promoted_to_fp16)[name = string("hidden_states_45_cast_fp16")]; + tensor var_3989 = const()[name = string("op_3989"), val = tensor([0, 2, 1])]; + tensor var_3992_axes_0 = const()[name = string("op_3992_axes_0"), val = tensor([2])]; + tensor var_3990_cast_fp16 = transpose(perm = var_3989, x = hidden_states_45_cast_fp16)[name = string("transpose_17")]; + tensor var_3992_cast_fp16 = expand_dims(axes = var_3992_axes_0, x = var_3990_cast_fp16)[name = string("op_3992_cast_fp16")]; + string query_states_43_pad_type_0 = const()[name = string("query_states_43_pad_type_0"), val = string("valid")]; + tensor query_states_43_strides_0 = const()[name = string("query_states_43_strides_0"), val = tensor([1, 1])]; + tensor query_states_43_pad_0 = const()[name = string("query_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_43_dilations_0 = const()[name = string("query_states_43_dilations_0"), val = tensor([1, 1])]; + int32 query_states_43_groups_0 = const()[name = string("query_states_43_groups_0"), val = int32(1)]; + tensor query_states_43 = conv(bias = model_model_layers_17_self_attn_q_proj_bias, dilations = query_states_43_dilations_0, groups = query_states_43_groups_0, pad = query_states_43_pad_0, pad_type = query_states_43_pad_type_0, strides = query_states_43_strides_0, weight = model_model_layers_17_self_attn_q_proj_weight_palettized, x = var_3992_cast_fp16)[name = string("query_states_43")]; + string key_states_57_pad_type_0 = const()[name = string("key_states_57_pad_type_0"), val = string("valid")]; + tensor key_states_57_strides_0 = const()[name = string("key_states_57_strides_0"), val = tensor([1, 1])]; + tensor key_states_57_pad_0 = const()[name = string("key_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_57_dilations_0 = const()[name = string("key_states_57_dilations_0"), val = tensor([1, 1])]; + int32 key_states_57_groups_0 = const()[name = string("key_states_57_groups_0"), val = int32(1)]; + tensor key_states_57 = conv(bias = model_model_layers_17_self_attn_k_proj_bias, dilations = key_states_57_dilations_0, groups = key_states_57_groups_0, pad = key_states_57_pad_0, pad_type = key_states_57_pad_type_0, strides = key_states_57_strides_0, weight = model_model_layers_17_self_attn_k_proj_weight_palettized, x = var_3992_cast_fp16)[name = string("key_states_57")]; + string value_states_57_pad_type_0 = const()[name = string("value_states_57_pad_type_0"), val = string("valid")]; + tensor value_states_57_strides_0 = const()[name = string("value_states_57_strides_0"), val = tensor([1, 1])]; + tensor value_states_57_pad_0 = const()[name = string("value_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_57_dilations_0 = const()[name = string("value_states_57_dilations_0"), val = tensor([1, 1])]; + int32 value_states_57_groups_0 = const()[name = string("value_states_57_groups_0"), val = int32(1)]; + tensor value_states_57 = conv(bias = model_model_layers_17_self_attn_v_proj_bias, dilations = value_states_57_dilations_0, groups = value_states_57_groups_0, pad = value_states_57_pad_0, pad_type = value_states_57_pad_type_0, strides = value_states_57_strides_0, weight = model_model_layers_17_self_attn_v_proj_weight_palettized, x = var_3992_cast_fp16)[name = string("value_states_57")]; + tensor var_4034 = const()[name = string("op_4034"), val = tensor([1, 12, 128, 64])]; + tensor var_4035 = reshape(shape = var_4034, x = query_states_43)[name = string("op_4035")]; + tensor var_4040 = const()[name = string("op_4040"), val = tensor([0, 1, 3, 2])]; + tensor var_4045 = const()[name = string("op_4045"), val = tensor([1, 2, 128, 64])]; + tensor var_4046 = reshape(shape = var_4045, x = key_states_57)[name = string("op_4046")]; + tensor var_4051 = const()[name = string("op_4051"), val = tensor([0, 1, 3, 2])]; + tensor var_4056 = const()[name = string("op_4056"), val = tensor([1, 2, 128, 64])]; + tensor var_4057 = reshape(shape = var_4056, x = value_states_57)[name = string("op_4057")]; + tensor var_4062 = const()[name = string("op_4062"), val = tensor([0, 1, 3, 2])]; + tensor q_29 = transpose(perm = var_4040, x = var_4035)[name = string("transpose_16")]; + tensor var_4076 = mul(x = q_29, y = cos_5)[name = string("op_4076")]; + tensor x1_29_begin_0 = const()[name = string("x1_29_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_29_end_0 = const()[name = string("x1_29_end_0"), val = tensor([1, 12, 64, 64])]; + tensor x1_29_end_mask_0 = const()[name = string("x1_29_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_29 = slice_by_index(begin = x1_29_begin_0, end = x1_29_end_0, end_mask = x1_29_end_mask_0, x = q_29)[name = string("x1_29")]; + tensor x2_29_begin_0 = const()[name = string("x2_29_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_29_end_0 = const()[name = string("x2_29_end_0"), val = tensor([1, 12, 64, 128])]; + tensor x2_29_end_mask_0 = const()[name = string("x2_29_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_29 = slice_by_index(begin = x2_29_begin_0, end = x2_29_end_0, end_mask = x2_29_end_mask_0, x = q_29)[name = string("x2_29")]; + fp16 const_190_promoted = const()[name = string("const_190_promoted"), val = fp16(-0x1p+0)]; + tensor var_4097 = mul(x = x2_29, y = const_190_promoted)[name = string("op_4097")]; + int32 var_4099 = const()[name = string("op_4099"), val = int32(-1)]; + bool var_4100_interleave_0 = const()[name = string("op_4100_interleave_0"), val = bool(false)]; + tensor var_4100 = concat(axis = var_4099, interleave = var_4100_interleave_0, values = (var_4097, x1_29))[name = string("op_4100")]; + tensor var_4101 = mul(x = var_4100, y = sin_5)[name = string("op_4101")]; + tensor query_states_45 = add(x = var_4076, y = var_4101)[name = string("query_states_45")]; + tensor k_29 = transpose(perm = var_4051, x = var_4046)[name = string("transpose_15")]; + tensor var_4104 = mul(x = k_29, y = cos_5)[name = string("op_4104")]; + tensor x1_31_begin_0 = const()[name = string("x1_31_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_31_end_0 = const()[name = string("x1_31_end_0"), val = tensor([1, 2, 64, 64])]; + tensor x1_31_end_mask_0 = const()[name = string("x1_31_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_31 = slice_by_index(begin = x1_31_begin_0, end = x1_31_end_0, end_mask = x1_31_end_mask_0, x = k_29)[name = string("x1_31")]; + tensor x2_31_begin_0 = const()[name = string("x2_31_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_31_end_0 = const()[name = string("x2_31_end_0"), val = tensor([1, 2, 64, 128])]; + tensor x2_31_end_mask_0 = const()[name = string("x2_31_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_31 = slice_by_index(begin = x2_31_begin_0, end = x2_31_end_0, end_mask = x2_31_end_mask_0, x = k_29)[name = string("x2_31")]; + fp16 const_193_promoted = const()[name = string("const_193_promoted"), val = fp16(-0x1p+0)]; + tensor var_4125 = mul(x = x2_31, y = const_193_promoted)[name = string("op_4125")]; + int32 var_4127 = const()[name = string("op_4127"), val = int32(-1)]; + bool var_4128_interleave_0 = const()[name = string("op_4128_interleave_0"), val = bool(false)]; + tensor var_4128 = concat(axis = var_4127, interleave = var_4128_interleave_0, values = (var_4125, x1_31))[name = string("op_4128")]; + tensor var_4129 = mul(x = var_4128, y = sin_5)[name = string("op_4129")]; + tensor key_states_59 = add(x = var_4104, y = var_4129)[name = string("key_states_59")]; + tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([17])]; + tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; + tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; + tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([18])]; + int32 concat_128_axis_0 = const()[name = string("concat_128_axis_0"), val = int32(0)]; + bool concat_128_interleave_0 = const()[name = string("concat_128_interleave_0"), val = bool(false)]; + tensor concat_128 = concat(axis = concat_128_axis_0, interleave = concat_128_interleave_0, values = (expand_dims_84, expand_dims_85, current_pos, expand_dims_87))[name = string("concat_128")]; + tensor concat_129_values1_0 = const()[name = string("concat_129_values1_0"), val = tensor([0])]; + tensor concat_129_values3_0 = const()[name = string("concat_129_values3_0"), val = tensor([0])]; + int32 concat_129_axis_0 = const()[name = string("concat_129_axis_0"), val = int32(0)]; + bool concat_129_interleave_0 = const()[name = string("concat_129_interleave_0"), val = bool(false)]; + tensor concat_129 = concat(axis = concat_129_axis_0, interleave = concat_129_interleave_0, values = (expand_dims_88, concat_129_values1_0, var_616, concat_129_values3_0))[name = string("concat_129")]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_128, begin_mask = model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0, end = concat_129, end_mask = model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_15_stride_0, update = key_states_59, x = coreml_update_state_31)[name = string("model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_32_write_state")]; + tensor coreml_update_state_32 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_32")]; + tensor expand_dims_90 = const()[name = string("expand_dims_90"), val = tensor([45])]; + tensor expand_dims_91 = const()[name = string("expand_dims_91"), val = tensor([0])]; + tensor expand_dims_93 = const()[name = string("expand_dims_93"), val = tensor([0])]; + tensor expand_dims_94 = const()[name = string("expand_dims_94"), val = tensor([46])]; + int32 concat_132_axis_0 = const()[name = string("concat_132_axis_0"), val = int32(0)]; + bool concat_132_interleave_0 = const()[name = string("concat_132_interleave_0"), val = bool(false)]; + tensor concat_132 = concat(axis = concat_132_axis_0, interleave = concat_132_interleave_0, values = (expand_dims_90, expand_dims_91, current_pos, expand_dims_93))[name = string("concat_132")]; + tensor concat_133_values1_0 = const()[name = string("concat_133_values1_0"), val = tensor([0])]; + tensor concat_133_values3_0 = const()[name = string("concat_133_values3_0"), val = tensor([0])]; + int32 concat_133_axis_0 = const()[name = string("concat_133_axis_0"), val = int32(0)]; + bool concat_133_interleave_0 = const()[name = string("concat_133_interleave_0"), val = bool(false)]; + tensor concat_133 = concat(axis = concat_133_axis_0, interleave = concat_133_interleave_0, values = (expand_dims_94, concat_133_values1_0, var_616, concat_133_values3_0))[name = string("concat_133")]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_59 = transpose(perm = var_4062, x = var_4057)[name = string("transpose_14")]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_132, begin_mask = model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0, end = concat_133, end_mask = model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_16_stride_0, update = value_states_59, x = coreml_update_state_32)[name = string("model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_33_write_state")]; + tensor coreml_update_state_33 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_33")]; + tensor var_4200_begin_0 = const()[name = string("op_4200_begin_0"), val = tensor([17, 0, 0, 0])]; + tensor var_4200_end_0 = const()[name = string("op_4200_end_0"), val = tensor([18, 2, 2048, 128])]; + tensor var_4200_end_mask_0 = const()[name = string("op_4200_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4200_cast_fp16 = slice_by_index(begin = var_4200_begin_0, end = var_4200_end_0, end_mask = var_4200_end_mask_0, x = coreml_update_state_33)[name = string("op_4200_cast_fp16")]; + tensor K_layer_cache_15_axes_0 = const()[name = string("K_layer_cache_15_axes_0"), val = tensor([0])]; + tensor K_layer_cache_15_cast_fp16 = squeeze(axes = K_layer_cache_15_axes_0, x = var_4200_cast_fp16)[name = string("K_layer_cache_15_cast_fp16")]; + tensor var_4207_begin_0 = const()[name = string("op_4207_begin_0"), val = tensor([45, 0, 0, 0])]; + tensor var_4207_end_0 = const()[name = string("op_4207_end_0"), val = tensor([46, 2, 2048, 128])]; + tensor var_4207_end_mask_0 = const()[name = string("op_4207_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4207_cast_fp16 = slice_by_index(begin = var_4207_begin_0, end = var_4207_end_0, end_mask = var_4207_end_mask_0, x = coreml_update_state_33)[name = string("op_4207_cast_fp16")]; + tensor V_layer_cache_15_axes_0 = const()[name = string("V_layer_cache_15_axes_0"), val = tensor([0])]; + tensor V_layer_cache_15_cast_fp16 = squeeze(axes = V_layer_cache_15_axes_0, x = var_4207_cast_fp16)[name = string("V_layer_cache_15_cast_fp16")]; + tensor x_115_axes_0 = const()[name = string("x_115_axes_0"), val = tensor([1])]; + tensor x_115_cast_fp16 = expand_dims(axes = x_115_axes_0, x = K_layer_cache_15_cast_fp16)[name = string("x_115_cast_fp16")]; + tensor var_4236 = const()[name = string("op_4236"), val = tensor([1, 6, 1, 1])]; + tensor x_117_cast_fp16 = tile(reps = var_4236, x = x_115_cast_fp16)[name = string("x_117_cast_fp16")]; + tensor var_4248 = const()[name = string("op_4248"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_63_cast_fp16 = reshape(shape = var_4248, x = x_117_cast_fp16)[name = string("key_states_63_cast_fp16")]; + tensor x_121_axes_0 = const()[name = string("x_121_axes_0"), val = tensor([1])]; + tensor x_121_cast_fp16 = expand_dims(axes = x_121_axes_0, x = V_layer_cache_15_cast_fp16)[name = string("x_121_cast_fp16")]; + tensor var_4256 = const()[name = string("op_4256"), val = tensor([1, 6, 1, 1])]; + tensor x_123_cast_fp16 = tile(reps = var_4256, x = x_121_cast_fp16)[name = string("x_123_cast_fp16")]; + bool var_4291_transpose_x_1 = const()[name = string("op_4291_transpose_x_1"), val = bool(false)]; + bool var_4291_transpose_y_1 = const()[name = string("op_4291_transpose_y_1"), val = bool(true)]; + tensor var_4291_cast_fp16 = matmul(transpose_x = var_4291_transpose_x_1, transpose_y = var_4291_transpose_y_1, x = query_states_45, y = key_states_63_cast_fp16)[name = string("op_4291_cast_fp16")]; + fp16 var_4292_to_fp16 = const()[name = string("op_4292_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_29_cast_fp16 = mul(x = var_4291_cast_fp16, y = var_4292_to_fp16)[name = string("attn_logits_29_cast_fp16")]; + tensor attn_logits_31_cast_fp16 = add(x = attn_logits_29_cast_fp16, y = causal_mask)[name = string("attn_logits_31_cast_fp16")]; + int32 var_4319 = const()[name = string("op_4319"), val = int32(-1)]; + tensor var_4321_cast_fp16 = softmax(axis = var_4319, x = attn_logits_31_cast_fp16)[name = string("op_4321_cast_fp16")]; + tensor concat_138 = const()[name = string("concat_138"), val = tensor([12, 64, 2048])]; + tensor reshape_21_cast_fp16 = reshape(shape = concat_138, x = var_4321_cast_fp16)[name = string("reshape_21_cast_fp16")]; + tensor concat_139 = const()[name = string("concat_139"), val = tensor([12, 2048, 128])]; + tensor reshape_22_cast_fp16 = reshape(shape = concat_139, x = x_123_cast_fp16)[name = string("reshape_22_cast_fp16")]; + bool matmul_7_transpose_x_0 = const()[name = string("matmul_7_transpose_x_0"), val = bool(false)]; + bool matmul_7_transpose_y_0 = const()[name = string("matmul_7_transpose_y_0"), val = bool(false)]; + tensor matmul_7_cast_fp16 = matmul(transpose_x = matmul_7_transpose_x_0, transpose_y = matmul_7_transpose_y_0, x = reshape_21_cast_fp16, y = reshape_22_cast_fp16)[name = string("matmul_7_cast_fp16")]; + tensor concat_143 = const()[name = string("concat_143"), val = tensor([1, 12, 64, 128])]; + tensor reshape_23_cast_fp16 = reshape(shape = concat_143, x = matmul_7_cast_fp16)[name = string("reshape_23_cast_fp16")]; + tensor var_4348_perm_0 = const()[name = string("op_4348_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_4367 = const()[name = string("op_4367"), val = tensor([1, 64, 1536])]; + tensor var_4348 = transpose(perm = var_4348_perm_0, x = reshape_23_cast_fp16)[name = string("transpose_13")]; + tensor attn_output_75 = reshape(shape = var_4367, x = var_4348)[name = string("attn_output_75")]; + tensor var_4372 = const()[name = string("op_4372"), val = tensor([0, 2, 1])]; + tensor squeeze_7_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(321039808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(322809344))))[name = string("squeeze_7_palettized")]; + string var_4388_pad_type_0 = const()[name = string("op_4388_pad_type_0"), val = string("valid")]; + int32 var_4388_groups_0 = const()[name = string("op_4388_groups_0"), val = int32(1)]; + tensor var_4388_strides_0 = const()[name = string("op_4388_strides_0"), val = tensor([1])]; + tensor var_4388_pad_0 = const()[name = string("op_4388_pad_0"), val = tensor([0, 0])]; + tensor var_4388_dilations_0 = const()[name = string("op_4388_dilations_0"), val = tensor([1])]; + tensor var_4373 = transpose(perm = var_4372, x = attn_output_75)[name = string("transpose_12")]; + tensor var_4388 = conv(dilations = var_4388_dilations_0, groups = var_4388_groups_0, pad = var_4388_pad_0, pad_type = var_4388_pad_type_0, strides = var_4388_strides_0, weight = squeeze_7_palettized, x = var_4373)[name = string("op_4388")]; + tensor var_4392 = const()[name = string("op_4392"), val = tensor([0, 2, 1])]; + tensor attn_output_79 = transpose(perm = var_4392, x = var_4388)[name = string("transpose_11")]; + tensor hidden_states_47_cast_fp16 = add(x = hidden_states_43_cast_fp16, y = attn_output_79)[name = string("hidden_states_47_cast_fp16")]; + int32 var_4405 = const()[name = string("op_4405"), val = int32(-1)]; + fp16 const_205_promoted_to_fp16 = const()[name = string("const_205_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4407_cast_fp16 = mul(x = hidden_states_47_cast_fp16, y = const_205_promoted_to_fp16)[name = string("op_4407_cast_fp16")]; + bool input_105_interleave_0 = const()[name = string("input_105_interleave_0"), val = bool(false)]; + tensor input_105_cast_fp16 = concat(axis = var_4405, interleave = input_105_interleave_0, values = (hidden_states_47_cast_fp16, var_4407_cast_fp16))[name = string("input_105_cast_fp16")]; + tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; + fp16 var_4402_to_fp16 = const()[name = string("op_4402_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_4402_to_fp16, x = input_105_cast_fp16)[name = string("normed_61_cast_fp16")]; + tensor normed_63_begin_0 = const()[name = string("normed_63_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_63_end_0 = const()[name = string("normed_63_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_63_end_mask_0 = const()[name = string("normed_63_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_63_cast_fp16 = slice_by_index(begin = normed_63_begin_0, end = normed_63_end_0, end_mask = normed_63_end_mask_0, x = normed_61_cast_fp16)[name = string("normed_63_cast_fp16")]; + tensor const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(322858560)))]; + tensor x_125_cast_fp16 = mul(x = normed_63_cast_fp16, y = const_208_promoted_to_fp16)[name = string("x_125_cast_fp16")]; + tensor var_4432 = const()[name = string("op_4432"), val = tensor([0, 2, 1])]; + tensor input_107_axes_0 = const()[name = string("input_107_axes_0"), val = tensor([2])]; + tensor var_4433 = transpose(perm = var_4432, x = x_125_cast_fp16)[name = string("transpose_10")]; + tensor input_107 = expand_dims(axes = input_107_axes_0, x = var_4433)[name = string("input_107")]; + string input_109_pad_type_0 = const()[name = string("input_109_pad_type_0"), val = string("valid")]; + tensor input_109_strides_0 = const()[name = string("input_109_strides_0"), val = tensor([1, 1])]; + tensor input_109_pad_0 = const()[name = string("input_109_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_109_dilations_0 = const()[name = string("input_109_dilations_0"), val = tensor([1, 1])]; + int32 input_109_groups_0 = const()[name = string("input_109_groups_0"), val = int32(1)]; + tensor input_109 = conv(dilations = input_109_dilations_0, groups = input_109_groups_0, pad = input_109_pad_0, pad_type = input_109_pad_type_0, strides = input_109_strides_0, weight = model_model_layers_17_mlp_gate_proj_weight_palettized, x = input_107)[name = string("input_109")]; + string b_15_pad_type_0 = const()[name = string("b_15_pad_type_0"), val = string("valid")]; + tensor b_15_strides_0 = const()[name = string("b_15_strides_0"), val = tensor([1, 1])]; + tensor b_15_pad_0 = const()[name = string("b_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_15_dilations_0 = const()[name = string("b_15_dilations_0"), val = tensor([1, 1])]; + int32 b_15_groups_0 = const()[name = string("b_15_groups_0"), val = int32(1)]; + tensor b_15 = conv(dilations = b_15_dilations_0, groups = b_15_groups_0, pad = b_15_pad_0, pad_type = b_15_pad_type_0, strides = b_15_strides_0, weight = model_model_layers_17_mlp_up_proj_weight_palettized, x = input_107)[name = string("b_15")]; + tensor c_15 = silu(x = input_109)[name = string("c_15")]; + tensor input_111 = mul(x = c_15, y = b_15)[name = string("input_111")]; + string e_15_pad_type_0 = const()[name = string("e_15_pad_type_0"), val = string("valid")]; + tensor e_15_strides_0 = const()[name = string("e_15_strides_0"), val = tensor([1, 1])]; + tensor e_15_pad_0 = const()[name = string("e_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_15_dilations_0 = const()[name = string("e_15_dilations_0"), val = tensor([1, 1])]; + int32 e_15_groups_0 = const()[name = string("e_15_groups_0"), val = int32(1)]; + tensor e_15 = conv(dilations = e_15_dilations_0, groups = e_15_groups_0, pad = e_15_pad_0, pad_type = e_15_pad_type_0, strides = e_15_strides_0, weight = model_model_layers_17_mlp_down_proj_weight_palettized, x = input_111)[name = string("e_15")]; + tensor var_4455_axes_0 = const()[name = string("op_4455_axes_0"), val = tensor([2])]; + tensor var_4455 = squeeze(axes = var_4455_axes_0, x = e_15)[name = string("op_4455")]; + tensor var_4456 = const()[name = string("op_4456"), val = tensor([0, 2, 1])]; + tensor var_4457 = transpose(perm = var_4456, x = var_4455)[name = string("transpose_9")]; + tensor hidden_states_49_cast_fp16 = add(x = hidden_states_47_cast_fp16, y = var_4457)[name = string("hidden_states_49_cast_fp16")]; + int32 var_4469 = const()[name = string("op_4469"), val = int32(-1)]; + fp16 const_209_promoted_to_fp16 = const()[name = string("const_209_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4471_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_209_promoted_to_fp16)[name = string("op_4471_cast_fp16")]; + bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; + tensor input_113_cast_fp16 = concat(axis = var_4469, interleave = input_113_interleave_0, values = (hidden_states_49_cast_fp16, var_4471_cast_fp16))[name = string("input_113_cast_fp16")]; + tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; + fp16 var_4466_to_fp16 = const()[name = string("op_4466_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_4466_to_fp16, x = input_113_cast_fp16)[name = string("normed_65_cast_fp16")]; + tensor normed_67_begin_0 = const()[name = string("normed_67_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_67_end_0 = const()[name = string("normed_67_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_67_end_mask_0 = const()[name = string("normed_67_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_67_cast_fp16 = slice_by_index(begin = normed_67_begin_0, end = normed_67_end_0, end_mask = normed_67_end_mask_0, x = normed_65_cast_fp16)[name = string("normed_67_cast_fp16")]; + tensor const_212_promoted_to_fp16 = const()[name = string("const_212_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(322861696)))]; + tensor hidden_states_51_cast_fp16 = mul(x = normed_67_cast_fp16, y = const_212_promoted_to_fp16)[name = string("hidden_states_51_cast_fp16")]; + tensor var_4494 = const()[name = string("op_4494"), val = tensor([0, 2, 1])]; + tensor var_4497_axes_0 = const()[name = string("op_4497_axes_0"), val = tensor([2])]; + tensor var_4495_cast_fp16 = transpose(perm = var_4494, x = hidden_states_51_cast_fp16)[name = string("transpose_8")]; + tensor var_4497_cast_fp16 = expand_dims(axes = var_4497_axes_0, x = var_4495_cast_fp16)[name = string("op_4497_cast_fp16")]; + string query_states_49_pad_type_0 = const()[name = string("query_states_49_pad_type_0"), val = string("valid")]; + tensor query_states_49_strides_0 = const()[name = string("query_states_49_strides_0"), val = tensor([1, 1])]; + tensor query_states_49_pad_0 = const()[name = string("query_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_49_dilations_0 = const()[name = string("query_states_49_dilations_0"), val = tensor([1, 1])]; + int32 query_states_49_groups_0 = const()[name = string("query_states_49_groups_0"), val = int32(1)]; + tensor query_states_49 = conv(bias = model_model_layers_18_self_attn_q_proj_bias, dilations = query_states_49_dilations_0, groups = query_states_49_groups_0, pad = query_states_49_pad_0, pad_type = query_states_49_pad_type_0, strides = query_states_49_strides_0, weight = model_model_layers_18_self_attn_q_proj_weight_palettized, x = var_4497_cast_fp16)[name = string("query_states_49")]; + string key_states_65_pad_type_0 = const()[name = string("key_states_65_pad_type_0"), val = string("valid")]; + tensor key_states_65_strides_0 = const()[name = string("key_states_65_strides_0"), val = tensor([1, 1])]; + tensor key_states_65_pad_0 = const()[name = string("key_states_65_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_65_dilations_0 = const()[name = string("key_states_65_dilations_0"), val = tensor([1, 1])]; + int32 key_states_65_groups_0 = const()[name = string("key_states_65_groups_0"), val = int32(1)]; + tensor key_states_65 = conv(bias = model_model_layers_18_self_attn_k_proj_bias, dilations = key_states_65_dilations_0, groups = key_states_65_groups_0, pad = key_states_65_pad_0, pad_type = key_states_65_pad_type_0, strides = key_states_65_strides_0, weight = model_model_layers_18_self_attn_k_proj_weight_palettized, x = var_4497_cast_fp16)[name = string("key_states_65")]; + string value_states_65_pad_type_0 = const()[name = string("value_states_65_pad_type_0"), val = string("valid")]; + tensor value_states_65_strides_0 = const()[name = string("value_states_65_strides_0"), val = tensor([1, 1])]; + tensor value_states_65_pad_0 = const()[name = string("value_states_65_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_65_dilations_0 = const()[name = string("value_states_65_dilations_0"), val = tensor([1, 1])]; + int32 value_states_65_groups_0 = const()[name = string("value_states_65_groups_0"), val = int32(1)]; + tensor value_states_65 = conv(bias = model_model_layers_18_self_attn_v_proj_bias, dilations = value_states_65_dilations_0, groups = value_states_65_groups_0, pad = value_states_65_pad_0, pad_type = value_states_65_pad_type_0, strides = value_states_65_strides_0, weight = model_model_layers_18_self_attn_v_proj_weight_palettized, x = var_4497_cast_fp16)[name = string("value_states_65")]; + tensor var_4539 = const()[name = string("op_4539"), val = tensor([1, 12, 128, 64])]; + tensor var_4540 = reshape(shape = var_4539, x = query_states_49)[name = string("op_4540")]; + tensor var_4545 = const()[name = string("op_4545"), val = tensor([0, 1, 3, 2])]; + tensor var_4550 = const()[name = string("op_4550"), val = tensor([1, 2, 128, 64])]; + tensor var_4551 = reshape(shape = var_4550, x = key_states_65)[name = string("op_4551")]; + tensor var_4556 = const()[name = string("op_4556"), val = tensor([0, 1, 3, 2])]; + tensor var_4561 = const()[name = string("op_4561"), val = tensor([1, 2, 128, 64])]; + tensor var_4562 = reshape(shape = var_4561, x = value_states_65)[name = string("op_4562")]; + tensor var_4567 = const()[name = string("op_4567"), val = tensor([0, 1, 3, 2])]; + tensor q_33 = transpose(perm = var_4545, x = var_4540)[name = string("transpose_7")]; + tensor var_4581 = mul(x = q_33, y = cos_5)[name = string("op_4581")]; + tensor x1_33_begin_0 = const()[name = string("x1_33_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_33_end_0 = const()[name = string("x1_33_end_0"), val = tensor([1, 12, 64, 64])]; + tensor x1_33_end_mask_0 = const()[name = string("x1_33_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_33 = slice_by_index(begin = x1_33_begin_0, end = x1_33_end_0, end_mask = x1_33_end_mask_0, x = q_33)[name = string("x1_33")]; + tensor x2_33_begin_0 = const()[name = string("x2_33_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_33_end_0 = const()[name = string("x2_33_end_0"), val = tensor([1, 12, 64, 128])]; + tensor x2_33_end_mask_0 = const()[name = string("x2_33_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_33 = slice_by_index(begin = x2_33_begin_0, end = x2_33_end_0, end_mask = x2_33_end_mask_0, x = q_33)[name = string("x2_33")]; + fp16 const_216_promoted = const()[name = string("const_216_promoted"), val = fp16(-0x1p+0)]; + tensor var_4602 = mul(x = x2_33, y = const_216_promoted)[name = string("op_4602")]; + int32 var_4604 = const()[name = string("op_4604"), val = int32(-1)]; + bool var_4605_interleave_0 = const()[name = string("op_4605_interleave_0"), val = bool(false)]; + tensor var_4605 = concat(axis = var_4604, interleave = var_4605_interleave_0, values = (var_4602, x1_33))[name = string("op_4605")]; + tensor var_4606 = mul(x = var_4605, y = sin_5)[name = string("op_4606")]; + tensor query_states_51 = add(x = var_4581, y = var_4606)[name = string("query_states_51")]; + tensor k_33 = transpose(perm = var_4556, x = var_4551)[name = string("transpose_6")]; + tensor var_4609 = mul(x = k_33, y = cos_5)[name = string("op_4609")]; + tensor x1_begin_0 = const()[name = string("x1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_end_0 = const()[name = string("x1_end_0"), val = tensor([1, 2, 64, 64])]; + tensor x1_end_mask_0 = const()[name = string("x1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1 = slice_by_index(begin = x1_begin_0, end = x1_end_0, end_mask = x1_end_mask_0, x = k_33)[name = string("x1")]; + tensor x2_begin_0 = const()[name = string("x2_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_end_0 = const()[name = string("x2_end_0"), val = tensor([1, 2, 64, 128])]; + tensor x2_end_mask_0 = const()[name = string("x2_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2 = slice_by_index(begin = x2_begin_0, end = x2_end_0, end_mask = x2_end_mask_0, x = k_33)[name = string("x2")]; + fp16 const_219_promoted = const()[name = string("const_219_promoted"), val = fp16(-0x1p+0)]; + tensor var_4630 = mul(x = x2, y = const_219_promoted)[name = string("op_4630")]; + int32 var_4632 = const()[name = string("op_4632"), val = int32(-1)]; + bool var_4633_interleave_0 = const()[name = string("op_4633_interleave_0"), val = bool(false)]; + tensor var_4633 = concat(axis = var_4632, interleave = var_4633_interleave_0, values = (var_4630, x1))[name = string("op_4633")]; + tensor var_4634 = mul(x = var_4633, y = sin_5)[name = string("op_4634")]; + tensor key_states_67 = add(x = var_4609, y = var_4634)[name = string("key_states_67")]; + tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([18])]; + tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; + tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; + tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([19])]; + int32 concat_146_axis_0 = const()[name = string("concat_146_axis_0"), val = int32(0)]; + bool concat_146_interleave_0 = const()[name = string("concat_146_interleave_0"), val = bool(false)]; + tensor concat_146 = concat(axis = concat_146_axis_0, interleave = concat_146_interleave_0, values = (expand_dims_96, expand_dims_97, current_pos, expand_dims_99))[name = string("concat_146")]; + tensor concat_147_values1_0 = const()[name = string("concat_147_values1_0"), val = tensor([0])]; + tensor concat_147_values3_0 = const()[name = string("concat_147_values3_0"), val = tensor([0])]; + int32 concat_147_axis_0 = const()[name = string("concat_147_axis_0"), val = int32(0)]; + bool concat_147_interleave_0 = const()[name = string("concat_147_interleave_0"), val = bool(false)]; + tensor concat_147 = concat(axis = concat_147_axis_0, interleave = concat_147_interleave_0, values = (expand_dims_100, concat_147_values1_0, var_616, concat_147_values3_0))[name = string("concat_147")]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_146, begin_mask = model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0, end = concat_147, end_mask = model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_17_stride_0, update = key_states_67, x = coreml_update_state_33)[name = string("model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_34_write_state")]; + tensor coreml_update_state_34 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_34")]; + tensor expand_dims_102 = const()[name = string("expand_dims_102"), val = tensor([46])]; + tensor expand_dims_103 = const()[name = string("expand_dims_103"), val = tensor([0])]; + tensor expand_dims_105 = const()[name = string("expand_dims_105"), val = tensor([0])]; + tensor expand_dims_106 = const()[name = string("expand_dims_106"), val = tensor([47])]; + int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; + bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; + tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_102, expand_dims_103, current_pos, expand_dims_105))[name = string("concat_150")]; + tensor concat_151_values1_0 = const()[name = string("concat_151_values1_0"), val = tensor([0])]; + tensor concat_151_values3_0 = const()[name = string("concat_151_values3_0"), val = tensor([0])]; + int32 concat_151_axis_0 = const()[name = string("concat_151_axis_0"), val = int32(0)]; + bool concat_151_interleave_0 = const()[name = string("concat_151_interleave_0"), val = bool(false)]; + tensor concat_151 = concat(axis = concat_151_axis_0, interleave = concat_151_interleave_0, values = (expand_dims_106, concat_151_values1_0, var_616, concat_151_values3_0))[name = string("concat_151")]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_67 = transpose(perm = var_4567, x = var_4562)[name = string("transpose_5")]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_150, begin_mask = model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0, end = concat_151, end_mask = model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_18_stride_0, update = value_states_67, x = coreml_update_state_34)[name = string("model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_35_write_state")]; + tensor coreml_update_state_35 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_35")]; + tensor var_4705_begin_0 = const()[name = string("op_4705_begin_0"), val = tensor([18, 0, 0, 0])]; + tensor var_4705_end_0 = const()[name = string("op_4705_end_0"), val = tensor([19, 2, 2048, 128])]; + tensor var_4705_end_mask_0 = const()[name = string("op_4705_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4705_cast_fp16 = slice_by_index(begin = var_4705_begin_0, end = var_4705_end_0, end_mask = var_4705_end_mask_0, x = coreml_update_state_35)[name = string("op_4705_cast_fp16")]; + tensor K_layer_cache_axes_0 = const()[name = string("K_layer_cache_axes_0"), val = tensor([0])]; + tensor K_layer_cache_cast_fp16 = squeeze(axes = K_layer_cache_axes_0, x = var_4705_cast_fp16)[name = string("K_layer_cache_cast_fp16")]; + tensor var_4712_begin_0 = const()[name = string("op_4712_begin_0"), val = tensor([46, 0, 0, 0])]; + tensor var_4712_end_0 = const()[name = string("op_4712_end_0"), val = tensor([47, 2, 2048, 128])]; + tensor var_4712_end_mask_0 = const()[name = string("op_4712_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4712_cast_fp16 = slice_by_index(begin = var_4712_begin_0, end = var_4712_end_0, end_mask = var_4712_end_mask_0, x = coreml_update_state_35)[name = string("op_4712_cast_fp16")]; + tensor V_layer_cache_axes_0 = const()[name = string("V_layer_cache_axes_0"), val = tensor([0])]; + tensor V_layer_cache_cast_fp16 = squeeze(axes = V_layer_cache_axes_0, x = var_4712_cast_fp16)[name = string("V_layer_cache_cast_fp16")]; + tensor x_131_axes_0 = const()[name = string("x_131_axes_0"), val = tensor([1])]; + tensor x_131_cast_fp16 = expand_dims(axes = x_131_axes_0, x = K_layer_cache_cast_fp16)[name = string("x_131_cast_fp16")]; + tensor var_4741 = const()[name = string("op_4741"), val = tensor([1, 6, 1, 1])]; + tensor x_133_cast_fp16 = tile(reps = var_4741, x = x_131_cast_fp16)[name = string("x_133_cast_fp16")]; + tensor var_4753 = const()[name = string("op_4753"), val = tensor([1, -1, 2048, 128])]; + tensor key_states_cast_fp16 = reshape(shape = var_4753, x = x_133_cast_fp16)[name = string("key_states_cast_fp16")]; + tensor x_137_axes_0 = const()[name = string("x_137_axes_0"), val = tensor([1])]; + tensor x_137_cast_fp16 = expand_dims(axes = x_137_axes_0, x = V_layer_cache_cast_fp16)[name = string("x_137_cast_fp16")]; + tensor var_4761 = const()[name = string("op_4761"), val = tensor([1, 6, 1, 1])]; + tensor x_139_cast_fp16 = tile(reps = var_4761, x = x_137_cast_fp16)[name = string("x_139_cast_fp16")]; + bool var_4796_transpose_x_1 = const()[name = string("op_4796_transpose_x_1"), val = bool(false)]; + bool var_4796_transpose_y_1 = const()[name = string("op_4796_transpose_y_1"), val = bool(true)]; + tensor var_4796_cast_fp16 = matmul(transpose_x = var_4796_transpose_x_1, transpose_y = var_4796_transpose_y_1, x = query_states_51, y = key_states_cast_fp16)[name = string("op_4796_cast_fp16")]; + fp16 var_4797_to_fp16 = const()[name = string("op_4797_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_logits_33_cast_fp16 = mul(x = var_4796_cast_fp16, y = var_4797_to_fp16)[name = string("attn_logits_33_cast_fp16")]; + tensor attn_logits_cast_fp16 = add(x = attn_logits_33_cast_fp16, y = causal_mask)[name = string("attn_logits_cast_fp16")]; + int32 var_4824 = const()[name = string("op_4824"), val = int32(-1)]; + tensor var_4826_cast_fp16 = softmax(axis = var_4824, x = attn_logits_cast_fp16)[name = string("op_4826_cast_fp16")]; + tensor concat_156 = const()[name = string("concat_156"), val = tensor([12, 64, 2048])]; + tensor reshape_24_cast_fp16 = reshape(shape = concat_156, x = var_4826_cast_fp16)[name = string("reshape_24_cast_fp16")]; + tensor concat_157 = const()[name = string("concat_157"), val = tensor([12, 2048, 128])]; + tensor reshape_25_cast_fp16 = reshape(shape = concat_157, x = x_139_cast_fp16)[name = string("reshape_25_cast_fp16")]; + bool matmul_8_transpose_x_0 = const()[name = string("matmul_8_transpose_x_0"), val = bool(false)]; + bool matmul_8_transpose_y_0 = const()[name = string("matmul_8_transpose_y_0"), val = bool(false)]; + tensor matmul_8_cast_fp16 = matmul(transpose_x = matmul_8_transpose_x_0, transpose_y = matmul_8_transpose_y_0, x = reshape_24_cast_fp16, y = reshape_25_cast_fp16)[name = string("matmul_8_cast_fp16")]; + tensor concat_161 = const()[name = string("concat_161"), val = tensor([1, 12, 64, 128])]; + tensor reshape_26_cast_fp16 = reshape(shape = concat_161, x = matmul_8_cast_fp16)[name = string("reshape_26_cast_fp16")]; + tensor var_4853_perm_0 = const()[name = string("op_4853_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_4872 = const()[name = string("op_4872"), val = tensor([1, 64, 1536])]; + tensor var_4853 = transpose(perm = var_4853_perm_0, x = reshape_26_cast_fp16)[name = string("transpose_4")]; + tensor attn_output_85 = reshape(shape = var_4872, x = var_4853)[name = string("attn_output_85")]; + tensor var_4877 = const()[name = string("op_4877"), val = tensor([0, 2, 1])]; + tensor squeeze_8_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(322864832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324634368))))[name = string("squeeze_8_palettized")]; + string var_4893_pad_type_0 = const()[name = string("op_4893_pad_type_0"), val = string("valid")]; + int32 var_4893_groups_0 = const()[name = string("op_4893_groups_0"), val = int32(1)]; + tensor var_4893_strides_0 = const()[name = string("op_4893_strides_0"), val = tensor([1])]; + tensor var_4893_pad_0 = const()[name = string("op_4893_pad_0"), val = tensor([0, 0])]; + tensor var_4893_dilations_0 = const()[name = string("op_4893_dilations_0"), val = tensor([1])]; + tensor var_4878 = transpose(perm = var_4877, x = attn_output_85)[name = string("transpose_3")]; + tensor var_4893 = conv(dilations = var_4893_dilations_0, groups = var_4893_groups_0, pad = var_4893_pad_0, pad_type = var_4893_pad_type_0, strides = var_4893_strides_0, weight = squeeze_8_palettized, x = var_4878)[name = string("op_4893")]; + tensor var_4897 = const()[name = string("op_4897"), val = tensor([0, 2, 1])]; + tensor attn_output = transpose(perm = var_4897, x = var_4893)[name = string("transpose_2")]; + tensor hidden_states_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = attn_output)[name = string("hidden_states_cast_fp16")]; + int32 var_4910 = const()[name = string("op_4910"), val = int32(-1)]; + fp16 const_231_promoted_to_fp16 = const()[name = string("const_231_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4912_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_231_promoted_to_fp16)[name = string("op_4912_cast_fp16")]; + bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; + tensor input_119_cast_fp16 = concat(axis = var_4910, interleave = input_119_interleave_0, values = (hidden_states_cast_fp16, var_4912_cast_fp16))[name = string("input_119_cast_fp16")]; + tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; + fp16 var_4907_to_fp16 = const()[name = string("op_4907_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_4907_to_fp16, x = input_119_cast_fp16)[name = string("normed_69_cast_fp16")]; + tensor normed_begin_0 = const()[name = string("normed_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_end_0 = const()[name = string("normed_end_0"), val = tensor([1, 64, 1536])]; + tensor normed_end_mask_0 = const()[name = string("normed_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_cast_fp16 = slice_by_index(begin = normed_begin_0, end = normed_end_0, end_mask = normed_end_mask_0, x = normed_69_cast_fp16)[name = string("normed_cast_fp16")]; + tensor const_234_promoted_to_fp16 = const()[name = string("const_234_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(324683584)))]; + tensor x_141_cast_fp16 = mul(x = normed_cast_fp16, y = const_234_promoted_to_fp16)[name = string("x_141_cast_fp16")]; + tensor var_4937 = const()[name = string("op_4937"), val = tensor([0, 2, 1])]; + tensor input_121_axes_0 = const()[name = string("input_121_axes_0"), val = tensor([2])]; + tensor var_4938 = transpose(perm = var_4937, x = x_141_cast_fp16)[name = string("transpose_1")]; + tensor input_121 = expand_dims(axes = input_121_axes_0, x = var_4938)[name = string("input_121")]; + string input_123_pad_type_0 = const()[name = string("input_123_pad_type_0"), val = string("valid")]; + tensor input_123_strides_0 = const()[name = string("input_123_strides_0"), val = tensor([1, 1])]; + tensor input_123_pad_0 = const()[name = string("input_123_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_123_dilations_0 = const()[name = string("input_123_dilations_0"), val = tensor([1, 1])]; + int32 input_123_groups_0 = const()[name = string("input_123_groups_0"), val = int32(1)]; + tensor input_123 = conv(dilations = input_123_dilations_0, groups = input_123_groups_0, pad = input_123_pad_0, pad_type = input_123_pad_type_0, strides = input_123_strides_0, weight = model_model_layers_18_mlp_gate_proj_weight_palettized, x = input_121)[name = string("input_123")]; + string b_pad_type_0 = const()[name = string("b_pad_type_0"), val = string("valid")]; + tensor b_strides_0 = const()[name = string("b_strides_0"), val = tensor([1, 1])]; + tensor b_pad_0 = const()[name = string("b_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_dilations_0 = const()[name = string("b_dilations_0"), val = tensor([1, 1])]; + int32 b_groups_0 = const()[name = string("b_groups_0"), val = int32(1)]; + tensor b = conv(dilations = b_dilations_0, groups = b_groups_0, pad = b_pad_0, pad_type = b_pad_type_0, strides = b_strides_0, weight = model_model_layers_18_mlp_up_proj_weight_palettized, x = input_121)[name = string("b")]; + tensor c = silu(x = input_123)[name = string("c")]; + tensor input = mul(x = c, y = b)[name = string("input")]; + string e_pad_type_0 = const()[name = string("e_pad_type_0"), val = string("valid")]; + tensor e_strides_0 = const()[name = string("e_strides_0"), val = tensor([1, 1])]; + tensor e_pad_0 = const()[name = string("e_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_dilations_0 = const()[name = string("e_dilations_0"), val = tensor([1, 1])]; + int32 e_groups_0 = const()[name = string("e_groups_0"), val = int32(1)]; + tensor e = conv(dilations = e_dilations_0, groups = e_groups_0, pad = e_pad_0, pad_type = e_pad_type_0, strides = e_strides_0, weight = model_model_layers_18_mlp_down_proj_weight_palettized, x = input)[name = string("e")]; + tensor var_4960_axes_0 = const()[name = string("op_4960_axes_0"), val = tensor([2])]; + tensor var_4960 = squeeze(axes = var_4960_axes_0, x = e)[name = string("op_4960")]; + tensor var_4961 = const()[name = string("op_4961"), val = tensor([0, 2, 1])]; + tensor var_4962 = transpose(perm = var_4961, x = var_4960)[name = string("transpose_0")]; + tensor output_hidden_states = add(x = hidden_states_cast_fp16, y = var_4962)[name = string("op_4964_cast_fp16")]; + } -> (output_hidden_states); +} \ No newline at end of file