program(1.3) [buildInfo = dict({{"coremlc-component-MIL", "3500.14.1"}, {"coremlc-version", "3500.32.1"}})] { func infer(tensor causal_mask, tensor current_pos, tensor hidden_states, state> model_model_kv_cache_0, tensor position_ids) { tensor model_model_layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2097280))))[name = string("model_model_layers_0_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2105536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2629888))))[name = string("model_model_layers_0_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2632000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3156352))))[name = string("model_model_layers_0_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_0_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3158464))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11547136))))[name = string("model_model_layers_0_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_0_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11579968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19968640))))[name = string("model_model_layers_0_mlp_up_proj_weight_palettized")]; tensor model_model_layers_0_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(20001472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28390144))))[name = string("model_model_layers_0_mlp_down_proj_weight_palettized")]; tensor model_model_layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28398400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30495616))))[name = string("model_model_layers_1_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30503872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31028224))))[name = string("model_model_layers_1_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31030336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31554688))))[name = string("model_model_layers_1_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_1_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31556800))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39945472))))[name = string("model_model_layers_1_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_1_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39978304))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48366976))))[name = string("model_model_layers_1_mlp_up_proj_weight_palettized")]; tensor model_model_layers_1_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48399808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(56788480))))[name = string("model_model_layers_1_mlp_down_proj_weight_palettized")]; tensor model_model_layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(56796736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(58893952))))[name = string("model_model_layers_2_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(58902208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59426560))))[name = string("model_model_layers_2_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59428672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59953024))))[name = string("model_model_layers_2_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_2_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59955136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68343808))))[name = string("model_model_layers_2_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_2_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68376640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76765312))))[name = string("model_model_layers_2_mlp_up_proj_weight_palettized")]; tensor model_model_layers_2_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76798144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85186816))))[name = string("model_model_layers_2_mlp_down_proj_weight_palettized")]; tensor model_model_layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85195072))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87292288))))[name = string("model_model_layers_3_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87300544))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87824896))))[name = string("model_model_layers_3_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87827008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(88351360))))[name = string("model_model_layers_3_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_3_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(88353472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(96742144))))[name = string("model_model_layers_3_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_3_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(96774976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105163648))))[name = string("model_model_layers_3_mlp_up_proj_weight_palettized")]; tensor model_model_layers_3_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105196480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(113585152))))[name = string("model_model_layers_3_mlp_down_proj_weight_palettized")]; tensor model_model_layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(113593408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115690624))))[name = string("model_model_layers_4_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115698880))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116223232))))[name = string("model_model_layers_4_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116225344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116749696))))[name = string("model_model_layers_4_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_4_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116751808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125140480))))[name = string("model_model_layers_4_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_4_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125173312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133561984))))[name = string("model_model_layers_4_mlp_up_proj_weight_palettized")]; tensor model_model_layers_4_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133594816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141983488))))[name = string("model_model_layers_4_mlp_down_proj_weight_palettized")]; tensor model_model_layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141991744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144088960))))[name = string("model_model_layers_5_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144097216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144621568))))[name = string("model_model_layers_5_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144623680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145148032))))[name = string("model_model_layers_5_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_5_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145150144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153538816))))[name = string("model_model_layers_5_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_5_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153571648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(161960320))))[name = string("model_model_layers_5_mlp_up_proj_weight_palettized")]; tensor model_model_layers_5_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(161993152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(170381824))))[name = string("model_model_layers_5_mlp_down_proj_weight_palettized")]; tensor model_model_layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(170390080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172487296))))[name = string("model_model_layers_6_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172495552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(173019904))))[name = string("model_model_layers_6_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(173022016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(173546368))))[name = string("model_model_layers_6_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_6_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(173548480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(181937152))))[name = string("model_model_layers_6_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_6_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(181969984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(190358656))))[name = string("model_model_layers_6_mlp_up_proj_weight_palettized")]; tensor model_model_layers_6_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(190391488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(198780160))))[name = string("model_model_layers_6_mlp_down_proj_weight_palettized")]; tensor model_model_layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(198788416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200885632))))[name = string("model_model_layers_7_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200893888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(201418240))))[name = string("model_model_layers_7_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(201420352))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(201944704))))[name = string("model_model_layers_7_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_7_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(201946816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210335488))))[name = string("model_model_layers_7_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_7_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210368320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218756992))))[name = string("model_model_layers_7_mlp_up_proj_weight_palettized")]; tensor model_model_layers_7_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218789824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(227178496))))[name = string("model_model_layers_7_mlp_down_proj_weight_palettized")]; tensor model_model_layers_8_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(227186752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(229283968))))[name = string("model_model_layers_8_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_8_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(229292224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(229816576))))[name = string("model_model_layers_8_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_8_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(229818688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(230343040))))[name = string("model_model_layers_8_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_8_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(230345152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(238733824))))[name = string("model_model_layers_8_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_8_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(238766656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(247155328))))[name = string("model_model_layers_8_mlp_up_proj_weight_palettized")]; tensor model_model_layers_8_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(247188160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(255576832))))[name = string("model_model_layers_8_mlp_down_proj_weight_palettized")]; tensor model_model_layers_9_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(255585088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257682304))))[name = string("model_model_layers_9_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_9_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257690560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258214912))))[name = string("model_model_layers_9_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_9_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258217024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258741376))))[name = string("model_model_layers_9_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_9_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258743488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(267132160))))[name = string("model_model_layers_9_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_9_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(267164992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275553664))))[name = string("model_model_layers_9_mlp_up_proj_weight_palettized")]; tensor model_model_layers_9_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275586496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283975168))))[name = string("model_model_layers_9_mlp_down_proj_weight_palettized")]; tensor model_model_layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283983424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286080640))))[name = string("model_model_layers_10_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286088896))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286613248))))[name = string("model_model_layers_10_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286615360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(287139712))))[name = string("model_model_layers_10_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_10_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(287141824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295530496))))[name = string("model_model_layers_10_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_10_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295563328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303952000))))[name = string("model_model_layers_10_mlp_up_proj_weight_palettized")]; tensor model_model_layers_10_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303984832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312373504))))[name = string("model_model_layers_10_mlp_down_proj_weight_palettized")]; tensor model_model_layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312381760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(314478976))))[name = string("model_model_layers_11_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(314487232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315011584))))[name = string("model_model_layers_11_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315013696))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315538048))))[name = string("model_model_layers_11_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_11_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315540160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323928832))))[name = string("model_model_layers_11_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_11_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323961664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332350336))))[name = string("model_model_layers_11_mlp_up_proj_weight_palettized")]; tensor model_model_layers_11_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332383168))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(340771840))))[name = string("model_model_layers_11_mlp_down_proj_weight_palettized")]; tensor model_model_layers_12_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(340780096))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342877312))))[name = string("model_model_layers_12_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_12_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342885568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(343409920))))[name = string("model_model_layers_12_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_12_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(343412032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(343936384))))[name = string("model_model_layers_12_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_12_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(343938496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352327168))))[name = string("model_model_layers_12_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_12_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352360000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360748672))))[name = string("model_model_layers_12_mlp_up_proj_weight_palettized")]; tensor model_model_layers_12_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360781504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(369170176))))[name = string("model_model_layers_12_mlp_down_proj_weight_palettized")]; tensor model_model_layers_13_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(369178432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(371275648))))[name = string("model_model_layers_13_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_13_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(371283904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(371808256))))[name = string("model_model_layers_13_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_13_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(371810368))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372334720))))[name = string("model_model_layers_13_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_13_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372336832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(380725504))))[name = string("model_model_layers_13_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_13_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(380758336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(389147008))))[name = string("model_model_layers_13_mlp_up_proj_weight_palettized")]; tensor model_model_layers_13_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(389179840))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397568512))))[name = string("model_model_layers_13_mlp_down_proj_weight_palettized")]; tensor model_model_layers_14_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397576768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399673984))))[name = string("model_model_layers_14_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_14_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399682240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400206592))))[name = string("model_model_layers_14_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_14_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400208704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400733056))))[name = string("model_model_layers_14_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_14_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400735168))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(409123840))))[name = string("model_model_layers_14_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_14_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(409156672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(417545344))))[name = string("model_model_layers_14_mlp_up_proj_weight_palettized")]; tensor model_model_layers_14_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(417578176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(425966848))))[name = string("model_model_layers_14_mlp_down_proj_weight_palettized")]; tensor model_model_layers_15_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(425975104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(428072320))))[name = string("model_model_layers_15_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_15_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(428080576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(428604928))))[name = string("model_model_layers_15_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_15_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(428607040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(429131392))))[name = string("model_model_layers_15_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_15_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(429133504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437522176))))[name = string("model_model_layers_15_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_15_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437555008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(445943680))))[name = string("model_model_layers_15_mlp_up_proj_weight_palettized")]; tensor model_model_layers_15_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(445976512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454365184))))[name = string("model_model_layers_15_mlp_down_proj_weight_palettized")]; int32 var_80 = const()[name = string("op_80"), val = int32(-1)]; int32 var_490_batch_dims_0 = const()[name = string("op_490_batch_dims_0"), val = int32(0)]; bool var_490_validate_indices_0 = const()[name = string("op_490_validate_indices_0"), val = bool(false)]; tensor var_85_to_fp16 = const()[name = string("op_85_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454373440)))]; string current_pos_to_int16_dtype_0 = const()[name = string("current_pos_to_int16_dtype_0"), val = string("int16")]; string cast_166_dtype_0 = const()[name = string("cast_166_dtype_0"), val = string("int32")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor current_pos_to_int16 = cast(dtype = current_pos_to_int16_dtype_0, x = current_pos)[name = string("cast_5")]; tensor cast_166 = cast(dtype = cast_166_dtype_0, x = current_pos_to_int16)[name = string("cast_4")]; tensor greater_equal_0 = greater_equal(x = cast_166, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(8192)]; tensor add_0 = add(x = cast_166, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = cast_166, b = add_0, cond = greater_equal_0)[name = string("select_0")]; string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(8192)]; tensor add_0_1 = add(x = cast_0, y = slice_by_index_0_1)[name = string("add_0_1")]; tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; int32 op_490_cast_fp16_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_490_cast_fp16_cast_uint16_cast_uint16_axis_0"), val = int32(1)]; tensor op_490_cast_fp16_cast_uint16_cast_uint16 = gather(axis = op_490_cast_fp16_cast_uint16_cast_uint16_axis_0, batch_dims = var_490_batch_dims_0, indices = select_0_1, validate_indices = var_490_validate_indices_0, x = var_85_to_fp16)[name = string("op_490_cast_fp16_cast_uint16_cast_uint16")]; tensor var_491 = const()[name = string("op_491"), val = tensor([1, 1, 1, -1])]; tensor sin_1_cast_fp16 = reshape(shape = var_491, x = op_490_cast_fp16_cast_uint16_cast_uint16)[name = string("sin_1_cast_fp16")]; int32 var_495_axis_0 = const()[name = string("op_495_axis_0"), val = int32(1)]; int32 var_495_batch_dims_0 = const()[name = string("op_495_batch_dims_0"), val = int32(0)]; bool var_495_validate_indices_0 = const()[name = string("op_495_validate_indices_0"), val = bool(false)]; tensor var_79_to_fp16 = const()[name = string("op_79_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(455422080)))]; string current_pos_to_uint16_dtype_0 = const()[name = string("current_pos_to_uint16_dtype_0"), val = string("uint16")]; tensor current_pos_to_uint16 = cast(dtype = current_pos_to_uint16_dtype_0, x = current_pos)[name = string("cast_1")]; tensor var_495_cast_fp16_cast_uint16 = gather(axis = var_495_axis_0, batch_dims = var_495_batch_dims_0, indices = current_pos_to_uint16, validate_indices = var_495_validate_indices_0, x = var_79_to_fp16)[name = string("op_495_cast_fp16_cast_uint16")]; tensor var_496 = const()[name = string("op_496"), val = tensor([1, 1, 1, -1])]; tensor cos_1_cast_fp16 = reshape(shape = var_496, x = var_495_cast_fp16_cast_uint16)[name = string("cos_1_cast_fp16")]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_499_cast_fp16 = mul(x = hidden_states, y = const_0_promoted_to_fp16)[name = string("op_499_cast_fp16")]; bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; tensor input_1_cast_fp16 = concat(axis = var_80, interleave = input_1_interleave_0, values = (hidden_states, var_499_cast_fp16))[name = string("input_1_cast_fp16")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_74_to_fp16 = const()[name = string("op_74_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_74_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; tensor normed_3_begin_0 = const()[name = string("normed_3_begin_0"), val = tensor([0, 0, 0])]; tensor normed_3_end_0 = const()[name = string("normed_3_end_0"), val = tensor([1, 1, 2048])]; tensor normed_3_end_mask_0 = const()[name = string("normed_3_end_mask_0"), val = tensor([true, true, false])]; tensor normed_3_cast_fp16 = slice_by_index(begin = normed_3_begin_0, end = normed_3_end_0, end_mask = normed_3_end_mask_0, x = normed_1_cast_fp16)[name = string("normed_3_cast_fp16")]; tensor const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(456470720)))]; tensor hidden_states_3_cast_fp16 = mul(x = normed_3_cast_fp16, y = const_3_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor var_513 = const()[name = string("op_513"), val = tensor([0, 2, 1])]; tensor var_515_axes_0 = const()[name = string("op_515_axes_0"), val = tensor([2])]; tensor var_514_cast_fp16 = transpose(perm = var_513, x = hidden_states_3_cast_fp16)[name = string("transpose_63")]; tensor var_515_cast_fp16 = expand_dims(axes = var_515_axes_0, x = var_514_cast_fp16)[name = string("op_515_cast_fp16")]; string var_522_pad_type_0 = const()[name = string("op_522_pad_type_0"), val = string("valid")]; tensor var_522_strides_0 = const()[name = string("op_522_strides_0"), val = tensor([1, 1])]; tensor var_522_pad_0 = const()[name = string("op_522_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_522_dilations_0 = const()[name = string("op_522_dilations_0"), val = tensor([1, 1])]; int32 var_522_groups_0 = const()[name = string("op_522_groups_0"), val = int32(1)]; tensor var_522 = conv(dilations = var_522_dilations_0, groups = var_522_groups_0, pad = var_522_pad_0, pad_type = var_522_pad_type_0, strides = var_522_strides_0, weight = model_model_layers_0_self_attn_q_proj_weight_palettized, x = var_515_cast_fp16)[name = string("op_522")]; tensor var_523 = const()[name = string("op_523"), val = tensor([1, 32, 1, 64])]; tensor var_524 = reshape(shape = var_523, x = var_522)[name = string("op_524")]; string var_531_pad_type_0 = const()[name = string("op_531_pad_type_0"), val = string("valid")]; tensor var_531_strides_0 = const()[name = string("op_531_strides_0"), val = tensor([1, 1])]; tensor var_531_pad_0 = const()[name = string("op_531_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_531_dilations_0 = const()[name = string("op_531_dilations_0"), val = tensor([1, 1])]; int32 var_531_groups_0 = const()[name = string("op_531_groups_0"), val = int32(1)]; tensor var_531 = conv(dilations = var_531_dilations_0, groups = var_531_groups_0, pad = var_531_pad_0, pad_type = var_531_pad_type_0, strides = var_531_strides_0, weight = model_model_layers_0_self_attn_k_proj_weight_palettized, x = var_515_cast_fp16)[name = string("op_531")]; tensor var_532 = const()[name = string("op_532"), val = tensor([1, 8, 1, 64])]; tensor var_533 = reshape(shape = var_532, x = var_531)[name = string("op_533")]; string var_540_pad_type_0 = const()[name = string("op_540_pad_type_0"), val = string("valid")]; tensor var_540_strides_0 = const()[name = string("op_540_strides_0"), val = tensor([1, 1])]; tensor var_540_pad_0 = const()[name = string("op_540_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_540_dilations_0 = const()[name = string("op_540_dilations_0"), val = tensor([1, 1])]; int32 var_540_groups_0 = const()[name = string("op_540_groups_0"), val = int32(1)]; tensor var_540 = conv(dilations = var_540_dilations_0, groups = var_540_groups_0, pad = var_540_pad_0, pad_type = var_540_pad_type_0, strides = var_540_strides_0, weight = model_model_layers_0_self_attn_v_proj_weight_palettized, x = var_515_cast_fp16)[name = string("op_540")]; tensor var_541 = const()[name = string("op_541"), val = tensor([1, 8, 1, 64])]; tensor var_542 = reshape(shape = var_541, x = var_540)[name = string("op_542")]; tensor x1_1_begin_0 = const()[name = string("x1_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_1_end_0 = const()[name = string("x1_1_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_1_end_mask_0 = const()[name = string("x1_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_1 = slice_by_index(begin = x1_1_begin_0, end = x1_1_end_0, end_mask = x1_1_end_mask_0, x = var_524)[name = string("x1_1")]; tensor x2_1_begin_0 = const()[name = string("x2_1_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_1_end_0 = const()[name = string("x2_1_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_1_end_mask_0 = const()[name = string("x2_1_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_1 = slice_by_index(begin = x2_1_begin_0, end = x2_1_end_0, end_mask = x2_1_end_mask_0, x = var_524)[name = string("x2_1")]; tensor cos_3_begin_0 = const()[name = string("cos_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor cos_3_end_0 = const()[name = string("cos_3_end_0"), val = tensor([1, 1, 1, 32])]; tensor cos_3_end_mask_0 = const()[name = string("cos_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor cos_3_cast_fp16 = slice_by_index(begin = cos_3_begin_0, end = cos_3_end_0, end_mask = cos_3_end_mask_0, x = cos_1_cast_fp16)[name = string("cos_3_cast_fp16")]; tensor sin_3_begin_0 = const()[name = string("sin_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor sin_3_end_0 = const()[name = string("sin_3_end_0"), val = tensor([1, 1, 1, 32])]; tensor sin_3_end_mask_0 = const()[name = string("sin_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor sin_3_cast_fp16 = slice_by_index(begin = sin_3_begin_0, end = sin_3_end_0, end_mask = sin_3_end_mask_0, x = sin_1_cast_fp16)[name = string("sin_3_cast_fp16")]; tensor var_556_cast_fp16 = mul(x = x1_1, y = cos_3_cast_fp16)[name = string("op_556_cast_fp16")]; tensor var_557_cast_fp16 = mul(x = x2_1, y = sin_3_cast_fp16)[name = string("op_557_cast_fp16")]; tensor var_558_cast_fp16 = sub(x = var_556_cast_fp16, y = var_557_cast_fp16)[name = string("op_558_cast_fp16")]; tensor var_559_cast_fp16 = mul(x = x2_1, y = cos_3_cast_fp16)[name = string("op_559_cast_fp16")]; tensor var_560_cast_fp16 = mul(x = x1_1, y = sin_3_cast_fp16)[name = string("op_560_cast_fp16")]; tensor var_561_cast_fp16 = add(x = var_559_cast_fp16, y = var_560_cast_fp16)[name = string("op_561_cast_fp16")]; bool rotated_1_interleave_0 = const()[name = string("rotated_1_interleave_0"), val = bool(false)]; tensor rotated_1_cast_fp16 = concat(axis = var_80, interleave = rotated_1_interleave_0, values = (var_558_cast_fp16, var_561_cast_fp16))[name = string("rotated_1_cast_fp16")]; tensor x1_3_begin_0 = const()[name = string("x1_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_3_end_0 = const()[name = string("x1_3_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_3_end_mask_0 = const()[name = string("x1_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_3 = slice_by_index(begin = x1_3_begin_0, end = x1_3_end_0, end_mask = x1_3_end_mask_0, x = var_533)[name = string("x1_3")]; tensor x2_3_begin_0 = const()[name = string("x2_3_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_3_end_0 = const()[name = string("x2_3_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_3_end_mask_0 = const()[name = string("x2_3_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_3 = slice_by_index(begin = x2_3_begin_0, end = x2_3_end_0, end_mask = x2_3_end_mask_0, x = var_533)[name = string("x2_3")]; tensor var_577_cast_fp16 = mul(x = x1_3, y = cos_3_cast_fp16)[name = string("op_577_cast_fp16")]; tensor var_578_cast_fp16 = mul(x = x2_3, y = sin_3_cast_fp16)[name = string("op_578_cast_fp16")]; tensor var_579_cast_fp16 = sub(x = var_577_cast_fp16, y = var_578_cast_fp16)[name = string("op_579_cast_fp16")]; tensor var_580_cast_fp16 = mul(x = x2_3, y = cos_3_cast_fp16)[name = string("op_580_cast_fp16")]; tensor var_581_cast_fp16 = mul(x = x1_3, y = sin_3_cast_fp16)[name = string("op_581_cast_fp16")]; tensor var_582_cast_fp16 = add(x = var_580_cast_fp16, y = var_581_cast_fp16)[name = string("op_582_cast_fp16")]; bool rotated_3_interleave_0 = const()[name = string("rotated_3_interleave_0"), val = bool(false)]; tensor rotated_3_cast_fp16 = concat(axis = var_80, interleave = rotated_3_interleave_0, values = (var_579_cast_fp16, var_582_cast_fp16))[name = string("rotated_3_cast_fp16")]; int32 var_586 = const()[name = string("op_586"), val = int32(1)]; tensor var_587 = add(x = current_pos, y = var_586)[name = string("op_587")]; tensor read_state_0 = read_state(input = model_model_kv_cache_0)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; int32 concat_2_axis_0 = const()[name = string("concat_2_axis_0"), val = int32(0)]; bool concat_2_interleave_0 = const()[name = string("concat_2_interleave_0"), val = bool(false)]; tensor concat_2 = concat(axis = concat_2_axis_0, interleave = concat_2_interleave_0, values = (expand_dims_0, expand_dims_1, current_pos, expand_dims_3))[name = string("concat_2")]; tensor concat_3_values1_0 = const()[name = string("concat_3_values1_0"), val = tensor([0])]; tensor concat_3_values3_0 = const()[name = string("concat_3_values3_0"), val = tensor([0])]; int32 concat_3_axis_0 = const()[name = string("concat_3_axis_0"), val = int32(0)]; bool concat_3_interleave_0 = const()[name = string("concat_3_interleave_0"), val = bool(false)]; tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (expand_dims_4, concat_3_values1_0, var_587, concat_3_values3_0))[name = string("concat_3")]; tensor model_model_kv_cache_0_internal_tensor_assign_1_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_2, begin_mask = model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0, end = concat_3, end_mask = model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_1_stride_0, update = rotated_3_cast_fp16, x = read_state_0)[name = string("model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_64_write_state")]; tensor coreml_update_state_32 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_64")]; tensor expand_dims_6 = const()[name = string("expand_dims_6"), val = tensor([16])]; tensor expand_dims_7 = const()[name = string("expand_dims_7"), val = tensor([0])]; tensor expand_dims_9 = const()[name = string("expand_dims_9"), val = tensor([0])]; tensor expand_dims_10 = const()[name = string("expand_dims_10"), val = tensor([17])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_6, expand_dims_7, current_pos, expand_dims_9))[name = string("concat_6")]; tensor concat_7_values1_0 = const()[name = string("concat_7_values1_0"), val = tensor([0])]; tensor concat_7_values3_0 = const()[name = string("concat_7_values3_0"), val = tensor([0])]; int32 concat_7_axis_0 = const()[name = string("concat_7_axis_0"), val = int32(0)]; bool concat_7_interleave_0 = const()[name = string("concat_7_interleave_0"), val = bool(false)]; tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (expand_dims_10, concat_7_values1_0, var_587, concat_7_values3_0))[name = string("concat_7")]; tensor model_model_kv_cache_0_internal_tensor_assign_2_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_6, begin_mask = model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0, end = concat_7, end_mask = model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_2_stride_0, update = var_542, x = coreml_update_state_32)[name = string("model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_65_write_state")]; tensor coreml_update_state_33 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_65")]; tensor var_602_begin_0 = const()[name = string("op_602_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_602_end_0 = const()[name = string("op_602_end_0"), val = tensor([1, 8, 4096, 64])]; tensor var_602_end_mask_0 = const()[name = string("op_602_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_602_cast_fp16 = slice_by_index(begin = var_602_begin_0, end = var_602_end_0, end_mask = var_602_end_mask_0, x = coreml_update_state_33)[name = string("op_602_cast_fp16")]; tensor K_layer_cache_1_axes_0 = const()[name = string("K_layer_cache_1_axes_0"), val = tensor([0])]; tensor K_layer_cache_1_cast_fp16 = squeeze(axes = K_layer_cache_1_axes_0, x = var_602_cast_fp16)[name = string("K_layer_cache_1_cast_fp16")]; tensor var_604_begin_0 = const()[name = string("op_604_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_604_end_0 = const()[name = string("op_604_end_0"), val = tensor([17, 8, 4096, 64])]; tensor var_604_end_mask_0 = const()[name = string("op_604_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_604_cast_fp16 = slice_by_index(begin = var_604_begin_0, end = var_604_end_0, end_mask = var_604_end_mask_0, x = coreml_update_state_33)[name = string("op_604_cast_fp16")]; tensor V_layer_cache_1_axes_0 = const()[name = string("V_layer_cache_1_axes_0"), val = tensor([0])]; tensor V_layer_cache_1_cast_fp16 = squeeze(axes = V_layer_cache_1_axes_0, x = var_604_cast_fp16)[name = string("V_layer_cache_1_cast_fp16")]; tensor x_11_axes_0 = const()[name = string("x_11_axes_0"), val = tensor([1])]; tensor x_11_cast_fp16 = expand_dims(axes = x_11_axes_0, x = K_layer_cache_1_cast_fp16)[name = string("x_11_cast_fp16")]; tensor var_613 = const()[name = string("op_613"), val = tensor([1, 4, 1, 1])]; tensor x_13_cast_fp16 = tile(reps = var_613, x = x_11_cast_fp16)[name = string("x_13_cast_fp16")]; tensor var_617 = const()[name = string("op_617"), val = tensor([1, -1, 4096, 64])]; tensor key_states_3_cast_fp16 = reshape(shape = var_617, x = x_13_cast_fp16)[name = string("key_states_3_cast_fp16")]; tensor x_17_axes_0 = const()[name = string("x_17_axes_0"), val = tensor([1])]; tensor x_17_cast_fp16 = expand_dims(axes = x_17_axes_0, x = V_layer_cache_1_cast_fp16)[name = string("x_17_cast_fp16")]; tensor var_620 = const()[name = string("op_620"), val = tensor([1, 4, 1, 1])]; tensor x_19_cast_fp16 = tile(reps = var_620, x = x_17_cast_fp16)[name = string("x_19_cast_fp16")]; tensor var_624 = const()[name = string("op_624"), val = tensor([1, -1, 4096, 64])]; tensor value_states_3_cast_fp16 = reshape(shape = var_624, x = x_19_cast_fp16)[name = string("value_states_3_cast_fp16")]; bool var_627_transpose_x_1 = const()[name = string("op_627_transpose_x_1"), val = bool(false)]; bool var_627_transpose_y_1 = const()[name = string("op_627_transpose_y_1"), val = bool(true)]; tensor var_627_cast_fp16 = matmul(transpose_x = var_627_transpose_x_1, transpose_y = var_627_transpose_y_1, x = rotated_1_cast_fp16, y = key_states_3_cast_fp16)[name = string("op_627_cast_fp16")]; fp16 var_628_to_fp16 = const()[name = string("op_628_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_1_cast_fp16 = mul(x = var_627_cast_fp16, y = var_628_to_fp16)[name = string("attn_weights_1_cast_fp16")]; tensor x_21_cast_fp16 = add(x = attn_weights_1_cast_fp16, y = causal_mask)[name = string("x_21_cast_fp16")]; tensor reduce_max_0_axes_0 = const()[name = string("reduce_max_0_axes_0"), val = tensor([-1])]; bool reduce_max_0_keep_dims_0 = const()[name = string("reduce_max_0_keep_dims_0"), val = bool(true)]; tensor reduce_max_0_cast_fp16 = reduce_max(axes = reduce_max_0_axes_0, keep_dims = reduce_max_0_keep_dims_0, x = x_21_cast_fp16)[name = string("reduce_max_0_cast_fp16")]; tensor x_23_cast_fp16 = sub(x = x_21_cast_fp16, y = reduce_max_0_cast_fp16)[name = string("x_23_cast_fp16")]; tensor exp_x_1_cast_fp16 = exp(x = x_23_cast_fp16)[name = string("exp_x_1_cast_fp16")]; tensor var_639_axes_0 = const()[name = string("op_639_axes_0"), val = tensor([-1])]; bool var_639_keep_dims_0 = const()[name = string("op_639_keep_dims_0"), val = bool(true)]; tensor var_639_cast_fp16 = reduce_sum(axes = var_639_axes_0, keep_dims = var_639_keep_dims_0, x = exp_x_1_cast_fp16)[name = string("op_639_cast_fp16")]; tensor attn_weights_3_cast_fp16 = real_div(x = exp_x_1_cast_fp16, y = var_639_cast_fp16)[name = string("attn_weights_3_cast_fp16")]; bool attn_output_1_transpose_x_0 = const()[name = string("attn_output_1_transpose_x_0"), val = bool(false)]; bool attn_output_1_transpose_y_0 = const()[name = string("attn_output_1_transpose_y_0"), val = bool(false)]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = attn_weights_3_cast_fp16, y = value_states_3_cast_fp16)[name = string("attn_output_1_cast_fp16")]; tensor var_642_perm_0 = const()[name = string("op_642_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_644 = const()[name = string("op_644"), val = tensor([1, 1, 2048])]; tensor var_642_cast_fp16 = transpose(perm = var_642_perm_0, x = attn_output_1_cast_fp16)[name = string("transpose_62")]; tensor input_5_cast_fp16 = reshape(shape = var_644, x = var_642_cast_fp16)[name = string("input_5_cast_fp16")]; tensor model_model_layers_0_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(456474880))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(458572096))))[name = string("model_model_layers_0_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_0_bias_0_to_fp16 = const()[name = string("linear_0_bias_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(458580352)))]; tensor linear_0_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_0_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_5_cast_fp16)[name = string("linear_0_cast_fp16")]; tensor hidden_states_5_cast_fp16 = add(x = hidden_states, y = linear_0_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_650_cast_fp16 = mul(x = hidden_states_5_cast_fp16, y = const_12_promoted_to_fp16)[name = string("op_650_cast_fp16")]; bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; tensor input_7_cast_fp16 = concat(axis = var_80, interleave = input_7_interleave_0, values = (hidden_states_5_cast_fp16, var_650_cast_fp16))[name = string("input_7_cast_fp16")]; tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_74_to_fp16, x = input_7_cast_fp16)[name = string("normed_5_cast_fp16")]; tensor normed_7_begin_0 = const()[name = string("normed_7_begin_0"), val = tensor([0, 0, 0])]; tensor normed_7_end_0 = const()[name = string("normed_7_end_0"), val = tensor([1, 1, 2048])]; tensor normed_7_end_mask_0 = const()[name = string("normed_7_end_mask_0"), val = tensor([true, true, false])]; tensor normed_7_cast_fp16 = slice_by_index(begin = normed_7_begin_0, end = normed_7_end_0, end_mask = normed_7_end_mask_0, x = normed_5_cast_fp16)[name = string("normed_7_cast_fp16")]; tensor const_15_promoted_to_fp16 = const()[name = string("const_15_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(458584512)))]; tensor x_25_cast_fp16 = mul(x = normed_7_cast_fp16, y = const_15_promoted_to_fp16)[name = string("x_25_cast_fp16")]; tensor var_668 = const()[name = string("op_668"), val = tensor([0, 2, 1])]; tensor input_9_axes_0 = const()[name = string("input_9_axes_0"), val = tensor([2])]; tensor var_669 = transpose(perm = var_668, x = x_25_cast_fp16)[name = string("transpose_61")]; tensor input_9 = expand_dims(axes = input_9_axes_0, x = var_669)[name = string("input_9")]; string input_11_pad_type_0 = const()[name = string("input_11_pad_type_0"), val = string("valid")]; tensor input_11_strides_0 = const()[name = string("input_11_strides_0"), val = tensor([1, 1])]; tensor input_11_pad_0 = const()[name = string("input_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_11_dilations_0 = const()[name = string("input_11_dilations_0"), val = tensor([1, 1])]; int32 input_11_groups_0 = const()[name = string("input_11_groups_0"), val = int32(1)]; tensor input_11 = conv(dilations = input_11_dilations_0, groups = input_11_groups_0, pad = input_11_pad_0, pad_type = input_11_pad_type_0, strides = input_11_strides_0, weight = model_model_layers_0_mlp_gate_proj_weight_palettized, x = input_9)[name = string("input_11")]; string up_states_1_pad_type_0 = const()[name = string("up_states_1_pad_type_0"), val = string("valid")]; tensor up_states_1_strides_0 = const()[name = string("up_states_1_strides_0"), val = tensor([1, 1])]; tensor up_states_1_pad_0 = const()[name = string("up_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_1_dilations_0 = const()[name = string("up_states_1_dilations_0"), val = tensor([1, 1])]; int32 up_states_1_groups_0 = const()[name = string("up_states_1_groups_0"), val = int32(1)]; tensor up_states_1 = conv(dilations = up_states_1_dilations_0, groups = up_states_1_groups_0, pad = up_states_1_pad_0, pad_type = up_states_1_pad_type_0, strides = up_states_1_strides_0, weight = model_model_layers_0_mlp_up_proj_weight_palettized, x = input_9)[name = string("up_states_1")]; tensor gate_states_1 = silu(x = input_11)[name = string("gate_states_1")]; tensor input_13 = mul(x = gate_states_1, y = up_states_1)[name = string("input_13")]; string hidden_states_7_pad_type_0 = const()[name = string("hidden_states_7_pad_type_0"), val = string("valid")]; tensor hidden_states_7_strides_0 = const()[name = string("hidden_states_7_strides_0"), val = tensor([1, 1])]; tensor hidden_states_7_pad_0 = const()[name = string("hidden_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_7_dilations_0 = const()[name = string("hidden_states_7_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_7_groups_0 = const()[name = string("hidden_states_7_groups_0"), val = int32(1)]; tensor hidden_states_7 = conv(dilations = hidden_states_7_dilations_0, groups = hidden_states_7_groups_0, pad = hidden_states_7_pad_0, pad_type = hidden_states_7_pad_type_0, strides = hidden_states_7_strides_0, weight = model_model_layers_0_mlp_down_proj_weight_palettized, x = input_13)[name = string("hidden_states_7")]; tensor var_691_axes_0 = const()[name = string("op_691_axes_0"), val = tensor([2])]; tensor var_691 = squeeze(axes = var_691_axes_0, x = hidden_states_7)[name = string("op_691")]; tensor var_692 = const()[name = string("op_692"), val = tensor([0, 2, 1])]; tensor var_693 = transpose(perm = var_692, x = var_691)[name = string("transpose_60")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = var_693)[name = string("hidden_states_9_cast_fp16")]; fp16 const_16_promoted_to_fp16 = const()[name = string("const_16_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_696_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_16_promoted_to_fp16)[name = string("op_696_cast_fp16")]; bool input_15_interleave_0 = const()[name = string("input_15_interleave_0"), val = bool(false)]; tensor input_15_cast_fp16 = concat(axis = var_80, interleave = input_15_interleave_0, values = (hidden_states_9_cast_fp16, var_696_cast_fp16))[name = string("input_15_cast_fp16")]; tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_74_to_fp16, x = input_15_cast_fp16)[name = string("normed_9_cast_fp16")]; tensor normed_11_begin_0 = const()[name = string("normed_11_begin_0"), val = tensor([0, 0, 0])]; tensor normed_11_end_0 = const()[name = string("normed_11_end_0"), val = tensor([1, 1, 2048])]; tensor normed_11_end_mask_0 = const()[name = string("normed_11_end_mask_0"), val = tensor([true, true, false])]; tensor normed_11_cast_fp16 = slice_by_index(begin = normed_11_begin_0, end = normed_11_end_0, end_mask = normed_11_end_mask_0, x = normed_9_cast_fp16)[name = string("normed_11_cast_fp16")]; tensor const_19_promoted_to_fp16 = const()[name = string("const_19_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(458588672)))]; tensor hidden_states_11_cast_fp16 = mul(x = normed_11_cast_fp16, y = const_19_promoted_to_fp16)[name = string("hidden_states_11_cast_fp16")]; tensor var_710 = const()[name = string("op_710"), val = tensor([0, 2, 1])]; tensor var_712_axes_0 = const()[name = string("op_712_axes_0"), val = tensor([2])]; tensor var_711_cast_fp16 = transpose(perm = var_710, x = hidden_states_11_cast_fp16)[name = string("transpose_59")]; tensor var_712_cast_fp16 = expand_dims(axes = var_712_axes_0, x = var_711_cast_fp16)[name = string("op_712_cast_fp16")]; string var_719_pad_type_0 = const()[name = string("op_719_pad_type_0"), val = string("valid")]; tensor var_719_strides_0 = const()[name = string("op_719_strides_0"), val = tensor([1, 1])]; tensor var_719_pad_0 = const()[name = string("op_719_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_719_dilations_0 = const()[name = string("op_719_dilations_0"), val = tensor([1, 1])]; int32 var_719_groups_0 = const()[name = string("op_719_groups_0"), val = int32(1)]; tensor var_719 = conv(dilations = var_719_dilations_0, groups = var_719_groups_0, pad = var_719_pad_0, pad_type = var_719_pad_type_0, strides = var_719_strides_0, weight = model_model_layers_1_self_attn_q_proj_weight_palettized, x = var_712_cast_fp16)[name = string("op_719")]; tensor var_720 = const()[name = string("op_720"), val = tensor([1, 32, 1, 64])]; tensor var_721 = reshape(shape = var_720, x = var_719)[name = string("op_721")]; string var_728_pad_type_0 = const()[name = string("op_728_pad_type_0"), val = string("valid")]; tensor var_728_strides_0 = const()[name = string("op_728_strides_0"), val = tensor([1, 1])]; tensor var_728_pad_0 = const()[name = string("op_728_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_728_dilations_0 = const()[name = string("op_728_dilations_0"), val = tensor([1, 1])]; int32 var_728_groups_0 = const()[name = string("op_728_groups_0"), val = int32(1)]; tensor var_728 = conv(dilations = var_728_dilations_0, groups = var_728_groups_0, pad = var_728_pad_0, pad_type = var_728_pad_type_0, strides = var_728_strides_0, weight = model_model_layers_1_self_attn_k_proj_weight_palettized, x = var_712_cast_fp16)[name = string("op_728")]; tensor var_729 = const()[name = string("op_729"), val = tensor([1, 8, 1, 64])]; tensor var_730 = reshape(shape = var_729, x = var_728)[name = string("op_730")]; string var_737_pad_type_0 = const()[name = string("op_737_pad_type_0"), val = string("valid")]; tensor var_737_strides_0 = const()[name = string("op_737_strides_0"), val = tensor([1, 1])]; tensor var_737_pad_0 = const()[name = string("op_737_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_737_dilations_0 = const()[name = string("op_737_dilations_0"), val = tensor([1, 1])]; int32 var_737_groups_0 = const()[name = string("op_737_groups_0"), val = int32(1)]; tensor var_737 = conv(dilations = var_737_dilations_0, groups = var_737_groups_0, pad = var_737_pad_0, pad_type = var_737_pad_type_0, strides = var_737_strides_0, weight = model_model_layers_1_self_attn_v_proj_weight_palettized, x = var_712_cast_fp16)[name = string("op_737")]; tensor var_738 = const()[name = string("op_738"), val = tensor([1, 8, 1, 64])]; tensor var_739 = reshape(shape = var_738, x = var_737)[name = string("op_739")]; tensor x1_5_begin_0 = const()[name = string("x1_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_5_end_0 = const()[name = string("x1_5_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_5_end_mask_0 = const()[name = string("x1_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_5 = slice_by_index(begin = x1_5_begin_0, end = x1_5_end_0, end_mask = x1_5_end_mask_0, x = var_721)[name = string("x1_5")]; tensor x2_5_begin_0 = const()[name = string("x2_5_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_5_end_0 = const()[name = string("x2_5_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_5_end_mask_0 = const()[name = string("x2_5_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_5 = slice_by_index(begin = x2_5_begin_0, end = x2_5_end_0, end_mask = x2_5_end_mask_0, x = var_721)[name = string("x2_5")]; tensor var_753_cast_fp16 = mul(x = x1_5, y = cos_3_cast_fp16)[name = string("op_753_cast_fp16")]; tensor var_754_cast_fp16 = mul(x = x2_5, y = sin_3_cast_fp16)[name = string("op_754_cast_fp16")]; tensor var_755_cast_fp16 = sub(x = var_753_cast_fp16, y = var_754_cast_fp16)[name = string("op_755_cast_fp16")]; tensor var_756_cast_fp16 = mul(x = x2_5, y = cos_3_cast_fp16)[name = string("op_756_cast_fp16")]; tensor var_757_cast_fp16 = mul(x = x1_5, y = sin_3_cast_fp16)[name = string("op_757_cast_fp16")]; tensor var_758_cast_fp16 = add(x = var_756_cast_fp16, y = var_757_cast_fp16)[name = string("op_758_cast_fp16")]; bool rotated_5_interleave_0 = const()[name = string("rotated_5_interleave_0"), val = bool(false)]; tensor rotated_5_cast_fp16 = concat(axis = var_80, interleave = rotated_5_interleave_0, values = (var_755_cast_fp16, var_758_cast_fp16))[name = string("rotated_5_cast_fp16")]; tensor x1_7_begin_0 = const()[name = string("x1_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_7_end_0 = const()[name = string("x1_7_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_7_end_mask_0 = const()[name = string("x1_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_7 = slice_by_index(begin = x1_7_begin_0, end = x1_7_end_0, end_mask = x1_7_end_mask_0, x = var_730)[name = string("x1_7")]; tensor x2_7_begin_0 = const()[name = string("x2_7_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_7_end_0 = const()[name = string("x2_7_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_7_end_mask_0 = const()[name = string("x2_7_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_7 = slice_by_index(begin = x2_7_begin_0, end = x2_7_end_0, end_mask = x2_7_end_mask_0, x = var_730)[name = string("x2_7")]; tensor var_774_cast_fp16 = mul(x = x1_7, y = cos_3_cast_fp16)[name = string("op_774_cast_fp16")]; tensor var_775_cast_fp16 = mul(x = x2_7, y = sin_3_cast_fp16)[name = string("op_775_cast_fp16")]; tensor var_776_cast_fp16 = sub(x = var_774_cast_fp16, y = var_775_cast_fp16)[name = string("op_776_cast_fp16")]; tensor var_777_cast_fp16 = mul(x = x2_7, y = cos_3_cast_fp16)[name = string("op_777_cast_fp16")]; tensor var_778_cast_fp16 = mul(x = x1_7, y = sin_3_cast_fp16)[name = string("op_778_cast_fp16")]; tensor var_779_cast_fp16 = add(x = var_777_cast_fp16, y = var_778_cast_fp16)[name = string("op_779_cast_fp16")]; bool rotated_7_interleave_0 = const()[name = string("rotated_7_interleave_0"), val = bool(false)]; tensor rotated_7_cast_fp16 = concat(axis = var_80, interleave = rotated_7_interleave_0, values = (var_776_cast_fp16, var_779_cast_fp16))[name = string("rotated_7_cast_fp16")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; int32 concat_10_axis_0 = const()[name = string("concat_10_axis_0"), val = int32(0)]; bool concat_10_interleave_0 = const()[name = string("concat_10_interleave_0"), val = bool(false)]; tensor concat_10 = concat(axis = concat_10_axis_0, interleave = concat_10_interleave_0, values = (expand_dims_12, expand_dims_13, current_pos, expand_dims_15))[name = string("concat_10")]; tensor concat_11_values1_0 = const()[name = string("concat_11_values1_0"), val = tensor([0])]; tensor concat_11_values3_0 = const()[name = string("concat_11_values3_0"), val = tensor([0])]; int32 concat_11_axis_0 = const()[name = string("concat_11_axis_0"), val = int32(0)]; bool concat_11_interleave_0 = const()[name = string("concat_11_interleave_0"), val = bool(false)]; tensor concat_11 = concat(axis = concat_11_axis_0, interleave = concat_11_interleave_0, values = (expand_dims_16, concat_11_values1_0, var_587, concat_11_values3_0))[name = string("concat_11")]; tensor model_model_kv_cache_0_internal_tensor_assign_3_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_10, begin_mask = model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0, end = concat_11, end_mask = model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_3_stride_0, update = rotated_7_cast_fp16, x = coreml_update_state_33)[name = string("model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_66_write_state")]; tensor coreml_update_state_34 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_66")]; tensor expand_dims_18 = const()[name = string("expand_dims_18"), val = tensor([17])]; tensor expand_dims_19 = const()[name = string("expand_dims_19"), val = tensor([0])]; tensor expand_dims_21 = const()[name = string("expand_dims_21"), val = tensor([0])]; tensor expand_dims_22 = const()[name = string("expand_dims_22"), val = tensor([18])]; int32 concat_14_axis_0 = const()[name = string("concat_14_axis_0"), val = int32(0)]; bool concat_14_interleave_0 = const()[name = string("concat_14_interleave_0"), val = bool(false)]; tensor concat_14 = concat(axis = concat_14_axis_0, interleave = concat_14_interleave_0, values = (expand_dims_18, expand_dims_19, current_pos, expand_dims_21))[name = string("concat_14")]; tensor concat_15_values1_0 = const()[name = string("concat_15_values1_0"), val = tensor([0])]; tensor concat_15_values3_0 = const()[name = string("concat_15_values3_0"), val = tensor([0])]; int32 concat_15_axis_0 = const()[name = string("concat_15_axis_0"), val = int32(0)]; bool concat_15_interleave_0 = const()[name = string("concat_15_interleave_0"), val = bool(false)]; tensor concat_15 = concat(axis = concat_15_axis_0, interleave = concat_15_interleave_0, values = (expand_dims_22, concat_15_values1_0, var_587, concat_15_values3_0))[name = string("concat_15")]; tensor model_model_kv_cache_0_internal_tensor_assign_4_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_14, begin_mask = model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0, end = concat_15, end_mask = model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_4_stride_0, update = var_739, x = coreml_update_state_34)[name = string("model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_67_write_state")]; tensor coreml_update_state_35 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_67")]; tensor var_799_begin_0 = const()[name = string("op_799_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_799_end_0 = const()[name = string("op_799_end_0"), val = tensor([2, 8, 4096, 64])]; tensor var_799_end_mask_0 = const()[name = string("op_799_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_799_cast_fp16 = slice_by_index(begin = var_799_begin_0, end = var_799_end_0, end_mask = var_799_end_mask_0, x = coreml_update_state_35)[name = string("op_799_cast_fp16")]; tensor K_layer_cache_3_axes_0 = const()[name = string("K_layer_cache_3_axes_0"), val = tensor([0])]; tensor K_layer_cache_3_cast_fp16 = squeeze(axes = K_layer_cache_3_axes_0, x = var_799_cast_fp16)[name = string("K_layer_cache_3_cast_fp16")]; tensor var_801_begin_0 = const()[name = string("op_801_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_801_end_0 = const()[name = string("op_801_end_0"), val = tensor([18, 8, 4096, 64])]; tensor var_801_end_mask_0 = const()[name = string("op_801_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_801_cast_fp16 = slice_by_index(begin = var_801_begin_0, end = var_801_end_0, end_mask = var_801_end_mask_0, x = coreml_update_state_35)[name = string("op_801_cast_fp16")]; tensor V_layer_cache_3_axes_0 = const()[name = string("V_layer_cache_3_axes_0"), val = tensor([0])]; tensor V_layer_cache_3_cast_fp16 = squeeze(axes = V_layer_cache_3_axes_0, x = var_801_cast_fp16)[name = string("V_layer_cache_3_cast_fp16")]; tensor x_39_axes_0 = const()[name = string("x_39_axes_0"), val = tensor([1])]; tensor x_39_cast_fp16 = expand_dims(axes = x_39_axes_0, x = K_layer_cache_3_cast_fp16)[name = string("x_39_cast_fp16")]; tensor var_810 = const()[name = string("op_810"), val = tensor([1, 4, 1, 1])]; tensor x_41_cast_fp16 = tile(reps = var_810, x = x_39_cast_fp16)[name = string("x_41_cast_fp16")]; tensor var_814 = const()[name = string("op_814"), val = tensor([1, -1, 4096, 64])]; tensor key_states_7_cast_fp16 = reshape(shape = var_814, x = x_41_cast_fp16)[name = string("key_states_7_cast_fp16")]; tensor x_45_axes_0 = const()[name = string("x_45_axes_0"), val = tensor([1])]; tensor x_45_cast_fp16 = expand_dims(axes = x_45_axes_0, x = V_layer_cache_3_cast_fp16)[name = string("x_45_cast_fp16")]; tensor var_817 = const()[name = string("op_817"), val = tensor([1, 4, 1, 1])]; tensor x_47_cast_fp16 = tile(reps = var_817, x = x_45_cast_fp16)[name = string("x_47_cast_fp16")]; tensor var_821 = const()[name = string("op_821"), val = tensor([1, -1, 4096, 64])]; tensor value_states_7_cast_fp16 = reshape(shape = var_821, x = x_47_cast_fp16)[name = string("value_states_7_cast_fp16")]; bool var_824_transpose_x_1 = const()[name = string("op_824_transpose_x_1"), val = bool(false)]; bool var_824_transpose_y_1 = const()[name = string("op_824_transpose_y_1"), val = bool(true)]; tensor var_824_cast_fp16 = matmul(transpose_x = var_824_transpose_x_1, transpose_y = var_824_transpose_y_1, x = rotated_5_cast_fp16, y = key_states_7_cast_fp16)[name = string("op_824_cast_fp16")]; fp16 var_825_to_fp16 = const()[name = string("op_825_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_5_cast_fp16 = mul(x = var_824_cast_fp16, y = var_825_to_fp16)[name = string("attn_weights_5_cast_fp16")]; tensor x_49_cast_fp16 = add(x = attn_weights_5_cast_fp16, y = causal_mask)[name = string("x_49_cast_fp16")]; tensor reduce_max_1_axes_0 = const()[name = string("reduce_max_1_axes_0"), val = tensor([-1])]; bool reduce_max_1_keep_dims_0 = const()[name = string("reduce_max_1_keep_dims_0"), val = bool(true)]; tensor reduce_max_1_cast_fp16 = reduce_max(axes = reduce_max_1_axes_0, keep_dims = reduce_max_1_keep_dims_0, x = x_49_cast_fp16)[name = string("reduce_max_1_cast_fp16")]; tensor x_51_cast_fp16 = sub(x = x_49_cast_fp16, y = reduce_max_1_cast_fp16)[name = string("x_51_cast_fp16")]; tensor exp_x_3_cast_fp16 = exp(x = x_51_cast_fp16)[name = string("exp_x_3_cast_fp16")]; tensor var_836_axes_0 = const()[name = string("op_836_axes_0"), val = tensor([-1])]; bool var_836_keep_dims_0 = const()[name = string("op_836_keep_dims_0"), val = bool(true)]; tensor var_836_cast_fp16 = reduce_sum(axes = var_836_axes_0, keep_dims = var_836_keep_dims_0, x = exp_x_3_cast_fp16)[name = string("op_836_cast_fp16")]; tensor attn_weights_7_cast_fp16 = real_div(x = exp_x_3_cast_fp16, y = var_836_cast_fp16)[name = string("attn_weights_7_cast_fp16")]; bool attn_output_7_transpose_x_0 = const()[name = string("attn_output_7_transpose_x_0"), val = bool(false)]; bool attn_output_7_transpose_y_0 = const()[name = string("attn_output_7_transpose_y_0"), val = bool(false)]; tensor attn_output_7_cast_fp16 = matmul(transpose_x = attn_output_7_transpose_x_0, transpose_y = attn_output_7_transpose_y_0, x = attn_weights_7_cast_fp16, y = value_states_7_cast_fp16)[name = string("attn_output_7_cast_fp16")]; tensor var_839_perm_0 = const()[name = string("op_839_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_841 = const()[name = string("op_841"), val = tensor([1, 1, 2048])]; tensor var_839_cast_fp16 = transpose(perm = var_839_perm_0, x = attn_output_7_cast_fp16)[name = string("transpose_58")]; tensor input_19_cast_fp16 = reshape(shape = var_841, x = var_839_cast_fp16)[name = string("input_19_cast_fp16")]; tensor model_model_layers_1_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(458592832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460690048))))[name = string("model_model_layers_1_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_1_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_1_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_19_cast_fp16)[name = string("linear_1_cast_fp16")]; tensor hidden_states_13_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = linear_1_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_847_cast_fp16 = mul(x = hidden_states_13_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_847_cast_fp16")]; bool input_21_interleave_0 = const()[name = string("input_21_interleave_0"), val = bool(false)]; tensor input_21_cast_fp16 = concat(axis = var_80, interleave = input_21_interleave_0, values = (hidden_states_13_cast_fp16, var_847_cast_fp16))[name = string("input_21_cast_fp16")]; tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_74_to_fp16, x = input_21_cast_fp16)[name = string("normed_13_cast_fp16")]; tensor normed_15_begin_0 = const()[name = string("normed_15_begin_0"), val = tensor([0, 0, 0])]; tensor normed_15_end_0 = const()[name = string("normed_15_end_0"), val = tensor([1, 1, 2048])]; tensor normed_15_end_mask_0 = const()[name = string("normed_15_end_mask_0"), val = tensor([true, true, false])]; tensor normed_15_cast_fp16 = slice_by_index(begin = normed_15_begin_0, end = normed_15_end_0, end_mask = normed_15_end_mask_0, x = normed_13_cast_fp16)[name = string("normed_15_cast_fp16")]; tensor const_31_promoted_to_fp16 = const()[name = string("const_31_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460698304)))]; tensor x_53_cast_fp16 = mul(x = normed_15_cast_fp16, y = const_31_promoted_to_fp16)[name = string("x_53_cast_fp16")]; tensor var_865 = const()[name = string("op_865"), val = tensor([0, 2, 1])]; tensor input_23_axes_0 = const()[name = string("input_23_axes_0"), val = tensor([2])]; tensor var_866 = transpose(perm = var_865, x = x_53_cast_fp16)[name = string("transpose_57")]; tensor input_23 = expand_dims(axes = input_23_axes_0, x = var_866)[name = string("input_23")]; string input_25_pad_type_0 = const()[name = string("input_25_pad_type_0"), val = string("valid")]; tensor input_25_strides_0 = const()[name = string("input_25_strides_0"), val = tensor([1, 1])]; tensor input_25_pad_0 = const()[name = string("input_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_25_dilations_0 = const()[name = string("input_25_dilations_0"), val = tensor([1, 1])]; int32 input_25_groups_0 = const()[name = string("input_25_groups_0"), val = int32(1)]; tensor input_25 = conv(dilations = input_25_dilations_0, groups = input_25_groups_0, pad = input_25_pad_0, pad_type = input_25_pad_type_0, strides = input_25_strides_0, weight = model_model_layers_1_mlp_gate_proj_weight_palettized, x = input_23)[name = string("input_25")]; string up_states_3_pad_type_0 = const()[name = string("up_states_3_pad_type_0"), val = string("valid")]; tensor up_states_3_strides_0 = const()[name = string("up_states_3_strides_0"), val = tensor([1, 1])]; tensor up_states_3_pad_0 = const()[name = string("up_states_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_3_dilations_0 = const()[name = string("up_states_3_dilations_0"), val = tensor([1, 1])]; int32 up_states_3_groups_0 = const()[name = string("up_states_3_groups_0"), val = int32(1)]; tensor up_states_3 = conv(dilations = up_states_3_dilations_0, groups = up_states_3_groups_0, pad = up_states_3_pad_0, pad_type = up_states_3_pad_type_0, strides = up_states_3_strides_0, weight = model_model_layers_1_mlp_up_proj_weight_palettized, x = input_23)[name = string("up_states_3")]; tensor gate_states_3 = silu(x = input_25)[name = string("gate_states_3")]; tensor input_27 = mul(x = gate_states_3, y = up_states_3)[name = string("input_27")]; string hidden_states_15_pad_type_0 = const()[name = string("hidden_states_15_pad_type_0"), val = string("valid")]; tensor hidden_states_15_strides_0 = const()[name = string("hidden_states_15_strides_0"), val = tensor([1, 1])]; tensor hidden_states_15_pad_0 = const()[name = string("hidden_states_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_15_dilations_0 = const()[name = string("hidden_states_15_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_15_groups_0 = const()[name = string("hidden_states_15_groups_0"), val = int32(1)]; tensor hidden_states_15 = conv(dilations = hidden_states_15_dilations_0, groups = hidden_states_15_groups_0, pad = hidden_states_15_pad_0, pad_type = hidden_states_15_pad_type_0, strides = hidden_states_15_strides_0, weight = model_model_layers_1_mlp_down_proj_weight_palettized, x = input_27)[name = string("hidden_states_15")]; tensor var_888_axes_0 = const()[name = string("op_888_axes_0"), val = tensor([2])]; tensor var_888 = squeeze(axes = var_888_axes_0, x = hidden_states_15)[name = string("op_888")]; tensor var_889 = const()[name = string("op_889"), val = tensor([0, 2, 1])]; tensor var_890 = transpose(perm = var_889, x = var_888)[name = string("transpose_56")]; tensor hidden_states_17_cast_fp16 = add(x = hidden_states_13_cast_fp16, y = var_890)[name = string("hidden_states_17_cast_fp16")]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_893_cast_fp16 = mul(x = hidden_states_17_cast_fp16, y = const_32_promoted_to_fp16)[name = string("op_893_cast_fp16")]; bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; tensor input_29_cast_fp16 = concat(axis = var_80, interleave = input_29_interleave_0, values = (hidden_states_17_cast_fp16, var_893_cast_fp16))[name = string("input_29_cast_fp16")]; tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_74_to_fp16, x = input_29_cast_fp16)[name = string("normed_17_cast_fp16")]; tensor normed_19_begin_0 = const()[name = string("normed_19_begin_0"), val = tensor([0, 0, 0])]; tensor normed_19_end_0 = const()[name = string("normed_19_end_0"), val = tensor([1, 1, 2048])]; tensor normed_19_end_mask_0 = const()[name = string("normed_19_end_mask_0"), val = tensor([true, true, false])]; tensor normed_19_cast_fp16 = slice_by_index(begin = normed_19_begin_0, end = normed_19_end_0, end_mask = normed_19_end_mask_0, x = normed_17_cast_fp16)[name = string("normed_19_cast_fp16")]; tensor const_35_promoted_to_fp16 = const()[name = string("const_35_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460702464)))]; tensor hidden_states_19_cast_fp16 = mul(x = normed_19_cast_fp16, y = const_35_promoted_to_fp16)[name = string("hidden_states_19_cast_fp16")]; tensor var_907 = const()[name = string("op_907"), val = tensor([0, 2, 1])]; tensor var_909_axes_0 = const()[name = string("op_909_axes_0"), val = tensor([2])]; tensor var_908_cast_fp16 = transpose(perm = var_907, x = hidden_states_19_cast_fp16)[name = string("transpose_55")]; tensor var_909_cast_fp16 = expand_dims(axes = var_909_axes_0, x = var_908_cast_fp16)[name = string("op_909_cast_fp16")]; string var_916_pad_type_0 = const()[name = string("op_916_pad_type_0"), val = string("valid")]; tensor var_916_strides_0 = const()[name = string("op_916_strides_0"), val = tensor([1, 1])]; tensor var_916_pad_0 = const()[name = string("op_916_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_916_dilations_0 = const()[name = string("op_916_dilations_0"), val = tensor([1, 1])]; int32 var_916_groups_0 = const()[name = string("op_916_groups_0"), val = int32(1)]; tensor var_916 = conv(dilations = var_916_dilations_0, groups = var_916_groups_0, pad = var_916_pad_0, pad_type = var_916_pad_type_0, strides = var_916_strides_0, weight = model_model_layers_2_self_attn_q_proj_weight_palettized, x = var_909_cast_fp16)[name = string("op_916")]; tensor var_917 = const()[name = string("op_917"), val = tensor([1, 32, 1, 64])]; tensor var_918 = reshape(shape = var_917, x = var_916)[name = string("op_918")]; string var_925_pad_type_0 = const()[name = string("op_925_pad_type_0"), val = string("valid")]; tensor var_925_strides_0 = const()[name = string("op_925_strides_0"), val = tensor([1, 1])]; tensor var_925_pad_0 = const()[name = string("op_925_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_925_dilations_0 = const()[name = string("op_925_dilations_0"), val = tensor([1, 1])]; int32 var_925_groups_0 = const()[name = string("op_925_groups_0"), val = int32(1)]; tensor var_925 = conv(dilations = var_925_dilations_0, groups = var_925_groups_0, pad = var_925_pad_0, pad_type = var_925_pad_type_0, strides = var_925_strides_0, weight = model_model_layers_2_self_attn_k_proj_weight_palettized, x = var_909_cast_fp16)[name = string("op_925")]; tensor var_926 = const()[name = string("op_926"), val = tensor([1, 8, 1, 64])]; tensor var_927 = reshape(shape = var_926, x = var_925)[name = string("op_927")]; string var_934_pad_type_0 = const()[name = string("op_934_pad_type_0"), val = string("valid")]; tensor var_934_strides_0 = const()[name = string("op_934_strides_0"), val = tensor([1, 1])]; tensor var_934_pad_0 = const()[name = string("op_934_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_934_dilations_0 = const()[name = string("op_934_dilations_0"), val = tensor([1, 1])]; int32 var_934_groups_0 = const()[name = string("op_934_groups_0"), val = int32(1)]; tensor var_934 = conv(dilations = var_934_dilations_0, groups = var_934_groups_0, pad = var_934_pad_0, pad_type = var_934_pad_type_0, strides = var_934_strides_0, weight = model_model_layers_2_self_attn_v_proj_weight_palettized, x = var_909_cast_fp16)[name = string("op_934")]; tensor var_935 = const()[name = string("op_935"), val = tensor([1, 8, 1, 64])]; tensor var_936 = reshape(shape = var_935, x = var_934)[name = string("op_936")]; tensor x1_9_begin_0 = const()[name = string("x1_9_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_9_end_0 = const()[name = string("x1_9_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_9_end_mask_0 = const()[name = string("x1_9_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_9 = slice_by_index(begin = x1_9_begin_0, end = x1_9_end_0, end_mask = x1_9_end_mask_0, x = var_918)[name = string("x1_9")]; tensor x2_9_begin_0 = const()[name = string("x2_9_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_9_end_0 = const()[name = string("x2_9_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_9_end_mask_0 = const()[name = string("x2_9_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_9 = slice_by_index(begin = x2_9_begin_0, end = x2_9_end_0, end_mask = x2_9_end_mask_0, x = var_918)[name = string("x2_9")]; tensor var_950_cast_fp16 = mul(x = x1_9, y = cos_3_cast_fp16)[name = string("op_950_cast_fp16")]; tensor var_951_cast_fp16 = mul(x = x2_9, y = sin_3_cast_fp16)[name = string("op_951_cast_fp16")]; tensor var_952_cast_fp16 = sub(x = var_950_cast_fp16, y = var_951_cast_fp16)[name = string("op_952_cast_fp16")]; tensor var_953_cast_fp16 = mul(x = x2_9, y = cos_3_cast_fp16)[name = string("op_953_cast_fp16")]; tensor var_954_cast_fp16 = mul(x = x1_9, y = sin_3_cast_fp16)[name = string("op_954_cast_fp16")]; tensor var_955_cast_fp16 = add(x = var_953_cast_fp16, y = var_954_cast_fp16)[name = string("op_955_cast_fp16")]; bool rotated_9_interleave_0 = const()[name = string("rotated_9_interleave_0"), val = bool(false)]; tensor rotated_9_cast_fp16 = concat(axis = var_80, interleave = rotated_9_interleave_0, values = (var_952_cast_fp16, var_955_cast_fp16))[name = string("rotated_9_cast_fp16")]; tensor x1_11_begin_0 = const()[name = string("x1_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_11_end_0 = const()[name = string("x1_11_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_11_end_mask_0 = const()[name = string("x1_11_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_11 = slice_by_index(begin = x1_11_begin_0, end = x1_11_end_0, end_mask = x1_11_end_mask_0, x = var_927)[name = string("x1_11")]; tensor x2_11_begin_0 = const()[name = string("x2_11_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_11_end_0 = const()[name = string("x2_11_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_11_end_mask_0 = const()[name = string("x2_11_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_11 = slice_by_index(begin = x2_11_begin_0, end = x2_11_end_0, end_mask = x2_11_end_mask_0, x = var_927)[name = string("x2_11")]; tensor var_971_cast_fp16 = mul(x = x1_11, y = cos_3_cast_fp16)[name = string("op_971_cast_fp16")]; tensor var_972_cast_fp16 = mul(x = x2_11, y = sin_3_cast_fp16)[name = string("op_972_cast_fp16")]; tensor var_973_cast_fp16 = sub(x = var_971_cast_fp16, y = var_972_cast_fp16)[name = string("op_973_cast_fp16")]; tensor var_974_cast_fp16 = mul(x = x2_11, y = cos_3_cast_fp16)[name = string("op_974_cast_fp16")]; tensor var_975_cast_fp16 = mul(x = x1_11, y = sin_3_cast_fp16)[name = string("op_975_cast_fp16")]; tensor var_976_cast_fp16 = add(x = var_974_cast_fp16, y = var_975_cast_fp16)[name = string("op_976_cast_fp16")]; bool rotated_11_interleave_0 = const()[name = string("rotated_11_interleave_0"), val = bool(false)]; tensor rotated_11_cast_fp16 = concat(axis = var_80, interleave = rotated_11_interleave_0, values = (var_973_cast_fp16, var_976_cast_fp16))[name = string("rotated_11_cast_fp16")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_24, expand_dims_25, current_pos, expand_dims_27))[name = string("concat_18")]; tensor concat_19_values1_0 = const()[name = string("concat_19_values1_0"), val = tensor([0])]; tensor concat_19_values3_0 = const()[name = string("concat_19_values3_0"), val = tensor([0])]; int32 concat_19_axis_0 = const()[name = string("concat_19_axis_0"), val = int32(0)]; bool concat_19_interleave_0 = const()[name = string("concat_19_interleave_0"), val = bool(false)]; tensor concat_19 = concat(axis = concat_19_axis_0, interleave = concat_19_interleave_0, values = (expand_dims_28, concat_19_values1_0, var_587, concat_19_values3_0))[name = string("concat_19")]; tensor model_model_kv_cache_0_internal_tensor_assign_5_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_18, begin_mask = model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0, end = concat_19, end_mask = model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_5_stride_0, update = rotated_11_cast_fp16, x = coreml_update_state_35)[name = string("model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_68_write_state")]; tensor coreml_update_state_36 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_68")]; tensor expand_dims_30 = const()[name = string("expand_dims_30"), val = tensor([18])]; tensor expand_dims_31 = const()[name = string("expand_dims_31"), val = tensor([0])]; tensor expand_dims_33 = const()[name = string("expand_dims_33"), val = tensor([0])]; tensor expand_dims_34 = const()[name = string("expand_dims_34"), val = tensor([19])]; int32 concat_22_axis_0 = const()[name = string("concat_22_axis_0"), val = int32(0)]; bool concat_22_interleave_0 = const()[name = string("concat_22_interleave_0"), val = bool(false)]; tensor concat_22 = concat(axis = concat_22_axis_0, interleave = concat_22_interleave_0, values = (expand_dims_30, expand_dims_31, current_pos, expand_dims_33))[name = string("concat_22")]; tensor concat_23_values1_0 = const()[name = string("concat_23_values1_0"), val = tensor([0])]; tensor concat_23_values3_0 = const()[name = string("concat_23_values3_0"), val = tensor([0])]; int32 concat_23_axis_0 = const()[name = string("concat_23_axis_0"), val = int32(0)]; bool concat_23_interleave_0 = const()[name = string("concat_23_interleave_0"), val = bool(false)]; tensor concat_23 = concat(axis = concat_23_axis_0, interleave = concat_23_interleave_0, values = (expand_dims_34, concat_23_values1_0, var_587, concat_23_values3_0))[name = string("concat_23")]; tensor model_model_kv_cache_0_internal_tensor_assign_6_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_22, begin_mask = model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0, end = concat_23, end_mask = model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_6_stride_0, update = var_936, x = coreml_update_state_36)[name = string("model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_69_write_state")]; tensor coreml_update_state_37 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_69")]; tensor var_996_begin_0 = const()[name = string("op_996_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_996_end_0 = const()[name = string("op_996_end_0"), val = tensor([3, 8, 4096, 64])]; tensor var_996_end_mask_0 = const()[name = string("op_996_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_996_cast_fp16 = slice_by_index(begin = var_996_begin_0, end = var_996_end_0, end_mask = var_996_end_mask_0, x = coreml_update_state_37)[name = string("op_996_cast_fp16")]; tensor K_layer_cache_5_axes_0 = const()[name = string("K_layer_cache_5_axes_0"), val = tensor([0])]; tensor K_layer_cache_5_cast_fp16 = squeeze(axes = K_layer_cache_5_axes_0, x = var_996_cast_fp16)[name = string("K_layer_cache_5_cast_fp16")]; tensor var_998_begin_0 = const()[name = string("op_998_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_998_end_0 = const()[name = string("op_998_end_0"), val = tensor([19, 8, 4096, 64])]; tensor var_998_end_mask_0 = const()[name = string("op_998_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_998_cast_fp16 = slice_by_index(begin = var_998_begin_0, end = var_998_end_0, end_mask = var_998_end_mask_0, x = coreml_update_state_37)[name = string("op_998_cast_fp16")]; tensor V_layer_cache_5_axes_0 = const()[name = string("V_layer_cache_5_axes_0"), val = tensor([0])]; tensor V_layer_cache_5_cast_fp16 = squeeze(axes = V_layer_cache_5_axes_0, x = var_998_cast_fp16)[name = string("V_layer_cache_5_cast_fp16")]; tensor x_67_axes_0 = const()[name = string("x_67_axes_0"), val = tensor([1])]; tensor x_67_cast_fp16 = expand_dims(axes = x_67_axes_0, x = K_layer_cache_5_cast_fp16)[name = string("x_67_cast_fp16")]; tensor var_1007 = const()[name = string("op_1007"), val = tensor([1, 4, 1, 1])]; tensor x_69_cast_fp16 = tile(reps = var_1007, x = x_67_cast_fp16)[name = string("x_69_cast_fp16")]; tensor var_1011 = const()[name = string("op_1011"), val = tensor([1, -1, 4096, 64])]; tensor key_states_11_cast_fp16 = reshape(shape = var_1011, x = x_69_cast_fp16)[name = string("key_states_11_cast_fp16")]; tensor x_73_axes_0 = const()[name = string("x_73_axes_0"), val = tensor([1])]; tensor x_73_cast_fp16 = expand_dims(axes = x_73_axes_0, x = V_layer_cache_5_cast_fp16)[name = string("x_73_cast_fp16")]; tensor var_1014 = const()[name = string("op_1014"), val = tensor([1, 4, 1, 1])]; tensor x_75_cast_fp16 = tile(reps = var_1014, x = x_73_cast_fp16)[name = string("x_75_cast_fp16")]; tensor var_1018 = const()[name = string("op_1018"), val = tensor([1, -1, 4096, 64])]; tensor value_states_11_cast_fp16 = reshape(shape = var_1018, x = x_75_cast_fp16)[name = string("value_states_11_cast_fp16")]; bool var_1021_transpose_x_1 = const()[name = string("op_1021_transpose_x_1"), val = bool(false)]; bool var_1021_transpose_y_1 = const()[name = string("op_1021_transpose_y_1"), val = bool(true)]; tensor var_1021_cast_fp16 = matmul(transpose_x = var_1021_transpose_x_1, transpose_y = var_1021_transpose_y_1, x = rotated_9_cast_fp16, y = key_states_11_cast_fp16)[name = string("op_1021_cast_fp16")]; fp16 var_1022_to_fp16 = const()[name = string("op_1022_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_9_cast_fp16 = mul(x = var_1021_cast_fp16, y = var_1022_to_fp16)[name = string("attn_weights_9_cast_fp16")]; tensor x_77_cast_fp16 = add(x = attn_weights_9_cast_fp16, y = causal_mask)[name = string("x_77_cast_fp16")]; tensor reduce_max_2_axes_0 = const()[name = string("reduce_max_2_axes_0"), val = tensor([-1])]; bool reduce_max_2_keep_dims_0 = const()[name = string("reduce_max_2_keep_dims_0"), val = bool(true)]; tensor reduce_max_2_cast_fp16 = reduce_max(axes = reduce_max_2_axes_0, keep_dims = reduce_max_2_keep_dims_0, x = x_77_cast_fp16)[name = string("reduce_max_2_cast_fp16")]; tensor x_79_cast_fp16 = sub(x = x_77_cast_fp16, y = reduce_max_2_cast_fp16)[name = string("x_79_cast_fp16")]; tensor exp_x_5_cast_fp16 = exp(x = x_79_cast_fp16)[name = string("exp_x_5_cast_fp16")]; tensor var_1033_axes_0 = const()[name = string("op_1033_axes_0"), val = tensor([-1])]; bool var_1033_keep_dims_0 = const()[name = string("op_1033_keep_dims_0"), val = bool(true)]; tensor var_1033_cast_fp16 = reduce_sum(axes = var_1033_axes_0, keep_dims = var_1033_keep_dims_0, x = exp_x_5_cast_fp16)[name = string("op_1033_cast_fp16")]; tensor attn_weights_11_cast_fp16 = real_div(x = exp_x_5_cast_fp16, y = var_1033_cast_fp16)[name = string("attn_weights_11_cast_fp16")]; bool attn_output_13_transpose_x_0 = const()[name = string("attn_output_13_transpose_x_0"), val = bool(false)]; bool attn_output_13_transpose_y_0 = const()[name = string("attn_output_13_transpose_y_0"), val = bool(false)]; tensor attn_output_13_cast_fp16 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = attn_weights_11_cast_fp16, y = value_states_11_cast_fp16)[name = string("attn_output_13_cast_fp16")]; tensor var_1036_perm_0 = const()[name = string("op_1036_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1038 = const()[name = string("op_1038"), val = tensor([1, 1, 2048])]; tensor var_1036_cast_fp16 = transpose(perm = var_1036_perm_0, x = attn_output_13_cast_fp16)[name = string("transpose_54")]; tensor input_33_cast_fp16 = reshape(shape = var_1038, x = var_1036_cast_fp16)[name = string("input_33_cast_fp16")]; tensor model_model_layers_2_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460706624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(462803840))))[name = string("model_model_layers_2_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_2_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_2_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_33_cast_fp16)[name = string("linear_2_cast_fp16")]; tensor hidden_states_21_cast_fp16 = add(x = hidden_states_17_cast_fp16, y = linear_2_cast_fp16)[name = string("hidden_states_21_cast_fp16")]; fp16 const_44_promoted_to_fp16 = const()[name = string("const_44_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1044_cast_fp16 = mul(x = hidden_states_21_cast_fp16, y = const_44_promoted_to_fp16)[name = string("op_1044_cast_fp16")]; bool input_35_interleave_0 = const()[name = string("input_35_interleave_0"), val = bool(false)]; tensor input_35_cast_fp16 = concat(axis = var_80, interleave = input_35_interleave_0, values = (hidden_states_21_cast_fp16, var_1044_cast_fp16))[name = string("input_35_cast_fp16")]; tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_74_to_fp16, x = input_35_cast_fp16)[name = string("normed_21_cast_fp16")]; tensor normed_23_begin_0 = const()[name = string("normed_23_begin_0"), val = tensor([0, 0, 0])]; tensor normed_23_end_0 = const()[name = string("normed_23_end_0"), val = tensor([1, 1, 2048])]; tensor normed_23_end_mask_0 = const()[name = string("normed_23_end_mask_0"), val = tensor([true, true, false])]; tensor normed_23_cast_fp16 = slice_by_index(begin = normed_23_begin_0, end = normed_23_end_0, end_mask = normed_23_end_mask_0, x = normed_21_cast_fp16)[name = string("normed_23_cast_fp16")]; tensor const_47_promoted_to_fp16 = const()[name = string("const_47_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(462812096)))]; tensor x_81_cast_fp16 = mul(x = normed_23_cast_fp16, y = const_47_promoted_to_fp16)[name = string("x_81_cast_fp16")]; tensor var_1062 = const()[name = string("op_1062"), val = tensor([0, 2, 1])]; tensor input_37_axes_0 = const()[name = string("input_37_axes_0"), val = tensor([2])]; tensor var_1063 = transpose(perm = var_1062, x = x_81_cast_fp16)[name = string("transpose_53")]; tensor input_37 = expand_dims(axes = input_37_axes_0, x = var_1063)[name = string("input_37")]; string input_39_pad_type_0 = const()[name = string("input_39_pad_type_0"), val = string("valid")]; tensor input_39_strides_0 = const()[name = string("input_39_strides_0"), val = tensor([1, 1])]; tensor input_39_pad_0 = const()[name = string("input_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_39_dilations_0 = const()[name = string("input_39_dilations_0"), val = tensor([1, 1])]; int32 input_39_groups_0 = const()[name = string("input_39_groups_0"), val = int32(1)]; tensor input_39 = conv(dilations = input_39_dilations_0, groups = input_39_groups_0, pad = input_39_pad_0, pad_type = input_39_pad_type_0, strides = input_39_strides_0, weight = model_model_layers_2_mlp_gate_proj_weight_palettized, x = input_37)[name = string("input_39")]; string up_states_5_pad_type_0 = const()[name = string("up_states_5_pad_type_0"), val = string("valid")]; tensor up_states_5_strides_0 = const()[name = string("up_states_5_strides_0"), val = tensor([1, 1])]; tensor up_states_5_pad_0 = const()[name = string("up_states_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_5_dilations_0 = const()[name = string("up_states_5_dilations_0"), val = tensor([1, 1])]; int32 up_states_5_groups_0 = const()[name = string("up_states_5_groups_0"), val = int32(1)]; tensor up_states_5 = conv(dilations = up_states_5_dilations_0, groups = up_states_5_groups_0, pad = up_states_5_pad_0, pad_type = up_states_5_pad_type_0, strides = up_states_5_strides_0, weight = model_model_layers_2_mlp_up_proj_weight_palettized, x = input_37)[name = string("up_states_5")]; tensor gate_states_5 = silu(x = input_39)[name = string("gate_states_5")]; tensor input_41 = mul(x = gate_states_5, y = up_states_5)[name = string("input_41")]; string hidden_states_23_pad_type_0 = const()[name = string("hidden_states_23_pad_type_0"), val = string("valid")]; tensor hidden_states_23_strides_0 = const()[name = string("hidden_states_23_strides_0"), val = tensor([1, 1])]; tensor hidden_states_23_pad_0 = const()[name = string("hidden_states_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_23_dilations_0 = const()[name = string("hidden_states_23_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_23_groups_0 = const()[name = string("hidden_states_23_groups_0"), val = int32(1)]; tensor hidden_states_23 = conv(dilations = hidden_states_23_dilations_0, groups = hidden_states_23_groups_0, pad = hidden_states_23_pad_0, pad_type = hidden_states_23_pad_type_0, strides = hidden_states_23_strides_0, weight = model_model_layers_2_mlp_down_proj_weight_palettized, x = input_41)[name = string("hidden_states_23")]; tensor var_1085_axes_0 = const()[name = string("op_1085_axes_0"), val = tensor([2])]; tensor var_1085 = squeeze(axes = var_1085_axes_0, x = hidden_states_23)[name = string("op_1085")]; tensor var_1086 = const()[name = string("op_1086"), val = tensor([0, 2, 1])]; tensor var_1087 = transpose(perm = var_1086, x = var_1085)[name = string("transpose_52")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_21_cast_fp16, y = var_1087)[name = string("hidden_states_25_cast_fp16")]; fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1090_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_48_promoted_to_fp16)[name = string("op_1090_cast_fp16")]; bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; tensor input_43_cast_fp16 = concat(axis = var_80, interleave = input_43_interleave_0, values = (hidden_states_25_cast_fp16, var_1090_cast_fp16))[name = string("input_43_cast_fp16")]; tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_74_to_fp16, x = input_43_cast_fp16)[name = string("normed_25_cast_fp16")]; tensor normed_27_begin_0 = const()[name = string("normed_27_begin_0"), val = tensor([0, 0, 0])]; tensor normed_27_end_0 = const()[name = string("normed_27_end_0"), val = tensor([1, 1, 2048])]; tensor normed_27_end_mask_0 = const()[name = string("normed_27_end_mask_0"), val = tensor([true, true, false])]; tensor normed_27_cast_fp16 = slice_by_index(begin = normed_27_begin_0, end = normed_27_end_0, end_mask = normed_27_end_mask_0, x = normed_25_cast_fp16)[name = string("normed_27_cast_fp16")]; tensor const_51_promoted_to_fp16 = const()[name = string("const_51_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(462816256)))]; tensor hidden_states_27_cast_fp16 = mul(x = normed_27_cast_fp16, y = const_51_promoted_to_fp16)[name = string("hidden_states_27_cast_fp16")]; tensor var_1104 = const()[name = string("op_1104"), val = tensor([0, 2, 1])]; tensor var_1106_axes_0 = const()[name = string("op_1106_axes_0"), val = tensor([2])]; tensor var_1105_cast_fp16 = transpose(perm = var_1104, x = hidden_states_27_cast_fp16)[name = string("transpose_51")]; tensor var_1106_cast_fp16 = expand_dims(axes = var_1106_axes_0, x = var_1105_cast_fp16)[name = string("op_1106_cast_fp16")]; string var_1113_pad_type_0 = const()[name = string("op_1113_pad_type_0"), val = string("valid")]; tensor var_1113_strides_0 = const()[name = string("op_1113_strides_0"), val = tensor([1, 1])]; tensor var_1113_pad_0 = const()[name = string("op_1113_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1113_dilations_0 = const()[name = string("op_1113_dilations_0"), val = tensor([1, 1])]; int32 var_1113_groups_0 = const()[name = string("op_1113_groups_0"), val = int32(1)]; tensor var_1113 = conv(dilations = var_1113_dilations_0, groups = var_1113_groups_0, pad = var_1113_pad_0, pad_type = var_1113_pad_type_0, strides = var_1113_strides_0, weight = model_model_layers_3_self_attn_q_proj_weight_palettized, x = var_1106_cast_fp16)[name = string("op_1113")]; tensor var_1114 = const()[name = string("op_1114"), val = tensor([1, 32, 1, 64])]; tensor var_1115 = reshape(shape = var_1114, x = var_1113)[name = string("op_1115")]; string var_1122_pad_type_0 = const()[name = string("op_1122_pad_type_0"), val = string("valid")]; tensor var_1122_strides_0 = const()[name = string("op_1122_strides_0"), val = tensor([1, 1])]; tensor var_1122_pad_0 = const()[name = string("op_1122_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1122_dilations_0 = const()[name = string("op_1122_dilations_0"), val = tensor([1, 1])]; int32 var_1122_groups_0 = const()[name = string("op_1122_groups_0"), val = int32(1)]; tensor var_1122 = conv(dilations = var_1122_dilations_0, groups = var_1122_groups_0, pad = var_1122_pad_0, pad_type = var_1122_pad_type_0, strides = var_1122_strides_0, weight = model_model_layers_3_self_attn_k_proj_weight_palettized, x = var_1106_cast_fp16)[name = string("op_1122")]; tensor var_1123 = const()[name = string("op_1123"), val = tensor([1, 8, 1, 64])]; tensor var_1124 = reshape(shape = var_1123, x = var_1122)[name = string("op_1124")]; string var_1131_pad_type_0 = const()[name = string("op_1131_pad_type_0"), val = string("valid")]; tensor var_1131_strides_0 = const()[name = string("op_1131_strides_0"), val = tensor([1, 1])]; tensor var_1131_pad_0 = const()[name = string("op_1131_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1131_dilations_0 = const()[name = string("op_1131_dilations_0"), val = tensor([1, 1])]; int32 var_1131_groups_0 = const()[name = string("op_1131_groups_0"), val = int32(1)]; tensor var_1131 = conv(dilations = var_1131_dilations_0, groups = var_1131_groups_0, pad = var_1131_pad_0, pad_type = var_1131_pad_type_0, strides = var_1131_strides_0, weight = model_model_layers_3_self_attn_v_proj_weight_palettized, x = var_1106_cast_fp16)[name = string("op_1131")]; tensor var_1132 = const()[name = string("op_1132"), val = tensor([1, 8, 1, 64])]; tensor var_1133 = reshape(shape = var_1132, x = var_1131)[name = string("op_1133")]; tensor x1_13_begin_0 = const()[name = string("x1_13_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_13_end_0 = const()[name = string("x1_13_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_13_end_mask_0 = const()[name = string("x1_13_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_13 = slice_by_index(begin = x1_13_begin_0, end = x1_13_end_0, end_mask = x1_13_end_mask_0, x = var_1115)[name = string("x1_13")]; tensor x2_13_begin_0 = const()[name = string("x2_13_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_13_end_0 = const()[name = string("x2_13_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_13_end_mask_0 = const()[name = string("x2_13_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_13 = slice_by_index(begin = x2_13_begin_0, end = x2_13_end_0, end_mask = x2_13_end_mask_0, x = var_1115)[name = string("x2_13")]; tensor var_1147_cast_fp16 = mul(x = x1_13, y = cos_3_cast_fp16)[name = string("op_1147_cast_fp16")]; tensor var_1148_cast_fp16 = mul(x = x2_13, y = sin_3_cast_fp16)[name = string("op_1148_cast_fp16")]; tensor var_1149_cast_fp16 = sub(x = var_1147_cast_fp16, y = var_1148_cast_fp16)[name = string("op_1149_cast_fp16")]; tensor var_1150_cast_fp16 = mul(x = x2_13, y = cos_3_cast_fp16)[name = string("op_1150_cast_fp16")]; tensor var_1151_cast_fp16 = mul(x = x1_13, y = sin_3_cast_fp16)[name = string("op_1151_cast_fp16")]; tensor var_1152_cast_fp16 = add(x = var_1150_cast_fp16, y = var_1151_cast_fp16)[name = string("op_1152_cast_fp16")]; bool rotated_13_interleave_0 = const()[name = string("rotated_13_interleave_0"), val = bool(false)]; tensor rotated_13_cast_fp16 = concat(axis = var_80, interleave = rotated_13_interleave_0, values = (var_1149_cast_fp16, var_1152_cast_fp16))[name = string("rotated_13_cast_fp16")]; tensor x1_15_begin_0 = const()[name = string("x1_15_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_15_end_0 = const()[name = string("x1_15_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_15_end_mask_0 = const()[name = string("x1_15_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_15 = slice_by_index(begin = x1_15_begin_0, end = x1_15_end_0, end_mask = x1_15_end_mask_0, x = var_1124)[name = string("x1_15")]; tensor x2_15_begin_0 = const()[name = string("x2_15_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_15_end_0 = const()[name = string("x2_15_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_15_end_mask_0 = const()[name = string("x2_15_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_15 = slice_by_index(begin = x2_15_begin_0, end = x2_15_end_0, end_mask = x2_15_end_mask_0, x = var_1124)[name = string("x2_15")]; tensor var_1168_cast_fp16 = mul(x = x1_15, y = cos_3_cast_fp16)[name = string("op_1168_cast_fp16")]; tensor var_1169_cast_fp16 = mul(x = x2_15, y = sin_3_cast_fp16)[name = string("op_1169_cast_fp16")]; tensor var_1170_cast_fp16 = sub(x = var_1168_cast_fp16, y = var_1169_cast_fp16)[name = string("op_1170_cast_fp16")]; tensor var_1171_cast_fp16 = mul(x = x2_15, y = cos_3_cast_fp16)[name = string("op_1171_cast_fp16")]; tensor var_1172_cast_fp16 = mul(x = x1_15, y = sin_3_cast_fp16)[name = string("op_1172_cast_fp16")]; tensor var_1173_cast_fp16 = add(x = var_1171_cast_fp16, y = var_1172_cast_fp16)[name = string("op_1173_cast_fp16")]; bool rotated_15_interleave_0 = const()[name = string("rotated_15_interleave_0"), val = bool(false)]; tensor rotated_15_cast_fp16 = concat(axis = var_80, interleave = rotated_15_interleave_0, values = (var_1170_cast_fp16, var_1173_cast_fp16))[name = string("rotated_15_cast_fp16")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; int32 concat_26_axis_0 = const()[name = string("concat_26_axis_0"), val = int32(0)]; bool concat_26_interleave_0 = const()[name = string("concat_26_interleave_0"), val = bool(false)]; tensor concat_26 = concat(axis = concat_26_axis_0, interleave = concat_26_interleave_0, values = (expand_dims_36, expand_dims_37, current_pos, expand_dims_39))[name = string("concat_26")]; tensor concat_27_values1_0 = const()[name = string("concat_27_values1_0"), val = tensor([0])]; tensor concat_27_values3_0 = const()[name = string("concat_27_values3_0"), val = tensor([0])]; int32 concat_27_axis_0 = const()[name = string("concat_27_axis_0"), val = int32(0)]; bool concat_27_interleave_0 = const()[name = string("concat_27_interleave_0"), val = bool(false)]; tensor concat_27 = concat(axis = concat_27_axis_0, interleave = concat_27_interleave_0, values = (expand_dims_40, concat_27_values1_0, var_587, concat_27_values3_0))[name = string("concat_27")]; tensor model_model_kv_cache_0_internal_tensor_assign_7_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_26, begin_mask = model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0, end = concat_27, end_mask = model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_7_stride_0, update = rotated_15_cast_fp16, x = coreml_update_state_37)[name = string("model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_70_write_state")]; tensor coreml_update_state_38 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_70")]; tensor expand_dims_42 = const()[name = string("expand_dims_42"), val = tensor([19])]; tensor expand_dims_43 = const()[name = string("expand_dims_43"), val = tensor([0])]; tensor expand_dims_45 = const()[name = string("expand_dims_45"), val = tensor([0])]; tensor expand_dims_46 = const()[name = string("expand_dims_46"), val = tensor([20])]; int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_42, expand_dims_43, current_pos, expand_dims_45))[name = string("concat_30")]; tensor concat_31_values1_0 = const()[name = string("concat_31_values1_0"), val = tensor([0])]; tensor concat_31_values3_0 = const()[name = string("concat_31_values3_0"), val = tensor([0])]; int32 concat_31_axis_0 = const()[name = string("concat_31_axis_0"), val = int32(0)]; bool concat_31_interleave_0 = const()[name = string("concat_31_interleave_0"), val = bool(false)]; tensor concat_31 = concat(axis = concat_31_axis_0, interleave = concat_31_interleave_0, values = (expand_dims_46, concat_31_values1_0, var_587, concat_31_values3_0))[name = string("concat_31")]; tensor model_model_kv_cache_0_internal_tensor_assign_8_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_30, begin_mask = model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0, end = concat_31, end_mask = model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_8_stride_0, update = var_1133, x = coreml_update_state_38)[name = string("model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_71_write_state")]; tensor coreml_update_state_39 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_71")]; tensor var_1193_begin_0 = const()[name = string("op_1193_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_1193_end_0 = const()[name = string("op_1193_end_0"), val = tensor([4, 8, 4096, 64])]; tensor var_1193_end_mask_0 = const()[name = string("op_1193_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1193_cast_fp16 = slice_by_index(begin = var_1193_begin_0, end = var_1193_end_0, end_mask = var_1193_end_mask_0, x = coreml_update_state_39)[name = string("op_1193_cast_fp16")]; tensor K_layer_cache_7_axes_0 = const()[name = string("K_layer_cache_7_axes_0"), val = tensor([0])]; tensor K_layer_cache_7_cast_fp16 = squeeze(axes = K_layer_cache_7_axes_0, x = var_1193_cast_fp16)[name = string("K_layer_cache_7_cast_fp16")]; tensor var_1195_begin_0 = const()[name = string("op_1195_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_1195_end_0 = const()[name = string("op_1195_end_0"), val = tensor([20, 8, 4096, 64])]; tensor var_1195_end_mask_0 = const()[name = string("op_1195_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1195_cast_fp16 = slice_by_index(begin = var_1195_begin_0, end = var_1195_end_0, end_mask = var_1195_end_mask_0, x = coreml_update_state_39)[name = string("op_1195_cast_fp16")]; tensor V_layer_cache_7_axes_0 = const()[name = string("V_layer_cache_7_axes_0"), val = tensor([0])]; tensor V_layer_cache_7_cast_fp16 = squeeze(axes = V_layer_cache_7_axes_0, x = var_1195_cast_fp16)[name = string("V_layer_cache_7_cast_fp16")]; tensor x_95_axes_0 = const()[name = string("x_95_axes_0"), val = tensor([1])]; tensor x_95_cast_fp16 = expand_dims(axes = x_95_axes_0, x = K_layer_cache_7_cast_fp16)[name = string("x_95_cast_fp16")]; tensor var_1204 = const()[name = string("op_1204"), val = tensor([1, 4, 1, 1])]; tensor x_97_cast_fp16 = tile(reps = var_1204, x = x_95_cast_fp16)[name = string("x_97_cast_fp16")]; tensor var_1208 = const()[name = string("op_1208"), val = tensor([1, -1, 4096, 64])]; tensor key_states_15_cast_fp16 = reshape(shape = var_1208, x = x_97_cast_fp16)[name = string("key_states_15_cast_fp16")]; tensor x_101_axes_0 = const()[name = string("x_101_axes_0"), val = tensor([1])]; tensor x_101_cast_fp16 = expand_dims(axes = x_101_axes_0, x = V_layer_cache_7_cast_fp16)[name = string("x_101_cast_fp16")]; tensor var_1211 = const()[name = string("op_1211"), val = tensor([1, 4, 1, 1])]; tensor x_103_cast_fp16 = tile(reps = var_1211, x = x_101_cast_fp16)[name = string("x_103_cast_fp16")]; tensor var_1215 = const()[name = string("op_1215"), val = tensor([1, -1, 4096, 64])]; tensor value_states_15_cast_fp16 = reshape(shape = var_1215, x = x_103_cast_fp16)[name = string("value_states_15_cast_fp16")]; bool var_1218_transpose_x_1 = const()[name = string("op_1218_transpose_x_1"), val = bool(false)]; bool var_1218_transpose_y_1 = const()[name = string("op_1218_transpose_y_1"), val = bool(true)]; tensor var_1218_cast_fp16 = matmul(transpose_x = var_1218_transpose_x_1, transpose_y = var_1218_transpose_y_1, x = rotated_13_cast_fp16, y = key_states_15_cast_fp16)[name = string("op_1218_cast_fp16")]; fp16 var_1219_to_fp16 = const()[name = string("op_1219_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_13_cast_fp16 = mul(x = var_1218_cast_fp16, y = var_1219_to_fp16)[name = string("attn_weights_13_cast_fp16")]; tensor x_105_cast_fp16 = add(x = attn_weights_13_cast_fp16, y = causal_mask)[name = string("x_105_cast_fp16")]; tensor reduce_max_3_axes_0 = const()[name = string("reduce_max_3_axes_0"), val = tensor([-1])]; bool reduce_max_3_keep_dims_0 = const()[name = string("reduce_max_3_keep_dims_0"), val = bool(true)]; tensor reduce_max_3_cast_fp16 = reduce_max(axes = reduce_max_3_axes_0, keep_dims = reduce_max_3_keep_dims_0, x = x_105_cast_fp16)[name = string("reduce_max_3_cast_fp16")]; tensor x_107_cast_fp16 = sub(x = x_105_cast_fp16, y = reduce_max_3_cast_fp16)[name = string("x_107_cast_fp16")]; tensor exp_x_7_cast_fp16 = exp(x = x_107_cast_fp16)[name = string("exp_x_7_cast_fp16")]; tensor var_1230_axes_0 = const()[name = string("op_1230_axes_0"), val = tensor([-1])]; bool var_1230_keep_dims_0 = const()[name = string("op_1230_keep_dims_0"), val = bool(true)]; tensor var_1230_cast_fp16 = reduce_sum(axes = var_1230_axes_0, keep_dims = var_1230_keep_dims_0, x = exp_x_7_cast_fp16)[name = string("op_1230_cast_fp16")]; tensor attn_weights_15_cast_fp16 = real_div(x = exp_x_7_cast_fp16, y = var_1230_cast_fp16)[name = string("attn_weights_15_cast_fp16")]; bool attn_output_19_transpose_x_0 = const()[name = string("attn_output_19_transpose_x_0"), val = bool(false)]; bool attn_output_19_transpose_y_0 = const()[name = string("attn_output_19_transpose_y_0"), val = bool(false)]; tensor attn_output_19_cast_fp16 = matmul(transpose_x = attn_output_19_transpose_x_0, transpose_y = attn_output_19_transpose_y_0, x = attn_weights_15_cast_fp16, y = value_states_15_cast_fp16)[name = string("attn_output_19_cast_fp16")]; tensor var_1233_perm_0 = const()[name = string("op_1233_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1235 = const()[name = string("op_1235"), val = tensor([1, 1, 2048])]; tensor var_1233_cast_fp16 = transpose(perm = var_1233_perm_0, x = attn_output_19_cast_fp16)[name = string("transpose_50")]; tensor input_47_cast_fp16 = reshape(shape = var_1235, x = var_1233_cast_fp16)[name = string("input_47_cast_fp16")]; tensor model_model_layers_3_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(462820416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(464917632))))[name = string("model_model_layers_3_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_3_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_3_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_47_cast_fp16)[name = string("linear_3_cast_fp16")]; tensor hidden_states_29_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = linear_3_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1241_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_1241_cast_fp16")]; bool input_49_interleave_0 = const()[name = string("input_49_interleave_0"), val = bool(false)]; tensor input_49_cast_fp16 = concat(axis = var_80, interleave = input_49_interleave_0, values = (hidden_states_29_cast_fp16, var_1241_cast_fp16))[name = string("input_49_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_74_to_fp16, x = input_49_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor normed_31_begin_0 = const()[name = string("normed_31_begin_0"), val = tensor([0, 0, 0])]; tensor normed_31_end_0 = const()[name = string("normed_31_end_0"), val = tensor([1, 1, 2048])]; tensor normed_31_end_mask_0 = const()[name = string("normed_31_end_mask_0"), val = tensor([true, true, false])]; tensor normed_31_cast_fp16 = slice_by_index(begin = normed_31_begin_0, end = normed_31_end_0, end_mask = normed_31_end_mask_0, x = normed_29_cast_fp16)[name = string("normed_31_cast_fp16")]; tensor const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(464925888)))]; tensor x_109_cast_fp16 = mul(x = normed_31_cast_fp16, y = const_63_promoted_to_fp16)[name = string("x_109_cast_fp16")]; tensor var_1259 = const()[name = string("op_1259"), val = tensor([0, 2, 1])]; tensor input_51_axes_0 = const()[name = string("input_51_axes_0"), val = tensor([2])]; tensor var_1260 = transpose(perm = var_1259, x = x_109_cast_fp16)[name = string("transpose_49")]; tensor input_51 = expand_dims(axes = input_51_axes_0, x = var_1260)[name = string("input_51")]; string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; tensor input_53 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = model_model_layers_3_mlp_gate_proj_weight_palettized, x = input_51)[name = string("input_53")]; string up_states_7_pad_type_0 = const()[name = string("up_states_7_pad_type_0"), val = string("valid")]; tensor up_states_7_strides_0 = const()[name = string("up_states_7_strides_0"), val = tensor([1, 1])]; tensor up_states_7_pad_0 = const()[name = string("up_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_7_dilations_0 = const()[name = string("up_states_7_dilations_0"), val = tensor([1, 1])]; int32 up_states_7_groups_0 = const()[name = string("up_states_7_groups_0"), val = int32(1)]; tensor up_states_7 = conv(dilations = up_states_7_dilations_0, groups = up_states_7_groups_0, pad = up_states_7_pad_0, pad_type = up_states_7_pad_type_0, strides = up_states_7_strides_0, weight = model_model_layers_3_mlp_up_proj_weight_palettized, x = input_51)[name = string("up_states_7")]; tensor gate_states_7 = silu(x = input_53)[name = string("gate_states_7")]; tensor input_55 = mul(x = gate_states_7, y = up_states_7)[name = string("input_55")]; string hidden_states_31_pad_type_0 = const()[name = string("hidden_states_31_pad_type_0"), val = string("valid")]; tensor hidden_states_31_strides_0 = const()[name = string("hidden_states_31_strides_0"), val = tensor([1, 1])]; tensor hidden_states_31_pad_0 = const()[name = string("hidden_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_31_dilations_0 = const()[name = string("hidden_states_31_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_31_groups_0 = const()[name = string("hidden_states_31_groups_0"), val = int32(1)]; tensor hidden_states_31 = conv(dilations = hidden_states_31_dilations_0, groups = hidden_states_31_groups_0, pad = hidden_states_31_pad_0, pad_type = hidden_states_31_pad_type_0, strides = hidden_states_31_strides_0, weight = model_model_layers_3_mlp_down_proj_weight_palettized, x = input_55)[name = string("hidden_states_31")]; tensor var_1282_axes_0 = const()[name = string("op_1282_axes_0"), val = tensor([2])]; tensor var_1282 = squeeze(axes = var_1282_axes_0, x = hidden_states_31)[name = string("op_1282")]; tensor var_1283 = const()[name = string("op_1283"), val = tensor([0, 2, 1])]; tensor var_1284 = transpose(perm = var_1283, x = var_1282)[name = string("transpose_48")]; tensor hidden_states_33_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = var_1284)[name = string("hidden_states_33_cast_fp16")]; fp16 const_64_promoted_to_fp16 = const()[name = string("const_64_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1287_cast_fp16 = mul(x = hidden_states_33_cast_fp16, y = const_64_promoted_to_fp16)[name = string("op_1287_cast_fp16")]; bool input_57_interleave_0 = const()[name = string("input_57_interleave_0"), val = bool(false)]; tensor input_57_cast_fp16 = concat(axis = var_80, interleave = input_57_interleave_0, values = (hidden_states_33_cast_fp16, var_1287_cast_fp16))[name = string("input_57_cast_fp16")]; tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_74_to_fp16, x = input_57_cast_fp16)[name = string("normed_33_cast_fp16")]; tensor normed_35_begin_0 = const()[name = string("normed_35_begin_0"), val = tensor([0, 0, 0])]; tensor normed_35_end_0 = const()[name = string("normed_35_end_0"), val = tensor([1, 1, 2048])]; tensor normed_35_end_mask_0 = const()[name = string("normed_35_end_mask_0"), val = tensor([true, true, false])]; tensor normed_35_cast_fp16 = slice_by_index(begin = normed_35_begin_0, end = normed_35_end_0, end_mask = normed_35_end_mask_0, x = normed_33_cast_fp16)[name = string("normed_35_cast_fp16")]; tensor const_67_promoted_to_fp16 = const()[name = string("const_67_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(464930048)))]; tensor hidden_states_35_cast_fp16 = mul(x = normed_35_cast_fp16, y = const_67_promoted_to_fp16)[name = string("hidden_states_35_cast_fp16")]; tensor var_1301 = const()[name = string("op_1301"), val = tensor([0, 2, 1])]; tensor var_1303_axes_0 = const()[name = string("op_1303_axes_0"), val = tensor([2])]; tensor var_1302_cast_fp16 = transpose(perm = var_1301, x = hidden_states_35_cast_fp16)[name = string("transpose_47")]; tensor var_1303_cast_fp16 = expand_dims(axes = var_1303_axes_0, x = var_1302_cast_fp16)[name = string("op_1303_cast_fp16")]; string var_1310_pad_type_0 = const()[name = string("op_1310_pad_type_0"), val = string("valid")]; tensor var_1310_strides_0 = const()[name = string("op_1310_strides_0"), val = tensor([1, 1])]; tensor var_1310_pad_0 = const()[name = string("op_1310_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1310_dilations_0 = const()[name = string("op_1310_dilations_0"), val = tensor([1, 1])]; int32 var_1310_groups_0 = const()[name = string("op_1310_groups_0"), val = int32(1)]; tensor var_1310 = conv(dilations = var_1310_dilations_0, groups = var_1310_groups_0, pad = var_1310_pad_0, pad_type = var_1310_pad_type_0, strides = var_1310_strides_0, weight = model_model_layers_4_self_attn_q_proj_weight_palettized, x = var_1303_cast_fp16)[name = string("op_1310")]; tensor var_1311 = const()[name = string("op_1311"), val = tensor([1, 32, 1, 64])]; tensor var_1312 = reshape(shape = var_1311, x = var_1310)[name = string("op_1312")]; string var_1319_pad_type_0 = const()[name = string("op_1319_pad_type_0"), val = string("valid")]; tensor var_1319_strides_0 = const()[name = string("op_1319_strides_0"), val = tensor([1, 1])]; tensor var_1319_pad_0 = const()[name = string("op_1319_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1319_dilations_0 = const()[name = string("op_1319_dilations_0"), val = tensor([1, 1])]; int32 var_1319_groups_0 = const()[name = string("op_1319_groups_0"), val = int32(1)]; tensor var_1319 = conv(dilations = var_1319_dilations_0, groups = var_1319_groups_0, pad = var_1319_pad_0, pad_type = var_1319_pad_type_0, strides = var_1319_strides_0, weight = model_model_layers_4_self_attn_k_proj_weight_palettized, x = var_1303_cast_fp16)[name = string("op_1319")]; tensor var_1320 = const()[name = string("op_1320"), val = tensor([1, 8, 1, 64])]; tensor var_1321 = reshape(shape = var_1320, x = var_1319)[name = string("op_1321")]; string var_1328_pad_type_0 = const()[name = string("op_1328_pad_type_0"), val = string("valid")]; tensor var_1328_strides_0 = const()[name = string("op_1328_strides_0"), val = tensor([1, 1])]; tensor var_1328_pad_0 = const()[name = string("op_1328_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1328_dilations_0 = const()[name = string("op_1328_dilations_0"), val = tensor([1, 1])]; int32 var_1328_groups_0 = const()[name = string("op_1328_groups_0"), val = int32(1)]; tensor var_1328 = conv(dilations = var_1328_dilations_0, groups = var_1328_groups_0, pad = var_1328_pad_0, pad_type = var_1328_pad_type_0, strides = var_1328_strides_0, weight = model_model_layers_4_self_attn_v_proj_weight_palettized, x = var_1303_cast_fp16)[name = string("op_1328")]; tensor var_1329 = const()[name = string("op_1329"), val = tensor([1, 8, 1, 64])]; tensor var_1330 = reshape(shape = var_1329, x = var_1328)[name = string("op_1330")]; tensor x1_17_begin_0 = const()[name = string("x1_17_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_17_end_0 = const()[name = string("x1_17_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_17_end_mask_0 = const()[name = string("x1_17_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_17 = slice_by_index(begin = x1_17_begin_0, end = x1_17_end_0, end_mask = x1_17_end_mask_0, x = var_1312)[name = string("x1_17")]; tensor x2_17_begin_0 = const()[name = string("x2_17_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_17_end_0 = const()[name = string("x2_17_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_17_end_mask_0 = const()[name = string("x2_17_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_17 = slice_by_index(begin = x2_17_begin_0, end = x2_17_end_0, end_mask = x2_17_end_mask_0, x = var_1312)[name = string("x2_17")]; tensor var_1344_cast_fp16 = mul(x = x1_17, y = cos_3_cast_fp16)[name = string("op_1344_cast_fp16")]; tensor var_1345_cast_fp16 = mul(x = x2_17, y = sin_3_cast_fp16)[name = string("op_1345_cast_fp16")]; tensor var_1346_cast_fp16 = sub(x = var_1344_cast_fp16, y = var_1345_cast_fp16)[name = string("op_1346_cast_fp16")]; tensor var_1347_cast_fp16 = mul(x = x2_17, y = cos_3_cast_fp16)[name = string("op_1347_cast_fp16")]; tensor var_1348_cast_fp16 = mul(x = x1_17, y = sin_3_cast_fp16)[name = string("op_1348_cast_fp16")]; tensor var_1349_cast_fp16 = add(x = var_1347_cast_fp16, y = var_1348_cast_fp16)[name = string("op_1349_cast_fp16")]; bool rotated_17_interleave_0 = const()[name = string("rotated_17_interleave_0"), val = bool(false)]; tensor rotated_17_cast_fp16 = concat(axis = var_80, interleave = rotated_17_interleave_0, values = (var_1346_cast_fp16, var_1349_cast_fp16))[name = string("rotated_17_cast_fp16")]; tensor x1_19_begin_0 = const()[name = string("x1_19_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_19_end_0 = const()[name = string("x1_19_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_19_end_mask_0 = const()[name = string("x1_19_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_19 = slice_by_index(begin = x1_19_begin_0, end = x1_19_end_0, end_mask = x1_19_end_mask_0, x = var_1321)[name = string("x1_19")]; tensor x2_19_begin_0 = const()[name = string("x2_19_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_19_end_0 = const()[name = string("x2_19_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_19_end_mask_0 = const()[name = string("x2_19_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_19 = slice_by_index(begin = x2_19_begin_0, end = x2_19_end_0, end_mask = x2_19_end_mask_0, x = var_1321)[name = string("x2_19")]; tensor var_1365_cast_fp16 = mul(x = x1_19, y = cos_3_cast_fp16)[name = string("op_1365_cast_fp16")]; tensor var_1366_cast_fp16 = mul(x = x2_19, y = sin_3_cast_fp16)[name = string("op_1366_cast_fp16")]; tensor var_1367_cast_fp16 = sub(x = var_1365_cast_fp16, y = var_1366_cast_fp16)[name = string("op_1367_cast_fp16")]; tensor var_1368_cast_fp16 = mul(x = x2_19, y = cos_3_cast_fp16)[name = string("op_1368_cast_fp16")]; tensor var_1369_cast_fp16 = mul(x = x1_19, y = sin_3_cast_fp16)[name = string("op_1369_cast_fp16")]; tensor var_1370_cast_fp16 = add(x = var_1368_cast_fp16, y = var_1369_cast_fp16)[name = string("op_1370_cast_fp16")]; bool rotated_19_interleave_0 = const()[name = string("rotated_19_interleave_0"), val = bool(false)]; tensor rotated_19_cast_fp16 = concat(axis = var_80, interleave = rotated_19_interleave_0, values = (var_1367_cast_fp16, var_1370_cast_fp16))[name = string("rotated_19_cast_fp16")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; int32 concat_34_axis_0 = const()[name = string("concat_34_axis_0"), val = int32(0)]; bool concat_34_interleave_0 = const()[name = string("concat_34_interleave_0"), val = bool(false)]; tensor concat_34 = concat(axis = concat_34_axis_0, interleave = concat_34_interleave_0, values = (expand_dims_48, expand_dims_49, current_pos, expand_dims_51))[name = string("concat_34")]; tensor concat_35_values1_0 = const()[name = string("concat_35_values1_0"), val = tensor([0])]; tensor concat_35_values3_0 = const()[name = string("concat_35_values3_0"), val = tensor([0])]; int32 concat_35_axis_0 = const()[name = string("concat_35_axis_0"), val = int32(0)]; bool concat_35_interleave_0 = const()[name = string("concat_35_interleave_0"), val = bool(false)]; tensor concat_35 = concat(axis = concat_35_axis_0, interleave = concat_35_interleave_0, values = (expand_dims_52, concat_35_values1_0, var_587, concat_35_values3_0))[name = string("concat_35")]; tensor model_model_kv_cache_0_internal_tensor_assign_9_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_34, begin_mask = model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0, end = concat_35, end_mask = model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_9_stride_0, update = rotated_19_cast_fp16, x = coreml_update_state_39)[name = string("model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_72_write_state")]; tensor coreml_update_state_40 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_72")]; tensor expand_dims_54 = const()[name = string("expand_dims_54"), val = tensor([20])]; tensor expand_dims_55 = const()[name = string("expand_dims_55"), val = tensor([0])]; tensor expand_dims_57 = const()[name = string("expand_dims_57"), val = tensor([0])]; tensor expand_dims_58 = const()[name = string("expand_dims_58"), val = tensor([21])]; int32 concat_38_axis_0 = const()[name = string("concat_38_axis_0"), val = int32(0)]; bool concat_38_interleave_0 = const()[name = string("concat_38_interleave_0"), val = bool(false)]; tensor concat_38 = concat(axis = concat_38_axis_0, interleave = concat_38_interleave_0, values = (expand_dims_54, expand_dims_55, current_pos, expand_dims_57))[name = string("concat_38")]; tensor concat_39_values1_0 = const()[name = string("concat_39_values1_0"), val = tensor([0])]; tensor concat_39_values3_0 = const()[name = string("concat_39_values3_0"), val = tensor([0])]; int32 concat_39_axis_0 = const()[name = string("concat_39_axis_0"), val = int32(0)]; bool concat_39_interleave_0 = const()[name = string("concat_39_interleave_0"), val = bool(false)]; tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (expand_dims_58, concat_39_values1_0, var_587, concat_39_values3_0))[name = string("concat_39")]; tensor model_model_kv_cache_0_internal_tensor_assign_10_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_38, begin_mask = model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0, end = concat_39, end_mask = model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_10_stride_0, update = var_1330, x = coreml_update_state_40)[name = string("model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_73_write_state")]; tensor coreml_update_state_41 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_73")]; tensor var_1390_begin_0 = const()[name = string("op_1390_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_1390_end_0 = const()[name = string("op_1390_end_0"), val = tensor([5, 8, 4096, 64])]; tensor var_1390_end_mask_0 = const()[name = string("op_1390_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1390_cast_fp16 = slice_by_index(begin = var_1390_begin_0, end = var_1390_end_0, end_mask = var_1390_end_mask_0, x = coreml_update_state_41)[name = string("op_1390_cast_fp16")]; tensor K_layer_cache_9_axes_0 = const()[name = string("K_layer_cache_9_axes_0"), val = tensor([0])]; tensor K_layer_cache_9_cast_fp16 = squeeze(axes = K_layer_cache_9_axes_0, x = var_1390_cast_fp16)[name = string("K_layer_cache_9_cast_fp16")]; tensor var_1392_begin_0 = const()[name = string("op_1392_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_1392_end_0 = const()[name = string("op_1392_end_0"), val = tensor([21, 8, 4096, 64])]; tensor var_1392_end_mask_0 = const()[name = string("op_1392_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1392_cast_fp16 = slice_by_index(begin = var_1392_begin_0, end = var_1392_end_0, end_mask = var_1392_end_mask_0, x = coreml_update_state_41)[name = string("op_1392_cast_fp16")]; tensor V_layer_cache_9_axes_0 = const()[name = string("V_layer_cache_9_axes_0"), val = tensor([0])]; tensor V_layer_cache_9_cast_fp16 = squeeze(axes = V_layer_cache_9_axes_0, x = var_1392_cast_fp16)[name = string("V_layer_cache_9_cast_fp16")]; tensor x_123_axes_0 = const()[name = string("x_123_axes_0"), val = tensor([1])]; tensor x_123_cast_fp16 = expand_dims(axes = x_123_axes_0, x = K_layer_cache_9_cast_fp16)[name = string("x_123_cast_fp16")]; tensor var_1401 = const()[name = string("op_1401"), val = tensor([1, 4, 1, 1])]; tensor x_125_cast_fp16 = tile(reps = var_1401, x = x_123_cast_fp16)[name = string("x_125_cast_fp16")]; tensor var_1405 = const()[name = string("op_1405"), val = tensor([1, -1, 4096, 64])]; tensor key_states_19_cast_fp16 = reshape(shape = var_1405, x = x_125_cast_fp16)[name = string("key_states_19_cast_fp16")]; tensor x_129_axes_0 = const()[name = string("x_129_axes_0"), val = tensor([1])]; tensor x_129_cast_fp16 = expand_dims(axes = x_129_axes_0, x = V_layer_cache_9_cast_fp16)[name = string("x_129_cast_fp16")]; tensor var_1408 = const()[name = string("op_1408"), val = tensor([1, 4, 1, 1])]; tensor x_131_cast_fp16 = tile(reps = var_1408, x = x_129_cast_fp16)[name = string("x_131_cast_fp16")]; tensor var_1412 = const()[name = string("op_1412"), val = tensor([1, -1, 4096, 64])]; tensor value_states_19_cast_fp16 = reshape(shape = var_1412, x = x_131_cast_fp16)[name = string("value_states_19_cast_fp16")]; bool var_1415_transpose_x_1 = const()[name = string("op_1415_transpose_x_1"), val = bool(false)]; bool var_1415_transpose_y_1 = const()[name = string("op_1415_transpose_y_1"), val = bool(true)]; tensor var_1415_cast_fp16 = matmul(transpose_x = var_1415_transpose_x_1, transpose_y = var_1415_transpose_y_1, x = rotated_17_cast_fp16, y = key_states_19_cast_fp16)[name = string("op_1415_cast_fp16")]; fp16 var_1416_to_fp16 = const()[name = string("op_1416_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_17_cast_fp16 = mul(x = var_1415_cast_fp16, y = var_1416_to_fp16)[name = string("attn_weights_17_cast_fp16")]; tensor x_133_cast_fp16 = add(x = attn_weights_17_cast_fp16, y = causal_mask)[name = string("x_133_cast_fp16")]; tensor reduce_max_4_axes_0 = const()[name = string("reduce_max_4_axes_0"), val = tensor([-1])]; bool reduce_max_4_keep_dims_0 = const()[name = string("reduce_max_4_keep_dims_0"), val = bool(true)]; tensor reduce_max_4_cast_fp16 = reduce_max(axes = reduce_max_4_axes_0, keep_dims = reduce_max_4_keep_dims_0, x = x_133_cast_fp16)[name = string("reduce_max_4_cast_fp16")]; tensor x_135_cast_fp16 = sub(x = x_133_cast_fp16, y = reduce_max_4_cast_fp16)[name = string("x_135_cast_fp16")]; tensor exp_x_9_cast_fp16 = exp(x = x_135_cast_fp16)[name = string("exp_x_9_cast_fp16")]; tensor var_1427_axes_0 = const()[name = string("op_1427_axes_0"), val = tensor([-1])]; bool var_1427_keep_dims_0 = const()[name = string("op_1427_keep_dims_0"), val = bool(true)]; tensor var_1427_cast_fp16 = reduce_sum(axes = var_1427_axes_0, keep_dims = var_1427_keep_dims_0, x = exp_x_9_cast_fp16)[name = string("op_1427_cast_fp16")]; tensor attn_weights_19_cast_fp16 = real_div(x = exp_x_9_cast_fp16, y = var_1427_cast_fp16)[name = string("attn_weights_19_cast_fp16")]; bool attn_output_25_transpose_x_0 = const()[name = string("attn_output_25_transpose_x_0"), val = bool(false)]; bool attn_output_25_transpose_y_0 = const()[name = string("attn_output_25_transpose_y_0"), val = bool(false)]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_0, transpose_y = attn_output_25_transpose_y_0, x = attn_weights_19_cast_fp16, y = value_states_19_cast_fp16)[name = string("attn_output_25_cast_fp16")]; tensor var_1430_perm_0 = const()[name = string("op_1430_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1432 = const()[name = string("op_1432"), val = tensor([1, 1, 2048])]; tensor var_1430_cast_fp16 = transpose(perm = var_1430_perm_0, x = attn_output_25_cast_fp16)[name = string("transpose_46")]; tensor input_61_cast_fp16 = reshape(shape = var_1432, x = var_1430_cast_fp16)[name = string("input_61_cast_fp16")]; tensor model_model_layers_4_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(464934208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(467031424))))[name = string("model_model_layers_4_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_4_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_4_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_61_cast_fp16)[name = string("linear_4_cast_fp16")]; tensor hidden_states_37_cast_fp16 = add(x = hidden_states_33_cast_fp16, y = linear_4_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; fp16 const_76_promoted_to_fp16 = const()[name = string("const_76_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1438_cast_fp16 = mul(x = hidden_states_37_cast_fp16, y = const_76_promoted_to_fp16)[name = string("op_1438_cast_fp16")]; bool input_63_interleave_0 = const()[name = string("input_63_interleave_0"), val = bool(false)]; tensor input_63_cast_fp16 = concat(axis = var_80, interleave = input_63_interleave_0, values = (hidden_states_37_cast_fp16, var_1438_cast_fp16))[name = string("input_63_cast_fp16")]; tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_74_to_fp16, x = input_63_cast_fp16)[name = string("normed_37_cast_fp16")]; tensor normed_39_begin_0 = const()[name = string("normed_39_begin_0"), val = tensor([0, 0, 0])]; tensor normed_39_end_0 = const()[name = string("normed_39_end_0"), val = tensor([1, 1, 2048])]; tensor normed_39_end_mask_0 = const()[name = string("normed_39_end_mask_0"), val = tensor([true, true, false])]; tensor normed_39_cast_fp16 = slice_by_index(begin = normed_39_begin_0, end = normed_39_end_0, end_mask = normed_39_end_mask_0, x = normed_37_cast_fp16)[name = string("normed_39_cast_fp16")]; tensor const_79_promoted_to_fp16 = const()[name = string("const_79_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(467039680)))]; tensor x_137_cast_fp16 = mul(x = normed_39_cast_fp16, y = const_79_promoted_to_fp16)[name = string("x_137_cast_fp16")]; tensor var_1456 = const()[name = string("op_1456"), val = tensor([0, 2, 1])]; tensor input_65_axes_0 = const()[name = string("input_65_axes_0"), val = tensor([2])]; tensor var_1457 = transpose(perm = var_1456, x = x_137_cast_fp16)[name = string("transpose_45")]; tensor input_65 = expand_dims(axes = input_65_axes_0, x = var_1457)[name = string("input_65")]; string input_67_pad_type_0 = const()[name = string("input_67_pad_type_0"), val = string("valid")]; tensor input_67_strides_0 = const()[name = string("input_67_strides_0"), val = tensor([1, 1])]; tensor input_67_pad_0 = const()[name = string("input_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_67_dilations_0 = const()[name = string("input_67_dilations_0"), val = tensor([1, 1])]; int32 input_67_groups_0 = const()[name = string("input_67_groups_0"), val = int32(1)]; tensor input_67 = conv(dilations = input_67_dilations_0, groups = input_67_groups_0, pad = input_67_pad_0, pad_type = input_67_pad_type_0, strides = input_67_strides_0, weight = model_model_layers_4_mlp_gate_proj_weight_palettized, x = input_65)[name = string("input_67")]; string up_states_9_pad_type_0 = const()[name = string("up_states_9_pad_type_0"), val = string("valid")]; tensor up_states_9_strides_0 = const()[name = string("up_states_9_strides_0"), val = tensor([1, 1])]; tensor up_states_9_pad_0 = const()[name = string("up_states_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_9_dilations_0 = const()[name = string("up_states_9_dilations_0"), val = tensor([1, 1])]; int32 up_states_9_groups_0 = const()[name = string("up_states_9_groups_0"), val = int32(1)]; tensor up_states_9 = conv(dilations = up_states_9_dilations_0, groups = up_states_9_groups_0, pad = up_states_9_pad_0, pad_type = up_states_9_pad_type_0, strides = up_states_9_strides_0, weight = model_model_layers_4_mlp_up_proj_weight_palettized, x = input_65)[name = string("up_states_9")]; tensor gate_states_9 = silu(x = input_67)[name = string("gate_states_9")]; tensor input_69 = mul(x = gate_states_9, y = up_states_9)[name = string("input_69")]; string hidden_states_39_pad_type_0 = const()[name = string("hidden_states_39_pad_type_0"), val = string("valid")]; tensor hidden_states_39_strides_0 = const()[name = string("hidden_states_39_strides_0"), val = tensor([1, 1])]; tensor hidden_states_39_pad_0 = const()[name = string("hidden_states_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_39_dilations_0 = const()[name = string("hidden_states_39_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_39_groups_0 = const()[name = string("hidden_states_39_groups_0"), val = int32(1)]; tensor hidden_states_39 = conv(dilations = hidden_states_39_dilations_0, groups = hidden_states_39_groups_0, pad = hidden_states_39_pad_0, pad_type = hidden_states_39_pad_type_0, strides = hidden_states_39_strides_0, weight = model_model_layers_4_mlp_down_proj_weight_palettized, x = input_69)[name = string("hidden_states_39")]; tensor var_1479_axes_0 = const()[name = string("op_1479_axes_0"), val = tensor([2])]; tensor var_1479 = squeeze(axes = var_1479_axes_0, x = hidden_states_39)[name = string("op_1479")]; tensor var_1480 = const()[name = string("op_1480"), val = tensor([0, 2, 1])]; tensor var_1481 = transpose(perm = var_1480, x = var_1479)[name = string("transpose_44")]; tensor hidden_states_41_cast_fp16 = add(x = hidden_states_37_cast_fp16, y = var_1481)[name = string("hidden_states_41_cast_fp16")]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1484_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_1484_cast_fp16")]; bool input_71_interleave_0 = const()[name = string("input_71_interleave_0"), val = bool(false)]; tensor input_71_cast_fp16 = concat(axis = var_80, interleave = input_71_interleave_0, values = (hidden_states_41_cast_fp16, var_1484_cast_fp16))[name = string("input_71_cast_fp16")]; tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_74_to_fp16, x = input_71_cast_fp16)[name = string("normed_41_cast_fp16")]; tensor normed_43_begin_0 = const()[name = string("normed_43_begin_0"), val = tensor([0, 0, 0])]; tensor normed_43_end_0 = const()[name = string("normed_43_end_0"), val = tensor([1, 1, 2048])]; tensor normed_43_end_mask_0 = const()[name = string("normed_43_end_mask_0"), val = tensor([true, true, false])]; tensor normed_43_cast_fp16 = slice_by_index(begin = normed_43_begin_0, end = normed_43_end_0, end_mask = normed_43_end_mask_0, x = normed_41_cast_fp16)[name = string("normed_43_cast_fp16")]; tensor const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(467043840)))]; tensor hidden_states_43_cast_fp16 = mul(x = normed_43_cast_fp16, y = const_83_promoted_to_fp16)[name = string("hidden_states_43_cast_fp16")]; tensor var_1498 = const()[name = string("op_1498"), val = tensor([0, 2, 1])]; tensor var_1500_axes_0 = const()[name = string("op_1500_axes_0"), val = tensor([2])]; tensor var_1499_cast_fp16 = transpose(perm = var_1498, x = hidden_states_43_cast_fp16)[name = string("transpose_43")]; tensor var_1500_cast_fp16 = expand_dims(axes = var_1500_axes_0, x = var_1499_cast_fp16)[name = string("op_1500_cast_fp16")]; string var_1507_pad_type_0 = const()[name = string("op_1507_pad_type_0"), val = string("valid")]; tensor var_1507_strides_0 = const()[name = string("op_1507_strides_0"), val = tensor([1, 1])]; tensor var_1507_pad_0 = const()[name = string("op_1507_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1507_dilations_0 = const()[name = string("op_1507_dilations_0"), val = tensor([1, 1])]; int32 var_1507_groups_0 = const()[name = string("op_1507_groups_0"), val = int32(1)]; tensor var_1507 = conv(dilations = var_1507_dilations_0, groups = var_1507_groups_0, pad = var_1507_pad_0, pad_type = var_1507_pad_type_0, strides = var_1507_strides_0, weight = model_model_layers_5_self_attn_q_proj_weight_palettized, x = var_1500_cast_fp16)[name = string("op_1507")]; tensor var_1508 = const()[name = string("op_1508"), val = tensor([1, 32, 1, 64])]; tensor var_1509 = reshape(shape = var_1508, x = var_1507)[name = string("op_1509")]; string var_1516_pad_type_0 = const()[name = string("op_1516_pad_type_0"), val = string("valid")]; tensor var_1516_strides_0 = const()[name = string("op_1516_strides_0"), val = tensor([1, 1])]; tensor var_1516_pad_0 = const()[name = string("op_1516_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1516_dilations_0 = const()[name = string("op_1516_dilations_0"), val = tensor([1, 1])]; int32 var_1516_groups_0 = const()[name = string("op_1516_groups_0"), val = int32(1)]; tensor var_1516 = conv(dilations = var_1516_dilations_0, groups = var_1516_groups_0, pad = var_1516_pad_0, pad_type = var_1516_pad_type_0, strides = var_1516_strides_0, weight = model_model_layers_5_self_attn_k_proj_weight_palettized, x = var_1500_cast_fp16)[name = string("op_1516")]; tensor var_1517 = const()[name = string("op_1517"), val = tensor([1, 8, 1, 64])]; tensor var_1518 = reshape(shape = var_1517, x = var_1516)[name = string("op_1518")]; string var_1525_pad_type_0 = const()[name = string("op_1525_pad_type_0"), val = string("valid")]; tensor var_1525_strides_0 = const()[name = string("op_1525_strides_0"), val = tensor([1, 1])]; tensor var_1525_pad_0 = const()[name = string("op_1525_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1525_dilations_0 = const()[name = string("op_1525_dilations_0"), val = tensor([1, 1])]; int32 var_1525_groups_0 = const()[name = string("op_1525_groups_0"), val = int32(1)]; tensor var_1525 = conv(dilations = var_1525_dilations_0, groups = var_1525_groups_0, pad = var_1525_pad_0, pad_type = var_1525_pad_type_0, strides = var_1525_strides_0, weight = model_model_layers_5_self_attn_v_proj_weight_palettized, x = var_1500_cast_fp16)[name = string("op_1525")]; tensor var_1526 = const()[name = string("op_1526"), val = tensor([1, 8, 1, 64])]; tensor var_1527 = reshape(shape = var_1526, x = var_1525)[name = string("op_1527")]; tensor x1_21_begin_0 = const()[name = string("x1_21_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_21_end_0 = const()[name = string("x1_21_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_21_end_mask_0 = const()[name = string("x1_21_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_21 = slice_by_index(begin = x1_21_begin_0, end = x1_21_end_0, end_mask = x1_21_end_mask_0, x = var_1509)[name = string("x1_21")]; tensor x2_21_begin_0 = const()[name = string("x2_21_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_21_end_0 = const()[name = string("x2_21_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_21_end_mask_0 = const()[name = string("x2_21_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_21 = slice_by_index(begin = x2_21_begin_0, end = x2_21_end_0, end_mask = x2_21_end_mask_0, x = var_1509)[name = string("x2_21")]; tensor var_1541_cast_fp16 = mul(x = x1_21, y = cos_3_cast_fp16)[name = string("op_1541_cast_fp16")]; tensor var_1542_cast_fp16 = mul(x = x2_21, y = sin_3_cast_fp16)[name = string("op_1542_cast_fp16")]; tensor var_1543_cast_fp16 = sub(x = var_1541_cast_fp16, y = var_1542_cast_fp16)[name = string("op_1543_cast_fp16")]; tensor var_1544_cast_fp16 = mul(x = x2_21, y = cos_3_cast_fp16)[name = string("op_1544_cast_fp16")]; tensor var_1545_cast_fp16 = mul(x = x1_21, y = sin_3_cast_fp16)[name = string("op_1545_cast_fp16")]; tensor var_1546_cast_fp16 = add(x = var_1544_cast_fp16, y = var_1545_cast_fp16)[name = string("op_1546_cast_fp16")]; bool rotated_21_interleave_0 = const()[name = string("rotated_21_interleave_0"), val = bool(false)]; tensor rotated_21_cast_fp16 = concat(axis = var_80, interleave = rotated_21_interleave_0, values = (var_1543_cast_fp16, var_1546_cast_fp16))[name = string("rotated_21_cast_fp16")]; tensor x1_23_begin_0 = const()[name = string("x1_23_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_23_end_0 = const()[name = string("x1_23_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_23_end_mask_0 = const()[name = string("x1_23_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_23 = slice_by_index(begin = x1_23_begin_0, end = x1_23_end_0, end_mask = x1_23_end_mask_0, x = var_1518)[name = string("x1_23")]; tensor x2_23_begin_0 = const()[name = string("x2_23_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_23_end_0 = const()[name = string("x2_23_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_23_end_mask_0 = const()[name = string("x2_23_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_23 = slice_by_index(begin = x2_23_begin_0, end = x2_23_end_0, end_mask = x2_23_end_mask_0, x = var_1518)[name = string("x2_23")]; tensor var_1562_cast_fp16 = mul(x = x1_23, y = cos_3_cast_fp16)[name = string("op_1562_cast_fp16")]; tensor var_1563_cast_fp16 = mul(x = x2_23, y = sin_3_cast_fp16)[name = string("op_1563_cast_fp16")]; tensor var_1564_cast_fp16 = sub(x = var_1562_cast_fp16, y = var_1563_cast_fp16)[name = string("op_1564_cast_fp16")]; tensor var_1565_cast_fp16 = mul(x = x2_23, y = cos_3_cast_fp16)[name = string("op_1565_cast_fp16")]; tensor var_1566_cast_fp16 = mul(x = x1_23, y = sin_3_cast_fp16)[name = string("op_1566_cast_fp16")]; tensor var_1567_cast_fp16 = add(x = var_1565_cast_fp16, y = var_1566_cast_fp16)[name = string("op_1567_cast_fp16")]; bool rotated_23_interleave_0 = const()[name = string("rotated_23_interleave_0"), val = bool(false)]; tensor rotated_23_cast_fp16 = concat(axis = var_80, interleave = rotated_23_interleave_0, values = (var_1564_cast_fp16, var_1567_cast_fp16))[name = string("rotated_23_cast_fp16")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_60, expand_dims_61, current_pos, expand_dims_63))[name = string("concat_42")]; tensor concat_43_values1_0 = const()[name = string("concat_43_values1_0"), val = tensor([0])]; tensor concat_43_values3_0 = const()[name = string("concat_43_values3_0"), val = tensor([0])]; int32 concat_43_axis_0 = const()[name = string("concat_43_axis_0"), val = int32(0)]; bool concat_43_interleave_0 = const()[name = string("concat_43_interleave_0"), val = bool(false)]; tensor concat_43 = concat(axis = concat_43_axis_0, interleave = concat_43_interleave_0, values = (expand_dims_64, concat_43_values1_0, var_587, concat_43_values3_0))[name = string("concat_43")]; tensor model_model_kv_cache_0_internal_tensor_assign_11_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_42, begin_mask = model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0, end = concat_43, end_mask = model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_11_stride_0, update = rotated_23_cast_fp16, x = coreml_update_state_41)[name = string("model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_74_write_state")]; tensor coreml_update_state_42 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_74")]; tensor expand_dims_66 = const()[name = string("expand_dims_66"), val = tensor([21])]; tensor expand_dims_67 = const()[name = string("expand_dims_67"), val = tensor([0])]; tensor expand_dims_69 = const()[name = string("expand_dims_69"), val = tensor([0])]; tensor expand_dims_70 = const()[name = string("expand_dims_70"), val = tensor([22])]; int32 concat_46_axis_0 = const()[name = string("concat_46_axis_0"), val = int32(0)]; bool concat_46_interleave_0 = const()[name = string("concat_46_interleave_0"), val = bool(false)]; tensor concat_46 = concat(axis = concat_46_axis_0, interleave = concat_46_interleave_0, values = (expand_dims_66, expand_dims_67, current_pos, expand_dims_69))[name = string("concat_46")]; tensor concat_47_values1_0 = const()[name = string("concat_47_values1_0"), val = tensor([0])]; tensor concat_47_values3_0 = const()[name = string("concat_47_values3_0"), val = tensor([0])]; int32 concat_47_axis_0 = const()[name = string("concat_47_axis_0"), val = int32(0)]; bool concat_47_interleave_0 = const()[name = string("concat_47_interleave_0"), val = bool(false)]; tensor concat_47 = concat(axis = concat_47_axis_0, interleave = concat_47_interleave_0, values = (expand_dims_70, concat_47_values1_0, var_587, concat_47_values3_0))[name = string("concat_47")]; tensor model_model_kv_cache_0_internal_tensor_assign_12_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_46, begin_mask = model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0, end = concat_47, end_mask = model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_12_stride_0, update = var_1527, x = coreml_update_state_42)[name = string("model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_75_write_state")]; tensor coreml_update_state_43 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_75")]; tensor var_1587_begin_0 = const()[name = string("op_1587_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_1587_end_0 = const()[name = string("op_1587_end_0"), val = tensor([6, 8, 4096, 64])]; tensor var_1587_end_mask_0 = const()[name = string("op_1587_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1587_cast_fp16 = slice_by_index(begin = var_1587_begin_0, end = var_1587_end_0, end_mask = var_1587_end_mask_0, x = coreml_update_state_43)[name = string("op_1587_cast_fp16")]; tensor K_layer_cache_11_axes_0 = const()[name = string("K_layer_cache_11_axes_0"), val = tensor([0])]; tensor K_layer_cache_11_cast_fp16 = squeeze(axes = K_layer_cache_11_axes_0, x = var_1587_cast_fp16)[name = string("K_layer_cache_11_cast_fp16")]; tensor var_1589_begin_0 = const()[name = string("op_1589_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_1589_end_0 = const()[name = string("op_1589_end_0"), val = tensor([22, 8, 4096, 64])]; tensor var_1589_end_mask_0 = const()[name = string("op_1589_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1589_cast_fp16 = slice_by_index(begin = var_1589_begin_0, end = var_1589_end_0, end_mask = var_1589_end_mask_0, x = coreml_update_state_43)[name = string("op_1589_cast_fp16")]; tensor V_layer_cache_11_axes_0 = const()[name = string("V_layer_cache_11_axes_0"), val = tensor([0])]; tensor V_layer_cache_11_cast_fp16 = squeeze(axes = V_layer_cache_11_axes_0, x = var_1589_cast_fp16)[name = string("V_layer_cache_11_cast_fp16")]; tensor x_151_axes_0 = const()[name = string("x_151_axes_0"), val = tensor([1])]; tensor x_151_cast_fp16 = expand_dims(axes = x_151_axes_0, x = K_layer_cache_11_cast_fp16)[name = string("x_151_cast_fp16")]; tensor var_1598 = const()[name = string("op_1598"), val = tensor([1, 4, 1, 1])]; tensor x_153_cast_fp16 = tile(reps = var_1598, x = x_151_cast_fp16)[name = string("x_153_cast_fp16")]; tensor var_1602 = const()[name = string("op_1602"), val = tensor([1, -1, 4096, 64])]; tensor key_states_23_cast_fp16 = reshape(shape = var_1602, x = x_153_cast_fp16)[name = string("key_states_23_cast_fp16")]; tensor x_157_axes_0 = const()[name = string("x_157_axes_0"), val = tensor([1])]; tensor x_157_cast_fp16 = expand_dims(axes = x_157_axes_0, x = V_layer_cache_11_cast_fp16)[name = string("x_157_cast_fp16")]; tensor var_1605 = const()[name = string("op_1605"), val = tensor([1, 4, 1, 1])]; tensor x_159_cast_fp16 = tile(reps = var_1605, x = x_157_cast_fp16)[name = string("x_159_cast_fp16")]; tensor var_1609 = const()[name = string("op_1609"), val = tensor([1, -1, 4096, 64])]; tensor value_states_23_cast_fp16 = reshape(shape = var_1609, x = x_159_cast_fp16)[name = string("value_states_23_cast_fp16")]; bool var_1612_transpose_x_1 = const()[name = string("op_1612_transpose_x_1"), val = bool(false)]; bool var_1612_transpose_y_1 = const()[name = string("op_1612_transpose_y_1"), val = bool(true)]; tensor var_1612_cast_fp16 = matmul(transpose_x = var_1612_transpose_x_1, transpose_y = var_1612_transpose_y_1, x = rotated_21_cast_fp16, y = key_states_23_cast_fp16)[name = string("op_1612_cast_fp16")]; fp16 var_1613_to_fp16 = const()[name = string("op_1613_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_21_cast_fp16 = mul(x = var_1612_cast_fp16, y = var_1613_to_fp16)[name = string("attn_weights_21_cast_fp16")]; tensor x_161_cast_fp16 = add(x = attn_weights_21_cast_fp16, y = causal_mask)[name = string("x_161_cast_fp16")]; tensor reduce_max_5_axes_0 = const()[name = string("reduce_max_5_axes_0"), val = tensor([-1])]; bool reduce_max_5_keep_dims_0 = const()[name = string("reduce_max_5_keep_dims_0"), val = bool(true)]; tensor reduce_max_5_cast_fp16 = reduce_max(axes = reduce_max_5_axes_0, keep_dims = reduce_max_5_keep_dims_0, x = x_161_cast_fp16)[name = string("reduce_max_5_cast_fp16")]; tensor x_163_cast_fp16 = sub(x = x_161_cast_fp16, y = reduce_max_5_cast_fp16)[name = string("x_163_cast_fp16")]; tensor exp_x_11_cast_fp16 = exp(x = x_163_cast_fp16)[name = string("exp_x_11_cast_fp16")]; tensor var_1624_axes_0 = const()[name = string("op_1624_axes_0"), val = tensor([-1])]; bool var_1624_keep_dims_0 = const()[name = string("op_1624_keep_dims_0"), val = bool(true)]; tensor var_1624_cast_fp16 = reduce_sum(axes = var_1624_axes_0, keep_dims = var_1624_keep_dims_0, x = exp_x_11_cast_fp16)[name = string("op_1624_cast_fp16")]; tensor attn_weights_23_cast_fp16 = real_div(x = exp_x_11_cast_fp16, y = var_1624_cast_fp16)[name = string("attn_weights_23_cast_fp16")]; bool attn_output_31_transpose_x_0 = const()[name = string("attn_output_31_transpose_x_0"), val = bool(false)]; bool attn_output_31_transpose_y_0 = const()[name = string("attn_output_31_transpose_y_0"), val = bool(false)]; tensor attn_output_31_cast_fp16 = matmul(transpose_x = attn_output_31_transpose_x_0, transpose_y = attn_output_31_transpose_y_0, x = attn_weights_23_cast_fp16, y = value_states_23_cast_fp16)[name = string("attn_output_31_cast_fp16")]; tensor var_1627_perm_0 = const()[name = string("op_1627_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1629 = const()[name = string("op_1629"), val = tensor([1, 1, 2048])]; tensor var_1627_cast_fp16 = transpose(perm = var_1627_perm_0, x = attn_output_31_cast_fp16)[name = string("transpose_42")]; tensor input_75_cast_fp16 = reshape(shape = var_1629, x = var_1627_cast_fp16)[name = string("input_75_cast_fp16")]; tensor model_model_layers_5_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(467048000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(469145216))))[name = string("model_model_layers_5_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_5_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_5_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_75_cast_fp16)[name = string("linear_5_cast_fp16")]; tensor hidden_states_45_cast_fp16 = add(x = hidden_states_41_cast_fp16, y = linear_5_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1635_cast_fp16 = mul(x = hidden_states_45_cast_fp16, y = const_92_promoted_to_fp16)[name = string("op_1635_cast_fp16")]; bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; tensor input_77_cast_fp16 = concat(axis = var_80, interleave = input_77_interleave_0, values = (hidden_states_45_cast_fp16, var_1635_cast_fp16))[name = string("input_77_cast_fp16")]; tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_74_to_fp16, x = input_77_cast_fp16)[name = string("normed_45_cast_fp16")]; tensor normed_47_begin_0 = const()[name = string("normed_47_begin_0"), val = tensor([0, 0, 0])]; tensor normed_47_end_0 = const()[name = string("normed_47_end_0"), val = tensor([1, 1, 2048])]; tensor normed_47_end_mask_0 = const()[name = string("normed_47_end_mask_0"), val = tensor([true, true, false])]; tensor normed_47_cast_fp16 = slice_by_index(begin = normed_47_begin_0, end = normed_47_end_0, end_mask = normed_47_end_mask_0, x = normed_45_cast_fp16)[name = string("normed_47_cast_fp16")]; tensor const_95_promoted_to_fp16 = const()[name = string("const_95_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(469153472)))]; tensor x_165_cast_fp16 = mul(x = normed_47_cast_fp16, y = const_95_promoted_to_fp16)[name = string("x_165_cast_fp16")]; tensor var_1653 = const()[name = string("op_1653"), val = tensor([0, 2, 1])]; tensor input_79_axes_0 = const()[name = string("input_79_axes_0"), val = tensor([2])]; tensor var_1654 = transpose(perm = var_1653, x = x_165_cast_fp16)[name = string("transpose_41")]; tensor input_79 = expand_dims(axes = input_79_axes_0, x = var_1654)[name = string("input_79")]; string input_81_pad_type_0 = const()[name = string("input_81_pad_type_0"), val = string("valid")]; tensor input_81_strides_0 = const()[name = string("input_81_strides_0"), val = tensor([1, 1])]; tensor input_81_pad_0 = const()[name = string("input_81_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_81_dilations_0 = const()[name = string("input_81_dilations_0"), val = tensor([1, 1])]; int32 input_81_groups_0 = const()[name = string("input_81_groups_0"), val = int32(1)]; tensor input_81 = conv(dilations = input_81_dilations_0, groups = input_81_groups_0, pad = input_81_pad_0, pad_type = input_81_pad_type_0, strides = input_81_strides_0, weight = model_model_layers_5_mlp_gate_proj_weight_palettized, x = input_79)[name = string("input_81")]; string up_states_11_pad_type_0 = const()[name = string("up_states_11_pad_type_0"), val = string("valid")]; tensor up_states_11_strides_0 = const()[name = string("up_states_11_strides_0"), val = tensor([1, 1])]; tensor up_states_11_pad_0 = const()[name = string("up_states_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_11_dilations_0 = const()[name = string("up_states_11_dilations_0"), val = tensor([1, 1])]; int32 up_states_11_groups_0 = const()[name = string("up_states_11_groups_0"), val = int32(1)]; tensor up_states_11 = conv(dilations = up_states_11_dilations_0, groups = up_states_11_groups_0, pad = up_states_11_pad_0, pad_type = up_states_11_pad_type_0, strides = up_states_11_strides_0, weight = model_model_layers_5_mlp_up_proj_weight_palettized, x = input_79)[name = string("up_states_11")]; tensor gate_states_11 = silu(x = input_81)[name = string("gate_states_11")]; tensor input_83 = mul(x = gate_states_11, y = up_states_11)[name = string("input_83")]; string hidden_states_47_pad_type_0 = const()[name = string("hidden_states_47_pad_type_0"), val = string("valid")]; tensor hidden_states_47_strides_0 = const()[name = string("hidden_states_47_strides_0"), val = tensor([1, 1])]; tensor hidden_states_47_pad_0 = const()[name = string("hidden_states_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_47_dilations_0 = const()[name = string("hidden_states_47_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_47_groups_0 = const()[name = string("hidden_states_47_groups_0"), val = int32(1)]; tensor hidden_states_47 = conv(dilations = hidden_states_47_dilations_0, groups = hidden_states_47_groups_0, pad = hidden_states_47_pad_0, pad_type = hidden_states_47_pad_type_0, strides = hidden_states_47_strides_0, weight = model_model_layers_5_mlp_down_proj_weight_palettized, x = input_83)[name = string("hidden_states_47")]; tensor var_1676_axes_0 = const()[name = string("op_1676_axes_0"), val = tensor([2])]; tensor var_1676 = squeeze(axes = var_1676_axes_0, x = hidden_states_47)[name = string("op_1676")]; tensor var_1677 = const()[name = string("op_1677"), val = tensor([0, 2, 1])]; tensor var_1678 = transpose(perm = var_1677, x = var_1676)[name = string("transpose_40")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = var_1678)[name = string("hidden_states_49_cast_fp16")]; fp16 const_96_promoted_to_fp16 = const()[name = string("const_96_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1681_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_96_promoted_to_fp16)[name = string("op_1681_cast_fp16")]; bool input_85_interleave_0 = const()[name = string("input_85_interleave_0"), val = bool(false)]; tensor input_85_cast_fp16 = concat(axis = var_80, interleave = input_85_interleave_0, values = (hidden_states_49_cast_fp16, var_1681_cast_fp16))[name = string("input_85_cast_fp16")]; tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_74_to_fp16, x = input_85_cast_fp16)[name = string("normed_49_cast_fp16")]; tensor normed_51_begin_0 = const()[name = string("normed_51_begin_0"), val = tensor([0, 0, 0])]; tensor normed_51_end_0 = const()[name = string("normed_51_end_0"), val = tensor([1, 1, 2048])]; tensor normed_51_end_mask_0 = const()[name = string("normed_51_end_mask_0"), val = tensor([true, true, false])]; tensor normed_51_cast_fp16 = slice_by_index(begin = normed_51_begin_0, end = normed_51_end_0, end_mask = normed_51_end_mask_0, x = normed_49_cast_fp16)[name = string("normed_51_cast_fp16")]; tensor const_99_promoted_to_fp16 = const()[name = string("const_99_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(469157632)))]; tensor hidden_states_51_cast_fp16 = mul(x = normed_51_cast_fp16, y = const_99_promoted_to_fp16)[name = string("hidden_states_51_cast_fp16")]; tensor var_1695 = const()[name = string("op_1695"), val = tensor([0, 2, 1])]; tensor var_1697_axes_0 = const()[name = string("op_1697_axes_0"), val = tensor([2])]; tensor var_1696_cast_fp16 = transpose(perm = var_1695, x = hidden_states_51_cast_fp16)[name = string("transpose_39")]; tensor var_1697_cast_fp16 = expand_dims(axes = var_1697_axes_0, x = var_1696_cast_fp16)[name = string("op_1697_cast_fp16")]; string var_1704_pad_type_0 = const()[name = string("op_1704_pad_type_0"), val = string("valid")]; tensor var_1704_strides_0 = const()[name = string("op_1704_strides_0"), val = tensor([1, 1])]; tensor var_1704_pad_0 = const()[name = string("op_1704_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1704_dilations_0 = const()[name = string("op_1704_dilations_0"), val = tensor([1, 1])]; int32 var_1704_groups_0 = const()[name = string("op_1704_groups_0"), val = int32(1)]; tensor var_1704 = conv(dilations = var_1704_dilations_0, groups = var_1704_groups_0, pad = var_1704_pad_0, pad_type = var_1704_pad_type_0, strides = var_1704_strides_0, weight = model_model_layers_6_self_attn_q_proj_weight_palettized, x = var_1697_cast_fp16)[name = string("op_1704")]; tensor var_1705 = const()[name = string("op_1705"), val = tensor([1, 32, 1, 64])]; tensor var_1706 = reshape(shape = var_1705, x = var_1704)[name = string("op_1706")]; string var_1713_pad_type_0 = const()[name = string("op_1713_pad_type_0"), val = string("valid")]; tensor var_1713_strides_0 = const()[name = string("op_1713_strides_0"), val = tensor([1, 1])]; tensor var_1713_pad_0 = const()[name = string("op_1713_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1713_dilations_0 = const()[name = string("op_1713_dilations_0"), val = tensor([1, 1])]; int32 var_1713_groups_0 = const()[name = string("op_1713_groups_0"), val = int32(1)]; tensor var_1713 = conv(dilations = var_1713_dilations_0, groups = var_1713_groups_0, pad = var_1713_pad_0, pad_type = var_1713_pad_type_0, strides = var_1713_strides_0, weight = model_model_layers_6_self_attn_k_proj_weight_palettized, x = var_1697_cast_fp16)[name = string("op_1713")]; tensor var_1714 = const()[name = string("op_1714"), val = tensor([1, 8, 1, 64])]; tensor var_1715 = reshape(shape = var_1714, x = var_1713)[name = string("op_1715")]; string var_1722_pad_type_0 = const()[name = string("op_1722_pad_type_0"), val = string("valid")]; tensor var_1722_strides_0 = const()[name = string("op_1722_strides_0"), val = tensor([1, 1])]; tensor var_1722_pad_0 = const()[name = string("op_1722_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1722_dilations_0 = const()[name = string("op_1722_dilations_0"), val = tensor([1, 1])]; int32 var_1722_groups_0 = const()[name = string("op_1722_groups_0"), val = int32(1)]; tensor var_1722 = conv(dilations = var_1722_dilations_0, groups = var_1722_groups_0, pad = var_1722_pad_0, pad_type = var_1722_pad_type_0, strides = var_1722_strides_0, weight = model_model_layers_6_self_attn_v_proj_weight_palettized, x = var_1697_cast_fp16)[name = string("op_1722")]; tensor var_1723 = const()[name = string("op_1723"), val = tensor([1, 8, 1, 64])]; tensor var_1724 = reshape(shape = var_1723, x = var_1722)[name = string("op_1724")]; tensor x1_25_begin_0 = const()[name = string("x1_25_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_25_end_0 = const()[name = string("x1_25_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_25_end_mask_0 = const()[name = string("x1_25_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_25 = slice_by_index(begin = x1_25_begin_0, end = x1_25_end_0, end_mask = x1_25_end_mask_0, x = var_1706)[name = string("x1_25")]; tensor x2_25_begin_0 = const()[name = string("x2_25_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_25_end_0 = const()[name = string("x2_25_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_25_end_mask_0 = const()[name = string("x2_25_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_25 = slice_by_index(begin = x2_25_begin_0, end = x2_25_end_0, end_mask = x2_25_end_mask_0, x = var_1706)[name = string("x2_25")]; tensor var_1738_cast_fp16 = mul(x = x1_25, y = cos_3_cast_fp16)[name = string("op_1738_cast_fp16")]; tensor var_1739_cast_fp16 = mul(x = x2_25, y = sin_3_cast_fp16)[name = string("op_1739_cast_fp16")]; tensor var_1740_cast_fp16 = sub(x = var_1738_cast_fp16, y = var_1739_cast_fp16)[name = string("op_1740_cast_fp16")]; tensor var_1741_cast_fp16 = mul(x = x2_25, y = cos_3_cast_fp16)[name = string("op_1741_cast_fp16")]; tensor var_1742_cast_fp16 = mul(x = x1_25, y = sin_3_cast_fp16)[name = string("op_1742_cast_fp16")]; tensor var_1743_cast_fp16 = add(x = var_1741_cast_fp16, y = var_1742_cast_fp16)[name = string("op_1743_cast_fp16")]; bool rotated_25_interleave_0 = const()[name = string("rotated_25_interleave_0"), val = bool(false)]; tensor rotated_25_cast_fp16 = concat(axis = var_80, interleave = rotated_25_interleave_0, values = (var_1740_cast_fp16, var_1743_cast_fp16))[name = string("rotated_25_cast_fp16")]; tensor x1_27_begin_0 = const()[name = string("x1_27_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_27_end_0 = const()[name = string("x1_27_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_27_end_mask_0 = const()[name = string("x1_27_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_27 = slice_by_index(begin = x1_27_begin_0, end = x1_27_end_0, end_mask = x1_27_end_mask_0, x = var_1715)[name = string("x1_27")]; tensor x2_27_begin_0 = const()[name = string("x2_27_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_27_end_0 = const()[name = string("x2_27_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_27_end_mask_0 = const()[name = string("x2_27_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_27 = slice_by_index(begin = x2_27_begin_0, end = x2_27_end_0, end_mask = x2_27_end_mask_0, x = var_1715)[name = string("x2_27")]; tensor var_1759_cast_fp16 = mul(x = x1_27, y = cos_3_cast_fp16)[name = string("op_1759_cast_fp16")]; tensor var_1760_cast_fp16 = mul(x = x2_27, y = sin_3_cast_fp16)[name = string("op_1760_cast_fp16")]; tensor var_1761_cast_fp16 = sub(x = var_1759_cast_fp16, y = var_1760_cast_fp16)[name = string("op_1761_cast_fp16")]; tensor var_1762_cast_fp16 = mul(x = x2_27, y = cos_3_cast_fp16)[name = string("op_1762_cast_fp16")]; tensor var_1763_cast_fp16 = mul(x = x1_27, y = sin_3_cast_fp16)[name = string("op_1763_cast_fp16")]; tensor var_1764_cast_fp16 = add(x = var_1762_cast_fp16, y = var_1763_cast_fp16)[name = string("op_1764_cast_fp16")]; bool rotated_27_interleave_0 = const()[name = string("rotated_27_interleave_0"), val = bool(false)]; tensor rotated_27_cast_fp16 = concat(axis = var_80, interleave = rotated_27_interleave_0, values = (var_1761_cast_fp16, var_1764_cast_fp16))[name = string("rotated_27_cast_fp16")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; int32 concat_50_axis_0 = const()[name = string("concat_50_axis_0"), val = int32(0)]; bool concat_50_interleave_0 = const()[name = string("concat_50_interleave_0"), val = bool(false)]; tensor concat_50 = concat(axis = concat_50_axis_0, interleave = concat_50_interleave_0, values = (expand_dims_72, expand_dims_73, current_pos, expand_dims_75))[name = string("concat_50")]; tensor concat_51_values1_0 = const()[name = string("concat_51_values1_0"), val = tensor([0])]; tensor concat_51_values3_0 = const()[name = string("concat_51_values3_0"), val = tensor([0])]; int32 concat_51_axis_0 = const()[name = string("concat_51_axis_0"), val = int32(0)]; bool concat_51_interleave_0 = const()[name = string("concat_51_interleave_0"), val = bool(false)]; tensor concat_51 = concat(axis = concat_51_axis_0, interleave = concat_51_interleave_0, values = (expand_dims_76, concat_51_values1_0, var_587, concat_51_values3_0))[name = string("concat_51")]; tensor model_model_kv_cache_0_internal_tensor_assign_13_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_50, begin_mask = model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0, end = concat_51, end_mask = model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_13_stride_0, update = rotated_27_cast_fp16, x = coreml_update_state_43)[name = string("model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_76_write_state")]; tensor coreml_update_state_44 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_76")]; tensor expand_dims_78 = const()[name = string("expand_dims_78"), val = tensor([22])]; tensor expand_dims_79 = const()[name = string("expand_dims_79"), val = tensor([0])]; tensor expand_dims_81 = const()[name = string("expand_dims_81"), val = tensor([0])]; tensor expand_dims_82 = const()[name = string("expand_dims_82"), val = tensor([23])]; int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_78, expand_dims_79, current_pos, expand_dims_81))[name = string("concat_54")]; tensor concat_55_values1_0 = const()[name = string("concat_55_values1_0"), val = tensor([0])]; tensor concat_55_values3_0 = const()[name = string("concat_55_values3_0"), val = tensor([0])]; int32 concat_55_axis_0 = const()[name = string("concat_55_axis_0"), val = int32(0)]; bool concat_55_interleave_0 = const()[name = string("concat_55_interleave_0"), val = bool(false)]; tensor concat_55 = concat(axis = concat_55_axis_0, interleave = concat_55_interleave_0, values = (expand_dims_82, concat_55_values1_0, var_587, concat_55_values3_0))[name = string("concat_55")]; tensor model_model_kv_cache_0_internal_tensor_assign_14_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_54, begin_mask = model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0, end = concat_55, end_mask = model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_14_stride_0, update = var_1724, x = coreml_update_state_44)[name = string("model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_77_write_state")]; tensor coreml_update_state_45 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_77")]; tensor var_1784_begin_0 = const()[name = string("op_1784_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_1784_end_0 = const()[name = string("op_1784_end_0"), val = tensor([7, 8, 4096, 64])]; tensor var_1784_end_mask_0 = const()[name = string("op_1784_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1784_cast_fp16 = slice_by_index(begin = var_1784_begin_0, end = var_1784_end_0, end_mask = var_1784_end_mask_0, x = coreml_update_state_45)[name = string("op_1784_cast_fp16")]; tensor K_layer_cache_13_axes_0 = const()[name = string("K_layer_cache_13_axes_0"), val = tensor([0])]; tensor K_layer_cache_13_cast_fp16 = squeeze(axes = K_layer_cache_13_axes_0, x = var_1784_cast_fp16)[name = string("K_layer_cache_13_cast_fp16")]; tensor var_1786_begin_0 = const()[name = string("op_1786_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_1786_end_0 = const()[name = string("op_1786_end_0"), val = tensor([23, 8, 4096, 64])]; tensor var_1786_end_mask_0 = const()[name = string("op_1786_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1786_cast_fp16 = slice_by_index(begin = var_1786_begin_0, end = var_1786_end_0, end_mask = var_1786_end_mask_0, x = coreml_update_state_45)[name = string("op_1786_cast_fp16")]; tensor V_layer_cache_13_axes_0 = const()[name = string("V_layer_cache_13_axes_0"), val = tensor([0])]; tensor V_layer_cache_13_cast_fp16 = squeeze(axes = V_layer_cache_13_axes_0, x = var_1786_cast_fp16)[name = string("V_layer_cache_13_cast_fp16")]; tensor x_179_axes_0 = const()[name = string("x_179_axes_0"), val = tensor([1])]; tensor x_179_cast_fp16 = expand_dims(axes = x_179_axes_0, x = K_layer_cache_13_cast_fp16)[name = string("x_179_cast_fp16")]; tensor var_1795 = const()[name = string("op_1795"), val = tensor([1, 4, 1, 1])]; tensor x_181_cast_fp16 = tile(reps = var_1795, x = x_179_cast_fp16)[name = string("x_181_cast_fp16")]; tensor var_1799 = const()[name = string("op_1799"), val = tensor([1, -1, 4096, 64])]; tensor key_states_27_cast_fp16 = reshape(shape = var_1799, x = x_181_cast_fp16)[name = string("key_states_27_cast_fp16")]; tensor x_185_axes_0 = const()[name = string("x_185_axes_0"), val = tensor([1])]; tensor x_185_cast_fp16 = expand_dims(axes = x_185_axes_0, x = V_layer_cache_13_cast_fp16)[name = string("x_185_cast_fp16")]; tensor var_1802 = const()[name = string("op_1802"), val = tensor([1, 4, 1, 1])]; tensor x_187_cast_fp16 = tile(reps = var_1802, x = x_185_cast_fp16)[name = string("x_187_cast_fp16")]; tensor var_1806 = const()[name = string("op_1806"), val = tensor([1, -1, 4096, 64])]; tensor value_states_27_cast_fp16 = reshape(shape = var_1806, x = x_187_cast_fp16)[name = string("value_states_27_cast_fp16")]; bool var_1809_transpose_x_1 = const()[name = string("op_1809_transpose_x_1"), val = bool(false)]; bool var_1809_transpose_y_1 = const()[name = string("op_1809_transpose_y_1"), val = bool(true)]; tensor var_1809_cast_fp16 = matmul(transpose_x = var_1809_transpose_x_1, transpose_y = var_1809_transpose_y_1, x = rotated_25_cast_fp16, y = key_states_27_cast_fp16)[name = string("op_1809_cast_fp16")]; fp16 var_1810_to_fp16 = const()[name = string("op_1810_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_25_cast_fp16 = mul(x = var_1809_cast_fp16, y = var_1810_to_fp16)[name = string("attn_weights_25_cast_fp16")]; tensor x_189_cast_fp16 = add(x = attn_weights_25_cast_fp16, y = causal_mask)[name = string("x_189_cast_fp16")]; tensor reduce_max_6_axes_0 = const()[name = string("reduce_max_6_axes_0"), val = tensor([-1])]; bool reduce_max_6_keep_dims_0 = const()[name = string("reduce_max_6_keep_dims_0"), val = bool(true)]; tensor reduce_max_6_cast_fp16 = reduce_max(axes = reduce_max_6_axes_0, keep_dims = reduce_max_6_keep_dims_0, x = x_189_cast_fp16)[name = string("reduce_max_6_cast_fp16")]; tensor x_191_cast_fp16 = sub(x = x_189_cast_fp16, y = reduce_max_6_cast_fp16)[name = string("x_191_cast_fp16")]; tensor exp_x_13_cast_fp16 = exp(x = x_191_cast_fp16)[name = string("exp_x_13_cast_fp16")]; tensor var_1821_axes_0 = const()[name = string("op_1821_axes_0"), val = tensor([-1])]; bool var_1821_keep_dims_0 = const()[name = string("op_1821_keep_dims_0"), val = bool(true)]; tensor var_1821_cast_fp16 = reduce_sum(axes = var_1821_axes_0, keep_dims = var_1821_keep_dims_0, x = exp_x_13_cast_fp16)[name = string("op_1821_cast_fp16")]; tensor attn_weights_27_cast_fp16 = real_div(x = exp_x_13_cast_fp16, y = var_1821_cast_fp16)[name = string("attn_weights_27_cast_fp16")]; bool attn_output_37_transpose_x_0 = const()[name = string("attn_output_37_transpose_x_0"), val = bool(false)]; bool attn_output_37_transpose_y_0 = const()[name = string("attn_output_37_transpose_y_0"), val = bool(false)]; tensor attn_output_37_cast_fp16 = matmul(transpose_x = attn_output_37_transpose_x_0, transpose_y = attn_output_37_transpose_y_0, x = attn_weights_27_cast_fp16, y = value_states_27_cast_fp16)[name = string("attn_output_37_cast_fp16")]; tensor var_1824_perm_0 = const()[name = string("op_1824_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1826 = const()[name = string("op_1826"), val = tensor([1, 1, 2048])]; tensor var_1824_cast_fp16 = transpose(perm = var_1824_perm_0, x = attn_output_37_cast_fp16)[name = string("transpose_38")]; tensor input_89_cast_fp16 = reshape(shape = var_1826, x = var_1824_cast_fp16)[name = string("input_89_cast_fp16")]; tensor model_model_layers_6_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(469161792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(471259008))))[name = string("model_model_layers_6_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_6_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_6_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_89_cast_fp16)[name = string("linear_6_cast_fp16")]; tensor hidden_states_53_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = linear_6_cast_fp16)[name = string("hidden_states_53_cast_fp16")]; fp16 const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1832_cast_fp16 = mul(x = hidden_states_53_cast_fp16, y = const_108_promoted_to_fp16)[name = string("op_1832_cast_fp16")]; bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; tensor input_91_cast_fp16 = concat(axis = var_80, interleave = input_91_interleave_0, values = (hidden_states_53_cast_fp16, var_1832_cast_fp16))[name = string("input_91_cast_fp16")]; tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_74_to_fp16, x = input_91_cast_fp16)[name = string("normed_53_cast_fp16")]; tensor normed_55_begin_0 = const()[name = string("normed_55_begin_0"), val = tensor([0, 0, 0])]; tensor normed_55_end_0 = const()[name = string("normed_55_end_0"), val = tensor([1, 1, 2048])]; tensor normed_55_end_mask_0 = const()[name = string("normed_55_end_mask_0"), val = tensor([true, true, false])]; tensor normed_55_cast_fp16 = slice_by_index(begin = normed_55_begin_0, end = normed_55_end_0, end_mask = normed_55_end_mask_0, x = normed_53_cast_fp16)[name = string("normed_55_cast_fp16")]; tensor const_111_promoted_to_fp16 = const()[name = string("const_111_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(471267264)))]; tensor x_193_cast_fp16 = mul(x = normed_55_cast_fp16, y = const_111_promoted_to_fp16)[name = string("x_193_cast_fp16")]; tensor var_1850 = const()[name = string("op_1850"), val = tensor([0, 2, 1])]; tensor input_93_axes_0 = const()[name = string("input_93_axes_0"), val = tensor([2])]; tensor var_1851 = transpose(perm = var_1850, x = x_193_cast_fp16)[name = string("transpose_37")]; tensor input_93 = expand_dims(axes = input_93_axes_0, x = var_1851)[name = string("input_93")]; string input_95_pad_type_0 = const()[name = string("input_95_pad_type_0"), val = string("valid")]; tensor input_95_strides_0 = const()[name = string("input_95_strides_0"), val = tensor([1, 1])]; tensor input_95_pad_0 = const()[name = string("input_95_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_95_dilations_0 = const()[name = string("input_95_dilations_0"), val = tensor([1, 1])]; int32 input_95_groups_0 = const()[name = string("input_95_groups_0"), val = int32(1)]; tensor input_95 = conv(dilations = input_95_dilations_0, groups = input_95_groups_0, pad = input_95_pad_0, pad_type = input_95_pad_type_0, strides = input_95_strides_0, weight = model_model_layers_6_mlp_gate_proj_weight_palettized, x = input_93)[name = string("input_95")]; string up_states_13_pad_type_0 = const()[name = string("up_states_13_pad_type_0"), val = string("valid")]; tensor up_states_13_strides_0 = const()[name = string("up_states_13_strides_0"), val = tensor([1, 1])]; tensor up_states_13_pad_0 = const()[name = string("up_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_13_dilations_0 = const()[name = string("up_states_13_dilations_0"), val = tensor([1, 1])]; int32 up_states_13_groups_0 = const()[name = string("up_states_13_groups_0"), val = int32(1)]; tensor up_states_13 = conv(dilations = up_states_13_dilations_0, groups = up_states_13_groups_0, pad = up_states_13_pad_0, pad_type = up_states_13_pad_type_0, strides = up_states_13_strides_0, weight = model_model_layers_6_mlp_up_proj_weight_palettized, x = input_93)[name = string("up_states_13")]; tensor gate_states_13 = silu(x = input_95)[name = string("gate_states_13")]; tensor input_97 = mul(x = gate_states_13, y = up_states_13)[name = string("input_97")]; string hidden_states_55_pad_type_0 = const()[name = string("hidden_states_55_pad_type_0"), val = string("valid")]; tensor hidden_states_55_strides_0 = const()[name = string("hidden_states_55_strides_0"), val = tensor([1, 1])]; tensor hidden_states_55_pad_0 = const()[name = string("hidden_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_55_dilations_0 = const()[name = string("hidden_states_55_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_55_groups_0 = const()[name = string("hidden_states_55_groups_0"), val = int32(1)]; tensor hidden_states_55 = conv(dilations = hidden_states_55_dilations_0, groups = hidden_states_55_groups_0, pad = hidden_states_55_pad_0, pad_type = hidden_states_55_pad_type_0, strides = hidden_states_55_strides_0, weight = model_model_layers_6_mlp_down_proj_weight_palettized, x = input_97)[name = string("hidden_states_55")]; tensor var_1873_axes_0 = const()[name = string("op_1873_axes_0"), val = tensor([2])]; tensor var_1873 = squeeze(axes = var_1873_axes_0, x = hidden_states_55)[name = string("op_1873")]; tensor var_1874 = const()[name = string("op_1874"), val = tensor([0, 2, 1])]; tensor var_1875 = transpose(perm = var_1874, x = var_1873)[name = string("transpose_36")]; tensor hidden_states_57_cast_fp16 = add(x = hidden_states_53_cast_fp16, y = var_1875)[name = string("hidden_states_57_cast_fp16")]; fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1878_cast_fp16 = mul(x = hidden_states_57_cast_fp16, y = const_112_promoted_to_fp16)[name = string("op_1878_cast_fp16")]; bool input_99_interleave_0 = const()[name = string("input_99_interleave_0"), val = bool(false)]; tensor input_99_cast_fp16 = concat(axis = var_80, interleave = input_99_interleave_0, values = (hidden_states_57_cast_fp16, var_1878_cast_fp16))[name = string("input_99_cast_fp16")]; tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_74_to_fp16, x = input_99_cast_fp16)[name = string("normed_57_cast_fp16")]; tensor normed_59_begin_0 = const()[name = string("normed_59_begin_0"), val = tensor([0, 0, 0])]; tensor normed_59_end_0 = const()[name = string("normed_59_end_0"), val = tensor([1, 1, 2048])]; tensor normed_59_end_mask_0 = const()[name = string("normed_59_end_mask_0"), val = tensor([true, true, false])]; tensor normed_59_cast_fp16 = slice_by_index(begin = normed_59_begin_0, end = normed_59_end_0, end_mask = normed_59_end_mask_0, x = normed_57_cast_fp16)[name = string("normed_59_cast_fp16")]; tensor const_115_promoted_to_fp16 = const()[name = string("const_115_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(471271424)))]; tensor hidden_states_59_cast_fp16 = mul(x = normed_59_cast_fp16, y = const_115_promoted_to_fp16)[name = string("hidden_states_59_cast_fp16")]; tensor var_1892 = const()[name = string("op_1892"), val = tensor([0, 2, 1])]; tensor var_1894_axes_0 = const()[name = string("op_1894_axes_0"), val = tensor([2])]; tensor var_1893_cast_fp16 = transpose(perm = var_1892, x = hidden_states_59_cast_fp16)[name = string("transpose_35")]; tensor var_1894_cast_fp16 = expand_dims(axes = var_1894_axes_0, x = var_1893_cast_fp16)[name = string("op_1894_cast_fp16")]; string var_1901_pad_type_0 = const()[name = string("op_1901_pad_type_0"), val = string("valid")]; tensor var_1901_strides_0 = const()[name = string("op_1901_strides_0"), val = tensor([1, 1])]; tensor var_1901_pad_0 = const()[name = string("op_1901_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1901_dilations_0 = const()[name = string("op_1901_dilations_0"), val = tensor([1, 1])]; int32 var_1901_groups_0 = const()[name = string("op_1901_groups_0"), val = int32(1)]; tensor var_1901 = conv(dilations = var_1901_dilations_0, groups = var_1901_groups_0, pad = var_1901_pad_0, pad_type = var_1901_pad_type_0, strides = var_1901_strides_0, weight = model_model_layers_7_self_attn_q_proj_weight_palettized, x = var_1894_cast_fp16)[name = string("op_1901")]; tensor var_1902 = const()[name = string("op_1902"), val = tensor([1, 32, 1, 64])]; tensor var_1903 = reshape(shape = var_1902, x = var_1901)[name = string("op_1903")]; string var_1910_pad_type_0 = const()[name = string("op_1910_pad_type_0"), val = string("valid")]; tensor var_1910_strides_0 = const()[name = string("op_1910_strides_0"), val = tensor([1, 1])]; tensor var_1910_pad_0 = const()[name = string("op_1910_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1910_dilations_0 = const()[name = string("op_1910_dilations_0"), val = tensor([1, 1])]; int32 var_1910_groups_0 = const()[name = string("op_1910_groups_0"), val = int32(1)]; tensor var_1910 = conv(dilations = var_1910_dilations_0, groups = var_1910_groups_0, pad = var_1910_pad_0, pad_type = var_1910_pad_type_0, strides = var_1910_strides_0, weight = model_model_layers_7_self_attn_k_proj_weight_palettized, x = var_1894_cast_fp16)[name = string("op_1910")]; tensor var_1911 = const()[name = string("op_1911"), val = tensor([1, 8, 1, 64])]; tensor var_1912 = reshape(shape = var_1911, x = var_1910)[name = string("op_1912")]; string var_1919_pad_type_0 = const()[name = string("op_1919_pad_type_0"), val = string("valid")]; tensor var_1919_strides_0 = const()[name = string("op_1919_strides_0"), val = tensor([1, 1])]; tensor var_1919_pad_0 = const()[name = string("op_1919_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1919_dilations_0 = const()[name = string("op_1919_dilations_0"), val = tensor([1, 1])]; int32 var_1919_groups_0 = const()[name = string("op_1919_groups_0"), val = int32(1)]; tensor var_1919 = conv(dilations = var_1919_dilations_0, groups = var_1919_groups_0, pad = var_1919_pad_0, pad_type = var_1919_pad_type_0, strides = var_1919_strides_0, weight = model_model_layers_7_self_attn_v_proj_weight_palettized, x = var_1894_cast_fp16)[name = string("op_1919")]; tensor var_1920 = const()[name = string("op_1920"), val = tensor([1, 8, 1, 64])]; tensor var_1921 = reshape(shape = var_1920, x = var_1919)[name = string("op_1921")]; tensor x1_29_begin_0 = const()[name = string("x1_29_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_29_end_0 = const()[name = string("x1_29_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_29_end_mask_0 = const()[name = string("x1_29_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_29 = slice_by_index(begin = x1_29_begin_0, end = x1_29_end_0, end_mask = x1_29_end_mask_0, x = var_1903)[name = string("x1_29")]; tensor x2_29_begin_0 = const()[name = string("x2_29_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_29_end_0 = const()[name = string("x2_29_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_29_end_mask_0 = const()[name = string("x2_29_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_29 = slice_by_index(begin = x2_29_begin_0, end = x2_29_end_0, end_mask = x2_29_end_mask_0, x = var_1903)[name = string("x2_29")]; tensor var_1935_cast_fp16 = mul(x = x1_29, y = cos_3_cast_fp16)[name = string("op_1935_cast_fp16")]; tensor var_1936_cast_fp16 = mul(x = x2_29, y = sin_3_cast_fp16)[name = string("op_1936_cast_fp16")]; tensor var_1937_cast_fp16 = sub(x = var_1935_cast_fp16, y = var_1936_cast_fp16)[name = string("op_1937_cast_fp16")]; tensor var_1938_cast_fp16 = mul(x = x2_29, y = cos_3_cast_fp16)[name = string("op_1938_cast_fp16")]; tensor var_1939_cast_fp16 = mul(x = x1_29, y = sin_3_cast_fp16)[name = string("op_1939_cast_fp16")]; tensor var_1940_cast_fp16 = add(x = var_1938_cast_fp16, y = var_1939_cast_fp16)[name = string("op_1940_cast_fp16")]; bool rotated_29_interleave_0 = const()[name = string("rotated_29_interleave_0"), val = bool(false)]; tensor rotated_29_cast_fp16 = concat(axis = var_80, interleave = rotated_29_interleave_0, values = (var_1937_cast_fp16, var_1940_cast_fp16))[name = string("rotated_29_cast_fp16")]; tensor x1_31_begin_0 = const()[name = string("x1_31_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_31_end_0 = const()[name = string("x1_31_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_31_end_mask_0 = const()[name = string("x1_31_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_31 = slice_by_index(begin = x1_31_begin_0, end = x1_31_end_0, end_mask = x1_31_end_mask_0, x = var_1912)[name = string("x1_31")]; tensor x2_31_begin_0 = const()[name = string("x2_31_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_31_end_0 = const()[name = string("x2_31_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_31_end_mask_0 = const()[name = string("x2_31_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_31 = slice_by_index(begin = x2_31_begin_0, end = x2_31_end_0, end_mask = x2_31_end_mask_0, x = var_1912)[name = string("x2_31")]; tensor var_1956_cast_fp16 = mul(x = x1_31, y = cos_3_cast_fp16)[name = string("op_1956_cast_fp16")]; tensor var_1957_cast_fp16 = mul(x = x2_31, y = sin_3_cast_fp16)[name = string("op_1957_cast_fp16")]; tensor var_1958_cast_fp16 = sub(x = var_1956_cast_fp16, y = var_1957_cast_fp16)[name = string("op_1958_cast_fp16")]; tensor var_1959_cast_fp16 = mul(x = x2_31, y = cos_3_cast_fp16)[name = string("op_1959_cast_fp16")]; tensor var_1960_cast_fp16 = mul(x = x1_31, y = sin_3_cast_fp16)[name = string("op_1960_cast_fp16")]; tensor var_1961_cast_fp16 = add(x = var_1959_cast_fp16, y = var_1960_cast_fp16)[name = string("op_1961_cast_fp16")]; bool rotated_31_interleave_0 = const()[name = string("rotated_31_interleave_0"), val = bool(false)]; tensor rotated_31_cast_fp16 = concat(axis = var_80, interleave = rotated_31_interleave_0, values = (var_1958_cast_fp16, var_1961_cast_fp16))[name = string("rotated_31_cast_fp16")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; int32 concat_58_axis_0 = const()[name = string("concat_58_axis_0"), val = int32(0)]; bool concat_58_interleave_0 = const()[name = string("concat_58_interleave_0"), val = bool(false)]; tensor concat_58 = concat(axis = concat_58_axis_0, interleave = concat_58_interleave_0, values = (expand_dims_84, expand_dims_85, current_pos, expand_dims_87))[name = string("concat_58")]; tensor concat_59_values1_0 = const()[name = string("concat_59_values1_0"), val = tensor([0])]; tensor concat_59_values3_0 = const()[name = string("concat_59_values3_0"), val = tensor([0])]; int32 concat_59_axis_0 = const()[name = string("concat_59_axis_0"), val = int32(0)]; bool concat_59_interleave_0 = const()[name = string("concat_59_interleave_0"), val = bool(false)]; tensor concat_59 = concat(axis = concat_59_axis_0, interleave = concat_59_interleave_0, values = (expand_dims_88, concat_59_values1_0, var_587, concat_59_values3_0))[name = string("concat_59")]; tensor model_model_kv_cache_0_internal_tensor_assign_15_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_58, begin_mask = model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0, end = concat_59, end_mask = model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_15_stride_0, update = rotated_31_cast_fp16, x = coreml_update_state_45)[name = string("model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_78_write_state")]; tensor coreml_update_state_46 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_78")]; tensor expand_dims_90 = const()[name = string("expand_dims_90"), val = tensor([23])]; tensor expand_dims_91 = const()[name = string("expand_dims_91"), val = tensor([0])]; tensor expand_dims_93 = const()[name = string("expand_dims_93"), val = tensor([0])]; tensor expand_dims_94 = const()[name = string("expand_dims_94"), val = tensor([24])]; int32 concat_62_axis_0 = const()[name = string("concat_62_axis_0"), val = int32(0)]; bool concat_62_interleave_0 = const()[name = string("concat_62_interleave_0"), val = bool(false)]; tensor concat_62 = concat(axis = concat_62_axis_0, interleave = concat_62_interleave_0, values = (expand_dims_90, expand_dims_91, current_pos, expand_dims_93))[name = string("concat_62")]; tensor concat_63_values1_0 = const()[name = string("concat_63_values1_0"), val = tensor([0])]; tensor concat_63_values3_0 = const()[name = string("concat_63_values3_0"), val = tensor([0])]; int32 concat_63_axis_0 = const()[name = string("concat_63_axis_0"), val = int32(0)]; bool concat_63_interleave_0 = const()[name = string("concat_63_interleave_0"), val = bool(false)]; tensor concat_63 = concat(axis = concat_63_axis_0, interleave = concat_63_interleave_0, values = (expand_dims_94, concat_63_values1_0, var_587, concat_63_values3_0))[name = string("concat_63")]; tensor model_model_kv_cache_0_internal_tensor_assign_16_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_62, begin_mask = model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0, end = concat_63, end_mask = model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_16_stride_0, update = var_1921, x = coreml_update_state_46)[name = string("model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_79_write_state")]; tensor coreml_update_state_47 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_79")]; tensor var_1981_begin_0 = const()[name = string("op_1981_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_1981_end_0 = const()[name = string("op_1981_end_0"), val = tensor([8, 8, 4096, 64])]; tensor var_1981_end_mask_0 = const()[name = string("op_1981_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1981_cast_fp16 = slice_by_index(begin = var_1981_begin_0, end = var_1981_end_0, end_mask = var_1981_end_mask_0, x = coreml_update_state_47)[name = string("op_1981_cast_fp16")]; tensor K_layer_cache_15_axes_0 = const()[name = string("K_layer_cache_15_axes_0"), val = tensor([0])]; tensor K_layer_cache_15_cast_fp16 = squeeze(axes = K_layer_cache_15_axes_0, x = var_1981_cast_fp16)[name = string("K_layer_cache_15_cast_fp16")]; tensor var_1983_begin_0 = const()[name = string("op_1983_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_1983_end_0 = const()[name = string("op_1983_end_0"), val = tensor([24, 8, 4096, 64])]; tensor var_1983_end_mask_0 = const()[name = string("op_1983_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1983_cast_fp16 = slice_by_index(begin = var_1983_begin_0, end = var_1983_end_0, end_mask = var_1983_end_mask_0, x = coreml_update_state_47)[name = string("op_1983_cast_fp16")]; tensor V_layer_cache_15_axes_0 = const()[name = string("V_layer_cache_15_axes_0"), val = tensor([0])]; tensor V_layer_cache_15_cast_fp16 = squeeze(axes = V_layer_cache_15_axes_0, x = var_1983_cast_fp16)[name = string("V_layer_cache_15_cast_fp16")]; tensor x_207_axes_0 = const()[name = string("x_207_axes_0"), val = tensor([1])]; tensor x_207_cast_fp16 = expand_dims(axes = x_207_axes_0, x = K_layer_cache_15_cast_fp16)[name = string("x_207_cast_fp16")]; tensor var_1992 = const()[name = string("op_1992"), val = tensor([1, 4, 1, 1])]; tensor x_209_cast_fp16 = tile(reps = var_1992, x = x_207_cast_fp16)[name = string("x_209_cast_fp16")]; tensor var_1996 = const()[name = string("op_1996"), val = tensor([1, -1, 4096, 64])]; tensor key_states_31_cast_fp16 = reshape(shape = var_1996, x = x_209_cast_fp16)[name = string("key_states_31_cast_fp16")]; tensor x_213_axes_0 = const()[name = string("x_213_axes_0"), val = tensor([1])]; tensor x_213_cast_fp16 = expand_dims(axes = x_213_axes_0, x = V_layer_cache_15_cast_fp16)[name = string("x_213_cast_fp16")]; tensor var_1999 = const()[name = string("op_1999"), val = tensor([1, 4, 1, 1])]; tensor x_215_cast_fp16 = tile(reps = var_1999, x = x_213_cast_fp16)[name = string("x_215_cast_fp16")]; tensor var_2003 = const()[name = string("op_2003"), val = tensor([1, -1, 4096, 64])]; tensor value_states_31_cast_fp16 = reshape(shape = var_2003, x = x_215_cast_fp16)[name = string("value_states_31_cast_fp16")]; bool var_2006_transpose_x_1 = const()[name = string("op_2006_transpose_x_1"), val = bool(false)]; bool var_2006_transpose_y_1 = const()[name = string("op_2006_transpose_y_1"), val = bool(true)]; tensor var_2006_cast_fp16 = matmul(transpose_x = var_2006_transpose_x_1, transpose_y = var_2006_transpose_y_1, x = rotated_29_cast_fp16, y = key_states_31_cast_fp16)[name = string("op_2006_cast_fp16")]; fp16 var_2007_to_fp16 = const()[name = string("op_2007_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_29_cast_fp16 = mul(x = var_2006_cast_fp16, y = var_2007_to_fp16)[name = string("attn_weights_29_cast_fp16")]; tensor x_217_cast_fp16 = add(x = attn_weights_29_cast_fp16, y = causal_mask)[name = string("x_217_cast_fp16")]; tensor reduce_max_7_axes_0 = const()[name = string("reduce_max_7_axes_0"), val = tensor([-1])]; bool reduce_max_7_keep_dims_0 = const()[name = string("reduce_max_7_keep_dims_0"), val = bool(true)]; tensor reduce_max_7_cast_fp16 = reduce_max(axes = reduce_max_7_axes_0, keep_dims = reduce_max_7_keep_dims_0, x = x_217_cast_fp16)[name = string("reduce_max_7_cast_fp16")]; tensor x_219_cast_fp16 = sub(x = x_217_cast_fp16, y = reduce_max_7_cast_fp16)[name = string("x_219_cast_fp16")]; tensor exp_x_15_cast_fp16 = exp(x = x_219_cast_fp16)[name = string("exp_x_15_cast_fp16")]; tensor var_2018_axes_0 = const()[name = string("op_2018_axes_0"), val = tensor([-1])]; bool var_2018_keep_dims_0 = const()[name = string("op_2018_keep_dims_0"), val = bool(true)]; tensor var_2018_cast_fp16 = reduce_sum(axes = var_2018_axes_0, keep_dims = var_2018_keep_dims_0, x = exp_x_15_cast_fp16)[name = string("op_2018_cast_fp16")]; tensor attn_weights_31_cast_fp16 = real_div(x = exp_x_15_cast_fp16, y = var_2018_cast_fp16)[name = string("attn_weights_31_cast_fp16")]; bool attn_output_43_transpose_x_0 = const()[name = string("attn_output_43_transpose_x_0"), val = bool(false)]; bool attn_output_43_transpose_y_0 = const()[name = string("attn_output_43_transpose_y_0"), val = bool(false)]; tensor attn_output_43_cast_fp16 = matmul(transpose_x = attn_output_43_transpose_x_0, transpose_y = attn_output_43_transpose_y_0, x = attn_weights_31_cast_fp16, y = value_states_31_cast_fp16)[name = string("attn_output_43_cast_fp16")]; tensor var_2021_perm_0 = const()[name = string("op_2021_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2023 = const()[name = string("op_2023"), val = tensor([1, 1, 2048])]; tensor var_2021_cast_fp16 = transpose(perm = var_2021_perm_0, x = attn_output_43_cast_fp16)[name = string("transpose_34")]; tensor input_103_cast_fp16 = reshape(shape = var_2023, x = var_2021_cast_fp16)[name = string("input_103_cast_fp16")]; tensor model_model_layers_7_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(471275584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473372800))))[name = string("model_model_layers_7_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_7_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_7_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_103_cast_fp16)[name = string("linear_7_cast_fp16")]; tensor hidden_states_61_cast_fp16 = add(x = hidden_states_57_cast_fp16, y = linear_7_cast_fp16)[name = string("hidden_states_61_cast_fp16")]; fp16 const_124_promoted_to_fp16 = const()[name = string("const_124_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2029_cast_fp16 = mul(x = hidden_states_61_cast_fp16, y = const_124_promoted_to_fp16)[name = string("op_2029_cast_fp16")]; bool input_105_interleave_0 = const()[name = string("input_105_interleave_0"), val = bool(false)]; tensor input_105_cast_fp16 = concat(axis = var_80, interleave = input_105_interleave_0, values = (hidden_states_61_cast_fp16, var_2029_cast_fp16))[name = string("input_105_cast_fp16")]; tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_74_to_fp16, x = input_105_cast_fp16)[name = string("normed_61_cast_fp16")]; tensor normed_63_begin_0 = const()[name = string("normed_63_begin_0"), val = tensor([0, 0, 0])]; tensor normed_63_end_0 = const()[name = string("normed_63_end_0"), val = tensor([1, 1, 2048])]; tensor normed_63_end_mask_0 = const()[name = string("normed_63_end_mask_0"), val = tensor([true, true, false])]; tensor normed_63_cast_fp16 = slice_by_index(begin = normed_63_begin_0, end = normed_63_end_0, end_mask = normed_63_end_mask_0, x = normed_61_cast_fp16)[name = string("normed_63_cast_fp16")]; tensor const_127_promoted_to_fp16 = const()[name = string("const_127_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473381056)))]; tensor x_221_cast_fp16 = mul(x = normed_63_cast_fp16, y = const_127_promoted_to_fp16)[name = string("x_221_cast_fp16")]; tensor var_2047 = const()[name = string("op_2047"), val = tensor([0, 2, 1])]; tensor input_107_axes_0 = const()[name = string("input_107_axes_0"), val = tensor([2])]; tensor var_2048 = transpose(perm = var_2047, x = x_221_cast_fp16)[name = string("transpose_33")]; tensor input_107 = expand_dims(axes = input_107_axes_0, x = var_2048)[name = string("input_107")]; string input_109_pad_type_0 = const()[name = string("input_109_pad_type_0"), val = string("valid")]; tensor input_109_strides_0 = const()[name = string("input_109_strides_0"), val = tensor([1, 1])]; tensor input_109_pad_0 = const()[name = string("input_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_109_dilations_0 = const()[name = string("input_109_dilations_0"), val = tensor([1, 1])]; int32 input_109_groups_0 = const()[name = string("input_109_groups_0"), val = int32(1)]; tensor input_109 = conv(dilations = input_109_dilations_0, groups = input_109_groups_0, pad = input_109_pad_0, pad_type = input_109_pad_type_0, strides = input_109_strides_0, weight = model_model_layers_7_mlp_gate_proj_weight_palettized, x = input_107)[name = string("input_109")]; string up_states_15_pad_type_0 = const()[name = string("up_states_15_pad_type_0"), val = string("valid")]; tensor up_states_15_strides_0 = const()[name = string("up_states_15_strides_0"), val = tensor([1, 1])]; tensor up_states_15_pad_0 = const()[name = string("up_states_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_15_dilations_0 = const()[name = string("up_states_15_dilations_0"), val = tensor([1, 1])]; int32 up_states_15_groups_0 = const()[name = string("up_states_15_groups_0"), val = int32(1)]; tensor up_states_15 = conv(dilations = up_states_15_dilations_0, groups = up_states_15_groups_0, pad = up_states_15_pad_0, pad_type = up_states_15_pad_type_0, strides = up_states_15_strides_0, weight = model_model_layers_7_mlp_up_proj_weight_palettized, x = input_107)[name = string("up_states_15")]; tensor gate_states_15 = silu(x = input_109)[name = string("gate_states_15")]; tensor input_111 = mul(x = gate_states_15, y = up_states_15)[name = string("input_111")]; string hidden_states_63_pad_type_0 = const()[name = string("hidden_states_63_pad_type_0"), val = string("valid")]; tensor hidden_states_63_strides_0 = const()[name = string("hidden_states_63_strides_0"), val = tensor([1, 1])]; tensor hidden_states_63_pad_0 = const()[name = string("hidden_states_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_63_dilations_0 = const()[name = string("hidden_states_63_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_63_groups_0 = const()[name = string("hidden_states_63_groups_0"), val = int32(1)]; tensor hidden_states_63 = conv(dilations = hidden_states_63_dilations_0, groups = hidden_states_63_groups_0, pad = hidden_states_63_pad_0, pad_type = hidden_states_63_pad_type_0, strides = hidden_states_63_strides_0, weight = model_model_layers_7_mlp_down_proj_weight_palettized, x = input_111)[name = string("hidden_states_63")]; tensor var_2070_axes_0 = const()[name = string("op_2070_axes_0"), val = tensor([2])]; tensor var_2070 = squeeze(axes = var_2070_axes_0, x = hidden_states_63)[name = string("op_2070")]; tensor var_2071 = const()[name = string("op_2071"), val = tensor([0, 2, 1])]; tensor var_2072 = transpose(perm = var_2071, x = var_2070)[name = string("transpose_32")]; tensor hidden_states_65_cast_fp16 = add(x = hidden_states_61_cast_fp16, y = var_2072)[name = string("hidden_states_65_cast_fp16")]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2075_cast_fp16 = mul(x = hidden_states_65_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_2075_cast_fp16")]; bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; tensor input_113_cast_fp16 = concat(axis = var_80, interleave = input_113_interleave_0, values = (hidden_states_65_cast_fp16, var_2075_cast_fp16))[name = string("input_113_cast_fp16")]; tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_74_to_fp16, x = input_113_cast_fp16)[name = string("normed_65_cast_fp16")]; tensor normed_67_begin_0 = const()[name = string("normed_67_begin_0"), val = tensor([0, 0, 0])]; tensor normed_67_end_0 = const()[name = string("normed_67_end_0"), val = tensor([1, 1, 2048])]; tensor normed_67_end_mask_0 = const()[name = string("normed_67_end_mask_0"), val = tensor([true, true, false])]; tensor normed_67_cast_fp16 = slice_by_index(begin = normed_67_begin_0, end = normed_67_end_0, end_mask = normed_67_end_mask_0, x = normed_65_cast_fp16)[name = string("normed_67_cast_fp16")]; tensor const_131_promoted_to_fp16 = const()[name = string("const_131_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473385216)))]; tensor hidden_states_67_cast_fp16 = mul(x = normed_67_cast_fp16, y = const_131_promoted_to_fp16)[name = string("hidden_states_67_cast_fp16")]; tensor var_2089 = const()[name = string("op_2089"), val = tensor([0, 2, 1])]; tensor var_2091_axes_0 = const()[name = string("op_2091_axes_0"), val = tensor([2])]; tensor var_2090_cast_fp16 = transpose(perm = var_2089, x = hidden_states_67_cast_fp16)[name = string("transpose_31")]; tensor var_2091_cast_fp16 = expand_dims(axes = var_2091_axes_0, x = var_2090_cast_fp16)[name = string("op_2091_cast_fp16")]; string var_2098_pad_type_0 = const()[name = string("op_2098_pad_type_0"), val = string("valid")]; tensor var_2098_strides_0 = const()[name = string("op_2098_strides_0"), val = tensor([1, 1])]; tensor var_2098_pad_0 = const()[name = string("op_2098_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2098_dilations_0 = const()[name = string("op_2098_dilations_0"), val = tensor([1, 1])]; int32 var_2098_groups_0 = const()[name = string("op_2098_groups_0"), val = int32(1)]; tensor var_2098 = conv(dilations = var_2098_dilations_0, groups = var_2098_groups_0, pad = var_2098_pad_0, pad_type = var_2098_pad_type_0, strides = var_2098_strides_0, weight = model_model_layers_8_self_attn_q_proj_weight_palettized, x = var_2091_cast_fp16)[name = string("op_2098")]; tensor var_2099 = const()[name = string("op_2099"), val = tensor([1, 32, 1, 64])]; tensor var_2100 = reshape(shape = var_2099, x = var_2098)[name = string("op_2100")]; string var_2107_pad_type_0 = const()[name = string("op_2107_pad_type_0"), val = string("valid")]; tensor var_2107_strides_0 = const()[name = string("op_2107_strides_0"), val = tensor([1, 1])]; tensor var_2107_pad_0 = const()[name = string("op_2107_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2107_dilations_0 = const()[name = string("op_2107_dilations_0"), val = tensor([1, 1])]; int32 var_2107_groups_0 = const()[name = string("op_2107_groups_0"), val = int32(1)]; tensor var_2107 = conv(dilations = var_2107_dilations_0, groups = var_2107_groups_0, pad = var_2107_pad_0, pad_type = var_2107_pad_type_0, strides = var_2107_strides_0, weight = model_model_layers_8_self_attn_k_proj_weight_palettized, x = var_2091_cast_fp16)[name = string("op_2107")]; tensor var_2108 = const()[name = string("op_2108"), val = tensor([1, 8, 1, 64])]; tensor var_2109 = reshape(shape = var_2108, x = var_2107)[name = string("op_2109")]; string var_2116_pad_type_0 = const()[name = string("op_2116_pad_type_0"), val = string("valid")]; tensor var_2116_strides_0 = const()[name = string("op_2116_strides_0"), val = tensor([1, 1])]; tensor var_2116_pad_0 = const()[name = string("op_2116_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2116_dilations_0 = const()[name = string("op_2116_dilations_0"), val = tensor([1, 1])]; int32 var_2116_groups_0 = const()[name = string("op_2116_groups_0"), val = int32(1)]; tensor var_2116 = conv(dilations = var_2116_dilations_0, groups = var_2116_groups_0, pad = var_2116_pad_0, pad_type = var_2116_pad_type_0, strides = var_2116_strides_0, weight = model_model_layers_8_self_attn_v_proj_weight_palettized, x = var_2091_cast_fp16)[name = string("op_2116")]; tensor var_2117 = const()[name = string("op_2117"), val = tensor([1, 8, 1, 64])]; tensor var_2118 = reshape(shape = var_2117, x = var_2116)[name = string("op_2118")]; tensor x1_33_begin_0 = const()[name = string("x1_33_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_33_end_0 = const()[name = string("x1_33_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_33_end_mask_0 = const()[name = string("x1_33_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_33 = slice_by_index(begin = x1_33_begin_0, end = x1_33_end_0, end_mask = x1_33_end_mask_0, x = var_2100)[name = string("x1_33")]; tensor x2_33_begin_0 = const()[name = string("x2_33_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_33_end_0 = const()[name = string("x2_33_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_33_end_mask_0 = const()[name = string("x2_33_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_33 = slice_by_index(begin = x2_33_begin_0, end = x2_33_end_0, end_mask = x2_33_end_mask_0, x = var_2100)[name = string("x2_33")]; tensor var_2132_cast_fp16 = mul(x = x1_33, y = cos_3_cast_fp16)[name = string("op_2132_cast_fp16")]; tensor var_2133_cast_fp16 = mul(x = x2_33, y = sin_3_cast_fp16)[name = string("op_2133_cast_fp16")]; tensor var_2134_cast_fp16 = sub(x = var_2132_cast_fp16, y = var_2133_cast_fp16)[name = string("op_2134_cast_fp16")]; tensor var_2135_cast_fp16 = mul(x = x2_33, y = cos_3_cast_fp16)[name = string("op_2135_cast_fp16")]; tensor var_2136_cast_fp16 = mul(x = x1_33, y = sin_3_cast_fp16)[name = string("op_2136_cast_fp16")]; tensor var_2137_cast_fp16 = add(x = var_2135_cast_fp16, y = var_2136_cast_fp16)[name = string("op_2137_cast_fp16")]; bool rotated_33_interleave_0 = const()[name = string("rotated_33_interleave_0"), val = bool(false)]; tensor rotated_33_cast_fp16 = concat(axis = var_80, interleave = rotated_33_interleave_0, values = (var_2134_cast_fp16, var_2137_cast_fp16))[name = string("rotated_33_cast_fp16")]; tensor x1_35_begin_0 = const()[name = string("x1_35_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_35_end_0 = const()[name = string("x1_35_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_35_end_mask_0 = const()[name = string("x1_35_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_35 = slice_by_index(begin = x1_35_begin_0, end = x1_35_end_0, end_mask = x1_35_end_mask_0, x = var_2109)[name = string("x1_35")]; tensor x2_35_begin_0 = const()[name = string("x2_35_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_35_end_0 = const()[name = string("x2_35_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_35_end_mask_0 = const()[name = string("x2_35_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_35 = slice_by_index(begin = x2_35_begin_0, end = x2_35_end_0, end_mask = x2_35_end_mask_0, x = var_2109)[name = string("x2_35")]; tensor var_2153_cast_fp16 = mul(x = x1_35, y = cos_3_cast_fp16)[name = string("op_2153_cast_fp16")]; tensor var_2154_cast_fp16 = mul(x = x2_35, y = sin_3_cast_fp16)[name = string("op_2154_cast_fp16")]; tensor var_2155_cast_fp16 = sub(x = var_2153_cast_fp16, y = var_2154_cast_fp16)[name = string("op_2155_cast_fp16")]; tensor var_2156_cast_fp16 = mul(x = x2_35, y = cos_3_cast_fp16)[name = string("op_2156_cast_fp16")]; tensor var_2157_cast_fp16 = mul(x = x1_35, y = sin_3_cast_fp16)[name = string("op_2157_cast_fp16")]; tensor var_2158_cast_fp16 = add(x = var_2156_cast_fp16, y = var_2157_cast_fp16)[name = string("op_2158_cast_fp16")]; bool rotated_35_interleave_0 = const()[name = string("rotated_35_interleave_0"), val = bool(false)]; tensor rotated_35_cast_fp16 = concat(axis = var_80, interleave = rotated_35_interleave_0, values = (var_2155_cast_fp16, var_2158_cast_fp16))[name = string("rotated_35_cast_fp16")]; tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_96, expand_dims_97, current_pos, expand_dims_99))[name = string("concat_66")]; tensor concat_67_values1_0 = const()[name = string("concat_67_values1_0"), val = tensor([0])]; tensor concat_67_values3_0 = const()[name = string("concat_67_values3_0"), val = tensor([0])]; int32 concat_67_axis_0 = const()[name = string("concat_67_axis_0"), val = int32(0)]; bool concat_67_interleave_0 = const()[name = string("concat_67_interleave_0"), val = bool(false)]; tensor concat_67 = concat(axis = concat_67_axis_0, interleave = concat_67_interleave_0, values = (expand_dims_100, concat_67_values1_0, var_587, concat_67_values3_0))[name = string("concat_67")]; tensor model_model_kv_cache_0_internal_tensor_assign_17_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_66, begin_mask = model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0, end = concat_67, end_mask = model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_17_stride_0, update = rotated_35_cast_fp16, x = coreml_update_state_47)[name = string("model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_80_write_state")]; tensor coreml_update_state_48 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_80")]; tensor expand_dims_102 = const()[name = string("expand_dims_102"), val = tensor([24])]; tensor expand_dims_103 = const()[name = string("expand_dims_103"), val = tensor([0])]; tensor expand_dims_105 = const()[name = string("expand_dims_105"), val = tensor([0])]; tensor expand_dims_106 = const()[name = string("expand_dims_106"), val = tensor([25])]; int32 concat_70_axis_0 = const()[name = string("concat_70_axis_0"), val = int32(0)]; bool concat_70_interleave_0 = const()[name = string("concat_70_interleave_0"), val = bool(false)]; tensor concat_70 = concat(axis = concat_70_axis_0, interleave = concat_70_interleave_0, values = (expand_dims_102, expand_dims_103, current_pos, expand_dims_105))[name = string("concat_70")]; tensor concat_71_values1_0 = const()[name = string("concat_71_values1_0"), val = tensor([0])]; tensor concat_71_values3_0 = const()[name = string("concat_71_values3_0"), val = tensor([0])]; int32 concat_71_axis_0 = const()[name = string("concat_71_axis_0"), val = int32(0)]; bool concat_71_interleave_0 = const()[name = string("concat_71_interleave_0"), val = bool(false)]; tensor concat_71 = concat(axis = concat_71_axis_0, interleave = concat_71_interleave_0, values = (expand_dims_106, concat_71_values1_0, var_587, concat_71_values3_0))[name = string("concat_71")]; tensor model_model_kv_cache_0_internal_tensor_assign_18_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_70, begin_mask = model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0, end = concat_71, end_mask = model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_18_stride_0, update = var_2118, x = coreml_update_state_48)[name = string("model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_81_write_state")]; tensor coreml_update_state_49 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_81")]; tensor var_2178_begin_0 = const()[name = string("op_2178_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_2178_end_0 = const()[name = string("op_2178_end_0"), val = tensor([9, 8, 4096, 64])]; tensor var_2178_end_mask_0 = const()[name = string("op_2178_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2178_cast_fp16 = slice_by_index(begin = var_2178_begin_0, end = var_2178_end_0, end_mask = var_2178_end_mask_0, x = coreml_update_state_49)[name = string("op_2178_cast_fp16")]; tensor K_layer_cache_17_axes_0 = const()[name = string("K_layer_cache_17_axes_0"), val = tensor([0])]; tensor K_layer_cache_17_cast_fp16 = squeeze(axes = K_layer_cache_17_axes_0, x = var_2178_cast_fp16)[name = string("K_layer_cache_17_cast_fp16")]; tensor var_2180_begin_0 = const()[name = string("op_2180_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_2180_end_0 = const()[name = string("op_2180_end_0"), val = tensor([25, 8, 4096, 64])]; tensor var_2180_end_mask_0 = const()[name = string("op_2180_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2180_cast_fp16 = slice_by_index(begin = var_2180_begin_0, end = var_2180_end_0, end_mask = var_2180_end_mask_0, x = coreml_update_state_49)[name = string("op_2180_cast_fp16")]; tensor V_layer_cache_17_axes_0 = const()[name = string("V_layer_cache_17_axes_0"), val = tensor([0])]; tensor V_layer_cache_17_cast_fp16 = squeeze(axes = V_layer_cache_17_axes_0, x = var_2180_cast_fp16)[name = string("V_layer_cache_17_cast_fp16")]; tensor x_235_axes_0 = const()[name = string("x_235_axes_0"), val = tensor([1])]; tensor x_235_cast_fp16 = expand_dims(axes = x_235_axes_0, x = K_layer_cache_17_cast_fp16)[name = string("x_235_cast_fp16")]; tensor var_2189 = const()[name = string("op_2189"), val = tensor([1, 4, 1, 1])]; tensor x_237_cast_fp16 = tile(reps = var_2189, x = x_235_cast_fp16)[name = string("x_237_cast_fp16")]; tensor var_2193 = const()[name = string("op_2193"), val = tensor([1, -1, 4096, 64])]; tensor key_states_35_cast_fp16 = reshape(shape = var_2193, x = x_237_cast_fp16)[name = string("key_states_35_cast_fp16")]; tensor x_241_axes_0 = const()[name = string("x_241_axes_0"), val = tensor([1])]; tensor x_241_cast_fp16 = expand_dims(axes = x_241_axes_0, x = V_layer_cache_17_cast_fp16)[name = string("x_241_cast_fp16")]; tensor var_2196 = const()[name = string("op_2196"), val = tensor([1, 4, 1, 1])]; tensor x_243_cast_fp16 = tile(reps = var_2196, x = x_241_cast_fp16)[name = string("x_243_cast_fp16")]; tensor var_2200 = const()[name = string("op_2200"), val = tensor([1, -1, 4096, 64])]; tensor value_states_35_cast_fp16 = reshape(shape = var_2200, x = x_243_cast_fp16)[name = string("value_states_35_cast_fp16")]; bool var_2203_transpose_x_1 = const()[name = string("op_2203_transpose_x_1"), val = bool(false)]; bool var_2203_transpose_y_1 = const()[name = string("op_2203_transpose_y_1"), val = bool(true)]; tensor var_2203_cast_fp16 = matmul(transpose_x = var_2203_transpose_x_1, transpose_y = var_2203_transpose_y_1, x = rotated_33_cast_fp16, y = key_states_35_cast_fp16)[name = string("op_2203_cast_fp16")]; fp16 var_2204_to_fp16 = const()[name = string("op_2204_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_33_cast_fp16 = mul(x = var_2203_cast_fp16, y = var_2204_to_fp16)[name = string("attn_weights_33_cast_fp16")]; tensor x_245_cast_fp16 = add(x = attn_weights_33_cast_fp16, y = causal_mask)[name = string("x_245_cast_fp16")]; tensor reduce_max_8_axes_0 = const()[name = string("reduce_max_8_axes_0"), val = tensor([-1])]; bool reduce_max_8_keep_dims_0 = const()[name = string("reduce_max_8_keep_dims_0"), val = bool(true)]; tensor reduce_max_8_cast_fp16 = reduce_max(axes = reduce_max_8_axes_0, keep_dims = reduce_max_8_keep_dims_0, x = x_245_cast_fp16)[name = string("reduce_max_8_cast_fp16")]; tensor x_247_cast_fp16 = sub(x = x_245_cast_fp16, y = reduce_max_8_cast_fp16)[name = string("x_247_cast_fp16")]; tensor exp_x_17_cast_fp16 = exp(x = x_247_cast_fp16)[name = string("exp_x_17_cast_fp16")]; tensor var_2215_axes_0 = const()[name = string("op_2215_axes_0"), val = tensor([-1])]; bool var_2215_keep_dims_0 = const()[name = string("op_2215_keep_dims_0"), val = bool(true)]; tensor var_2215_cast_fp16 = reduce_sum(axes = var_2215_axes_0, keep_dims = var_2215_keep_dims_0, x = exp_x_17_cast_fp16)[name = string("op_2215_cast_fp16")]; tensor attn_weights_35_cast_fp16 = real_div(x = exp_x_17_cast_fp16, y = var_2215_cast_fp16)[name = string("attn_weights_35_cast_fp16")]; bool attn_output_49_transpose_x_0 = const()[name = string("attn_output_49_transpose_x_0"), val = bool(false)]; bool attn_output_49_transpose_y_0 = const()[name = string("attn_output_49_transpose_y_0"), val = bool(false)]; tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_0, transpose_y = attn_output_49_transpose_y_0, x = attn_weights_35_cast_fp16, y = value_states_35_cast_fp16)[name = string("attn_output_49_cast_fp16")]; tensor var_2218_perm_0 = const()[name = string("op_2218_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2220 = const()[name = string("op_2220"), val = tensor([1, 1, 2048])]; tensor var_2218_cast_fp16 = transpose(perm = var_2218_perm_0, x = attn_output_49_cast_fp16)[name = string("transpose_30")]; tensor input_117_cast_fp16 = reshape(shape = var_2220, x = var_2218_cast_fp16)[name = string("input_117_cast_fp16")]; tensor model_model_layers_8_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473389376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(475486592))))[name = string("model_model_layers_8_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_8_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_8_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_117_cast_fp16)[name = string("linear_8_cast_fp16")]; tensor hidden_states_69_cast_fp16 = add(x = hidden_states_65_cast_fp16, y = linear_8_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; fp16 const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2226_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_140_promoted_to_fp16)[name = string("op_2226_cast_fp16")]; bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; tensor input_119_cast_fp16 = concat(axis = var_80, interleave = input_119_interleave_0, values = (hidden_states_69_cast_fp16, var_2226_cast_fp16))[name = string("input_119_cast_fp16")]; tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_74_to_fp16, x = input_119_cast_fp16)[name = string("normed_69_cast_fp16")]; tensor normed_71_begin_0 = const()[name = string("normed_71_begin_0"), val = tensor([0, 0, 0])]; tensor normed_71_end_0 = const()[name = string("normed_71_end_0"), val = tensor([1, 1, 2048])]; tensor normed_71_end_mask_0 = const()[name = string("normed_71_end_mask_0"), val = tensor([true, true, false])]; tensor normed_71_cast_fp16 = slice_by_index(begin = normed_71_begin_0, end = normed_71_end_0, end_mask = normed_71_end_mask_0, x = normed_69_cast_fp16)[name = string("normed_71_cast_fp16")]; tensor const_143_promoted_to_fp16 = const()[name = string("const_143_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(475494848)))]; tensor x_249_cast_fp16 = mul(x = normed_71_cast_fp16, y = const_143_promoted_to_fp16)[name = string("x_249_cast_fp16")]; tensor var_2244 = const()[name = string("op_2244"), val = tensor([0, 2, 1])]; tensor input_121_axes_0 = const()[name = string("input_121_axes_0"), val = tensor([2])]; tensor var_2245 = transpose(perm = var_2244, x = x_249_cast_fp16)[name = string("transpose_29")]; tensor input_121 = expand_dims(axes = input_121_axes_0, x = var_2245)[name = string("input_121")]; string input_123_pad_type_0 = const()[name = string("input_123_pad_type_0"), val = string("valid")]; tensor input_123_strides_0 = const()[name = string("input_123_strides_0"), val = tensor([1, 1])]; tensor input_123_pad_0 = const()[name = string("input_123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_123_dilations_0 = const()[name = string("input_123_dilations_0"), val = tensor([1, 1])]; int32 input_123_groups_0 = const()[name = string("input_123_groups_0"), val = int32(1)]; tensor input_123 = conv(dilations = input_123_dilations_0, groups = input_123_groups_0, pad = input_123_pad_0, pad_type = input_123_pad_type_0, strides = input_123_strides_0, weight = model_model_layers_8_mlp_gate_proj_weight_palettized, x = input_121)[name = string("input_123")]; string up_states_17_pad_type_0 = const()[name = string("up_states_17_pad_type_0"), val = string("valid")]; tensor up_states_17_strides_0 = const()[name = string("up_states_17_strides_0"), val = tensor([1, 1])]; tensor up_states_17_pad_0 = const()[name = string("up_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_17_dilations_0 = const()[name = string("up_states_17_dilations_0"), val = tensor([1, 1])]; int32 up_states_17_groups_0 = const()[name = string("up_states_17_groups_0"), val = int32(1)]; tensor up_states_17 = conv(dilations = up_states_17_dilations_0, groups = up_states_17_groups_0, pad = up_states_17_pad_0, pad_type = up_states_17_pad_type_0, strides = up_states_17_strides_0, weight = model_model_layers_8_mlp_up_proj_weight_palettized, x = input_121)[name = string("up_states_17")]; tensor gate_states_17 = silu(x = input_123)[name = string("gate_states_17")]; tensor input_125 = mul(x = gate_states_17, y = up_states_17)[name = string("input_125")]; string hidden_states_71_pad_type_0 = const()[name = string("hidden_states_71_pad_type_0"), val = string("valid")]; tensor hidden_states_71_strides_0 = const()[name = string("hidden_states_71_strides_0"), val = tensor([1, 1])]; tensor hidden_states_71_pad_0 = const()[name = string("hidden_states_71_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_71_dilations_0 = const()[name = string("hidden_states_71_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_71_groups_0 = const()[name = string("hidden_states_71_groups_0"), val = int32(1)]; tensor hidden_states_71 = conv(dilations = hidden_states_71_dilations_0, groups = hidden_states_71_groups_0, pad = hidden_states_71_pad_0, pad_type = hidden_states_71_pad_type_0, strides = hidden_states_71_strides_0, weight = model_model_layers_8_mlp_down_proj_weight_palettized, x = input_125)[name = string("hidden_states_71")]; tensor var_2267_axes_0 = const()[name = string("op_2267_axes_0"), val = tensor([2])]; tensor var_2267 = squeeze(axes = var_2267_axes_0, x = hidden_states_71)[name = string("op_2267")]; tensor var_2268 = const()[name = string("op_2268"), val = tensor([0, 2, 1])]; tensor var_2269 = transpose(perm = var_2268, x = var_2267)[name = string("transpose_28")]; tensor hidden_states_73_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = var_2269)[name = string("hidden_states_73_cast_fp16")]; fp16 const_144_promoted_to_fp16 = const()[name = string("const_144_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2272_cast_fp16 = mul(x = hidden_states_73_cast_fp16, y = const_144_promoted_to_fp16)[name = string("op_2272_cast_fp16")]; bool input_127_interleave_0 = const()[name = string("input_127_interleave_0"), val = bool(false)]; tensor input_127_cast_fp16 = concat(axis = var_80, interleave = input_127_interleave_0, values = (hidden_states_73_cast_fp16, var_2272_cast_fp16))[name = string("input_127_cast_fp16")]; tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_74_to_fp16, x = input_127_cast_fp16)[name = string("normed_73_cast_fp16")]; tensor normed_75_begin_0 = const()[name = string("normed_75_begin_0"), val = tensor([0, 0, 0])]; tensor normed_75_end_0 = const()[name = string("normed_75_end_0"), val = tensor([1, 1, 2048])]; tensor normed_75_end_mask_0 = const()[name = string("normed_75_end_mask_0"), val = tensor([true, true, false])]; tensor normed_75_cast_fp16 = slice_by_index(begin = normed_75_begin_0, end = normed_75_end_0, end_mask = normed_75_end_mask_0, x = normed_73_cast_fp16)[name = string("normed_75_cast_fp16")]; tensor const_147_promoted_to_fp16 = const()[name = string("const_147_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(475499008)))]; tensor hidden_states_75_cast_fp16 = mul(x = normed_75_cast_fp16, y = const_147_promoted_to_fp16)[name = string("hidden_states_75_cast_fp16")]; tensor var_2286 = const()[name = string("op_2286"), val = tensor([0, 2, 1])]; tensor var_2288_axes_0 = const()[name = string("op_2288_axes_0"), val = tensor([2])]; tensor var_2287_cast_fp16 = transpose(perm = var_2286, x = hidden_states_75_cast_fp16)[name = string("transpose_27")]; tensor var_2288_cast_fp16 = expand_dims(axes = var_2288_axes_0, x = var_2287_cast_fp16)[name = string("op_2288_cast_fp16")]; string var_2295_pad_type_0 = const()[name = string("op_2295_pad_type_0"), val = string("valid")]; tensor var_2295_strides_0 = const()[name = string("op_2295_strides_0"), val = tensor([1, 1])]; tensor var_2295_pad_0 = const()[name = string("op_2295_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2295_dilations_0 = const()[name = string("op_2295_dilations_0"), val = tensor([1, 1])]; int32 var_2295_groups_0 = const()[name = string("op_2295_groups_0"), val = int32(1)]; tensor var_2295 = conv(dilations = var_2295_dilations_0, groups = var_2295_groups_0, pad = var_2295_pad_0, pad_type = var_2295_pad_type_0, strides = var_2295_strides_0, weight = model_model_layers_9_self_attn_q_proj_weight_palettized, x = var_2288_cast_fp16)[name = string("op_2295")]; tensor var_2296 = const()[name = string("op_2296"), val = tensor([1, 32, 1, 64])]; tensor var_2297 = reshape(shape = var_2296, x = var_2295)[name = string("op_2297")]; string var_2304_pad_type_0 = const()[name = string("op_2304_pad_type_0"), val = string("valid")]; tensor var_2304_strides_0 = const()[name = string("op_2304_strides_0"), val = tensor([1, 1])]; tensor var_2304_pad_0 = const()[name = string("op_2304_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2304_dilations_0 = const()[name = string("op_2304_dilations_0"), val = tensor([1, 1])]; int32 var_2304_groups_0 = const()[name = string("op_2304_groups_0"), val = int32(1)]; tensor var_2304 = conv(dilations = var_2304_dilations_0, groups = var_2304_groups_0, pad = var_2304_pad_0, pad_type = var_2304_pad_type_0, strides = var_2304_strides_0, weight = model_model_layers_9_self_attn_k_proj_weight_palettized, x = var_2288_cast_fp16)[name = string("op_2304")]; tensor var_2305 = const()[name = string("op_2305"), val = tensor([1, 8, 1, 64])]; tensor var_2306 = reshape(shape = var_2305, x = var_2304)[name = string("op_2306")]; string var_2313_pad_type_0 = const()[name = string("op_2313_pad_type_0"), val = string("valid")]; tensor var_2313_strides_0 = const()[name = string("op_2313_strides_0"), val = tensor([1, 1])]; tensor var_2313_pad_0 = const()[name = string("op_2313_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2313_dilations_0 = const()[name = string("op_2313_dilations_0"), val = tensor([1, 1])]; int32 var_2313_groups_0 = const()[name = string("op_2313_groups_0"), val = int32(1)]; tensor var_2313 = conv(dilations = var_2313_dilations_0, groups = var_2313_groups_0, pad = var_2313_pad_0, pad_type = var_2313_pad_type_0, strides = var_2313_strides_0, weight = model_model_layers_9_self_attn_v_proj_weight_palettized, x = var_2288_cast_fp16)[name = string("op_2313")]; tensor var_2314 = const()[name = string("op_2314"), val = tensor([1, 8, 1, 64])]; tensor var_2315 = reshape(shape = var_2314, x = var_2313)[name = string("op_2315")]; tensor x1_37_begin_0 = const()[name = string("x1_37_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_37_end_0 = const()[name = string("x1_37_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_37_end_mask_0 = const()[name = string("x1_37_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_37 = slice_by_index(begin = x1_37_begin_0, end = x1_37_end_0, end_mask = x1_37_end_mask_0, x = var_2297)[name = string("x1_37")]; tensor x2_37_begin_0 = const()[name = string("x2_37_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_37_end_0 = const()[name = string("x2_37_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_37_end_mask_0 = const()[name = string("x2_37_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_37 = slice_by_index(begin = x2_37_begin_0, end = x2_37_end_0, end_mask = x2_37_end_mask_0, x = var_2297)[name = string("x2_37")]; tensor var_2329_cast_fp16 = mul(x = x1_37, y = cos_3_cast_fp16)[name = string("op_2329_cast_fp16")]; tensor var_2330_cast_fp16 = mul(x = x2_37, y = sin_3_cast_fp16)[name = string("op_2330_cast_fp16")]; tensor var_2331_cast_fp16 = sub(x = var_2329_cast_fp16, y = var_2330_cast_fp16)[name = string("op_2331_cast_fp16")]; tensor var_2332_cast_fp16 = mul(x = x2_37, y = cos_3_cast_fp16)[name = string("op_2332_cast_fp16")]; tensor var_2333_cast_fp16 = mul(x = x1_37, y = sin_3_cast_fp16)[name = string("op_2333_cast_fp16")]; tensor var_2334_cast_fp16 = add(x = var_2332_cast_fp16, y = var_2333_cast_fp16)[name = string("op_2334_cast_fp16")]; bool rotated_37_interleave_0 = const()[name = string("rotated_37_interleave_0"), val = bool(false)]; tensor rotated_37_cast_fp16 = concat(axis = var_80, interleave = rotated_37_interleave_0, values = (var_2331_cast_fp16, var_2334_cast_fp16))[name = string("rotated_37_cast_fp16")]; tensor x1_39_begin_0 = const()[name = string("x1_39_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_39_end_0 = const()[name = string("x1_39_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_39_end_mask_0 = const()[name = string("x1_39_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_39 = slice_by_index(begin = x1_39_begin_0, end = x1_39_end_0, end_mask = x1_39_end_mask_0, x = var_2306)[name = string("x1_39")]; tensor x2_39_begin_0 = const()[name = string("x2_39_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_39_end_0 = const()[name = string("x2_39_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_39_end_mask_0 = const()[name = string("x2_39_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_39 = slice_by_index(begin = x2_39_begin_0, end = x2_39_end_0, end_mask = x2_39_end_mask_0, x = var_2306)[name = string("x2_39")]; tensor var_2350_cast_fp16 = mul(x = x1_39, y = cos_3_cast_fp16)[name = string("op_2350_cast_fp16")]; tensor var_2351_cast_fp16 = mul(x = x2_39, y = sin_3_cast_fp16)[name = string("op_2351_cast_fp16")]; tensor var_2352_cast_fp16 = sub(x = var_2350_cast_fp16, y = var_2351_cast_fp16)[name = string("op_2352_cast_fp16")]; tensor var_2353_cast_fp16 = mul(x = x2_39, y = cos_3_cast_fp16)[name = string("op_2353_cast_fp16")]; tensor var_2354_cast_fp16 = mul(x = x1_39, y = sin_3_cast_fp16)[name = string("op_2354_cast_fp16")]; tensor var_2355_cast_fp16 = add(x = var_2353_cast_fp16, y = var_2354_cast_fp16)[name = string("op_2355_cast_fp16")]; bool rotated_39_interleave_0 = const()[name = string("rotated_39_interleave_0"), val = bool(false)]; tensor rotated_39_cast_fp16 = concat(axis = var_80, interleave = rotated_39_interleave_0, values = (var_2352_cast_fp16, var_2355_cast_fp16))[name = string("rotated_39_cast_fp16")]; tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; int32 concat_74_axis_0 = const()[name = string("concat_74_axis_0"), val = int32(0)]; bool concat_74_interleave_0 = const()[name = string("concat_74_interleave_0"), val = bool(false)]; tensor concat_74 = concat(axis = concat_74_axis_0, interleave = concat_74_interleave_0, values = (expand_dims_108, expand_dims_109, current_pos, expand_dims_111))[name = string("concat_74")]; tensor concat_75_values1_0 = const()[name = string("concat_75_values1_0"), val = tensor([0])]; tensor concat_75_values3_0 = const()[name = string("concat_75_values3_0"), val = tensor([0])]; int32 concat_75_axis_0 = const()[name = string("concat_75_axis_0"), val = int32(0)]; bool concat_75_interleave_0 = const()[name = string("concat_75_interleave_0"), val = bool(false)]; tensor concat_75 = concat(axis = concat_75_axis_0, interleave = concat_75_interleave_0, values = (expand_dims_112, concat_75_values1_0, var_587, concat_75_values3_0))[name = string("concat_75")]; tensor model_model_kv_cache_0_internal_tensor_assign_19_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_74, begin_mask = model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0, end = concat_75, end_mask = model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_19_stride_0, update = rotated_39_cast_fp16, x = coreml_update_state_49)[name = string("model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_82_write_state")]; tensor coreml_update_state_50 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_82")]; tensor expand_dims_114 = const()[name = string("expand_dims_114"), val = tensor([25])]; tensor expand_dims_115 = const()[name = string("expand_dims_115"), val = tensor([0])]; tensor expand_dims_117 = const()[name = string("expand_dims_117"), val = tensor([0])]; tensor expand_dims_118 = const()[name = string("expand_dims_118"), val = tensor([26])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_114, expand_dims_115, current_pos, expand_dims_117))[name = string("concat_78")]; tensor concat_79_values1_0 = const()[name = string("concat_79_values1_0"), val = tensor([0])]; tensor concat_79_values3_0 = const()[name = string("concat_79_values3_0"), val = tensor([0])]; int32 concat_79_axis_0 = const()[name = string("concat_79_axis_0"), val = int32(0)]; bool concat_79_interleave_0 = const()[name = string("concat_79_interleave_0"), val = bool(false)]; tensor concat_79 = concat(axis = concat_79_axis_0, interleave = concat_79_interleave_0, values = (expand_dims_118, concat_79_values1_0, var_587, concat_79_values3_0))[name = string("concat_79")]; tensor model_model_kv_cache_0_internal_tensor_assign_20_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_78, begin_mask = model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0, end = concat_79, end_mask = model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_20_stride_0, update = var_2315, x = coreml_update_state_50)[name = string("model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_83_write_state")]; tensor coreml_update_state_51 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_83")]; tensor var_2375_begin_0 = const()[name = string("op_2375_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_2375_end_0 = const()[name = string("op_2375_end_0"), val = tensor([10, 8, 4096, 64])]; tensor var_2375_end_mask_0 = const()[name = string("op_2375_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2375_cast_fp16 = slice_by_index(begin = var_2375_begin_0, end = var_2375_end_0, end_mask = var_2375_end_mask_0, x = coreml_update_state_51)[name = string("op_2375_cast_fp16")]; tensor K_layer_cache_19_axes_0 = const()[name = string("K_layer_cache_19_axes_0"), val = tensor([0])]; tensor K_layer_cache_19_cast_fp16 = squeeze(axes = K_layer_cache_19_axes_0, x = var_2375_cast_fp16)[name = string("K_layer_cache_19_cast_fp16")]; tensor var_2377_begin_0 = const()[name = string("op_2377_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_2377_end_0 = const()[name = string("op_2377_end_0"), val = tensor([26, 8, 4096, 64])]; tensor var_2377_end_mask_0 = const()[name = string("op_2377_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2377_cast_fp16 = slice_by_index(begin = var_2377_begin_0, end = var_2377_end_0, end_mask = var_2377_end_mask_0, x = coreml_update_state_51)[name = string("op_2377_cast_fp16")]; tensor V_layer_cache_19_axes_0 = const()[name = string("V_layer_cache_19_axes_0"), val = tensor([0])]; tensor V_layer_cache_19_cast_fp16 = squeeze(axes = V_layer_cache_19_axes_0, x = var_2377_cast_fp16)[name = string("V_layer_cache_19_cast_fp16")]; tensor x_263_axes_0 = const()[name = string("x_263_axes_0"), val = tensor([1])]; tensor x_263_cast_fp16 = expand_dims(axes = x_263_axes_0, x = K_layer_cache_19_cast_fp16)[name = string("x_263_cast_fp16")]; tensor var_2386 = const()[name = string("op_2386"), val = tensor([1, 4, 1, 1])]; tensor x_265_cast_fp16 = tile(reps = var_2386, x = x_263_cast_fp16)[name = string("x_265_cast_fp16")]; tensor var_2390 = const()[name = string("op_2390"), val = tensor([1, -1, 4096, 64])]; tensor key_states_39_cast_fp16 = reshape(shape = var_2390, x = x_265_cast_fp16)[name = string("key_states_39_cast_fp16")]; tensor x_269_axes_0 = const()[name = string("x_269_axes_0"), val = tensor([1])]; tensor x_269_cast_fp16 = expand_dims(axes = x_269_axes_0, x = V_layer_cache_19_cast_fp16)[name = string("x_269_cast_fp16")]; tensor var_2393 = const()[name = string("op_2393"), val = tensor([1, 4, 1, 1])]; tensor x_271_cast_fp16 = tile(reps = var_2393, x = x_269_cast_fp16)[name = string("x_271_cast_fp16")]; tensor var_2397 = const()[name = string("op_2397"), val = tensor([1, -1, 4096, 64])]; tensor value_states_39_cast_fp16 = reshape(shape = var_2397, x = x_271_cast_fp16)[name = string("value_states_39_cast_fp16")]; bool var_2400_transpose_x_1 = const()[name = string("op_2400_transpose_x_1"), val = bool(false)]; bool var_2400_transpose_y_1 = const()[name = string("op_2400_transpose_y_1"), val = bool(true)]; tensor var_2400_cast_fp16 = matmul(transpose_x = var_2400_transpose_x_1, transpose_y = var_2400_transpose_y_1, x = rotated_37_cast_fp16, y = key_states_39_cast_fp16)[name = string("op_2400_cast_fp16")]; fp16 var_2401_to_fp16 = const()[name = string("op_2401_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_37_cast_fp16 = mul(x = var_2400_cast_fp16, y = var_2401_to_fp16)[name = string("attn_weights_37_cast_fp16")]; tensor x_273_cast_fp16 = add(x = attn_weights_37_cast_fp16, y = causal_mask)[name = string("x_273_cast_fp16")]; tensor reduce_max_9_axes_0 = const()[name = string("reduce_max_9_axes_0"), val = tensor([-1])]; bool reduce_max_9_keep_dims_0 = const()[name = string("reduce_max_9_keep_dims_0"), val = bool(true)]; tensor reduce_max_9_cast_fp16 = reduce_max(axes = reduce_max_9_axes_0, keep_dims = reduce_max_9_keep_dims_0, x = x_273_cast_fp16)[name = string("reduce_max_9_cast_fp16")]; tensor x_275_cast_fp16 = sub(x = x_273_cast_fp16, y = reduce_max_9_cast_fp16)[name = string("x_275_cast_fp16")]; tensor exp_x_19_cast_fp16 = exp(x = x_275_cast_fp16)[name = string("exp_x_19_cast_fp16")]; tensor var_2412_axes_0 = const()[name = string("op_2412_axes_0"), val = tensor([-1])]; bool var_2412_keep_dims_0 = const()[name = string("op_2412_keep_dims_0"), val = bool(true)]; tensor var_2412_cast_fp16 = reduce_sum(axes = var_2412_axes_0, keep_dims = var_2412_keep_dims_0, x = exp_x_19_cast_fp16)[name = string("op_2412_cast_fp16")]; tensor attn_weights_39_cast_fp16 = real_div(x = exp_x_19_cast_fp16, y = var_2412_cast_fp16)[name = string("attn_weights_39_cast_fp16")]; bool attn_output_55_transpose_x_0 = const()[name = string("attn_output_55_transpose_x_0"), val = bool(false)]; bool attn_output_55_transpose_y_0 = const()[name = string("attn_output_55_transpose_y_0"), val = bool(false)]; tensor attn_output_55_cast_fp16 = matmul(transpose_x = attn_output_55_transpose_x_0, transpose_y = attn_output_55_transpose_y_0, x = attn_weights_39_cast_fp16, y = value_states_39_cast_fp16)[name = string("attn_output_55_cast_fp16")]; tensor var_2415_perm_0 = const()[name = string("op_2415_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2417 = const()[name = string("op_2417"), val = tensor([1, 1, 2048])]; tensor var_2415_cast_fp16 = transpose(perm = var_2415_perm_0, x = attn_output_55_cast_fp16)[name = string("transpose_26")]; tensor input_131_cast_fp16 = reshape(shape = var_2417, x = var_2415_cast_fp16)[name = string("input_131_cast_fp16")]; tensor model_model_layers_9_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(475503168))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477600384))))[name = string("model_model_layers_9_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_9_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_9_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_131_cast_fp16)[name = string("linear_9_cast_fp16")]; tensor hidden_states_77_cast_fp16 = add(x = hidden_states_73_cast_fp16, y = linear_9_cast_fp16)[name = string("hidden_states_77_cast_fp16")]; fp16 const_156_promoted_to_fp16 = const()[name = string("const_156_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2423_cast_fp16 = mul(x = hidden_states_77_cast_fp16, y = const_156_promoted_to_fp16)[name = string("op_2423_cast_fp16")]; bool input_133_interleave_0 = const()[name = string("input_133_interleave_0"), val = bool(false)]; tensor input_133_cast_fp16 = concat(axis = var_80, interleave = input_133_interleave_0, values = (hidden_states_77_cast_fp16, var_2423_cast_fp16))[name = string("input_133_cast_fp16")]; tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_74_to_fp16, x = input_133_cast_fp16)[name = string("normed_77_cast_fp16")]; tensor normed_79_begin_0 = const()[name = string("normed_79_begin_0"), val = tensor([0, 0, 0])]; tensor normed_79_end_0 = const()[name = string("normed_79_end_0"), val = tensor([1, 1, 2048])]; tensor normed_79_end_mask_0 = const()[name = string("normed_79_end_mask_0"), val = tensor([true, true, false])]; tensor normed_79_cast_fp16 = slice_by_index(begin = normed_79_begin_0, end = normed_79_end_0, end_mask = normed_79_end_mask_0, x = normed_77_cast_fp16)[name = string("normed_79_cast_fp16")]; tensor const_159_promoted_to_fp16 = const()[name = string("const_159_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477608640)))]; tensor x_277_cast_fp16 = mul(x = normed_79_cast_fp16, y = const_159_promoted_to_fp16)[name = string("x_277_cast_fp16")]; tensor var_2441 = const()[name = string("op_2441"), val = tensor([0, 2, 1])]; tensor input_135_axes_0 = const()[name = string("input_135_axes_0"), val = tensor([2])]; tensor var_2442 = transpose(perm = var_2441, x = x_277_cast_fp16)[name = string("transpose_25")]; tensor input_135 = expand_dims(axes = input_135_axes_0, x = var_2442)[name = string("input_135")]; string input_137_pad_type_0 = const()[name = string("input_137_pad_type_0"), val = string("valid")]; tensor input_137_strides_0 = const()[name = string("input_137_strides_0"), val = tensor([1, 1])]; tensor input_137_pad_0 = const()[name = string("input_137_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_137_dilations_0 = const()[name = string("input_137_dilations_0"), val = tensor([1, 1])]; int32 input_137_groups_0 = const()[name = string("input_137_groups_0"), val = int32(1)]; tensor input_137 = conv(dilations = input_137_dilations_0, groups = input_137_groups_0, pad = input_137_pad_0, pad_type = input_137_pad_type_0, strides = input_137_strides_0, weight = model_model_layers_9_mlp_gate_proj_weight_palettized, x = input_135)[name = string("input_137")]; string up_states_19_pad_type_0 = const()[name = string("up_states_19_pad_type_0"), val = string("valid")]; tensor up_states_19_strides_0 = const()[name = string("up_states_19_strides_0"), val = tensor([1, 1])]; tensor up_states_19_pad_0 = const()[name = string("up_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_19_dilations_0 = const()[name = string("up_states_19_dilations_0"), val = tensor([1, 1])]; int32 up_states_19_groups_0 = const()[name = string("up_states_19_groups_0"), val = int32(1)]; tensor up_states_19 = conv(dilations = up_states_19_dilations_0, groups = up_states_19_groups_0, pad = up_states_19_pad_0, pad_type = up_states_19_pad_type_0, strides = up_states_19_strides_0, weight = model_model_layers_9_mlp_up_proj_weight_palettized, x = input_135)[name = string("up_states_19")]; tensor gate_states_19 = silu(x = input_137)[name = string("gate_states_19")]; tensor input_139 = mul(x = gate_states_19, y = up_states_19)[name = string("input_139")]; string hidden_states_79_pad_type_0 = const()[name = string("hidden_states_79_pad_type_0"), val = string("valid")]; tensor hidden_states_79_strides_0 = const()[name = string("hidden_states_79_strides_0"), val = tensor([1, 1])]; tensor hidden_states_79_pad_0 = const()[name = string("hidden_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_79_dilations_0 = const()[name = string("hidden_states_79_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_79_groups_0 = const()[name = string("hidden_states_79_groups_0"), val = int32(1)]; tensor hidden_states_79 = conv(dilations = hidden_states_79_dilations_0, groups = hidden_states_79_groups_0, pad = hidden_states_79_pad_0, pad_type = hidden_states_79_pad_type_0, strides = hidden_states_79_strides_0, weight = model_model_layers_9_mlp_down_proj_weight_palettized, x = input_139)[name = string("hidden_states_79")]; tensor var_2464_axes_0 = const()[name = string("op_2464_axes_0"), val = tensor([2])]; tensor var_2464 = squeeze(axes = var_2464_axes_0, x = hidden_states_79)[name = string("op_2464")]; tensor var_2465 = const()[name = string("op_2465"), val = tensor([0, 2, 1])]; tensor var_2466 = transpose(perm = var_2465, x = var_2464)[name = string("transpose_24")]; tensor hidden_states_81_cast_fp16 = add(x = hidden_states_77_cast_fp16, y = var_2466)[name = string("hidden_states_81_cast_fp16")]; fp16 const_160_promoted_to_fp16 = const()[name = string("const_160_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2469_cast_fp16 = mul(x = hidden_states_81_cast_fp16, y = const_160_promoted_to_fp16)[name = string("op_2469_cast_fp16")]; bool input_141_interleave_0 = const()[name = string("input_141_interleave_0"), val = bool(false)]; tensor input_141_cast_fp16 = concat(axis = var_80, interleave = input_141_interleave_0, values = (hidden_states_81_cast_fp16, var_2469_cast_fp16))[name = string("input_141_cast_fp16")]; tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_74_to_fp16, x = input_141_cast_fp16)[name = string("normed_81_cast_fp16")]; tensor normed_83_begin_0 = const()[name = string("normed_83_begin_0"), val = tensor([0, 0, 0])]; tensor normed_83_end_0 = const()[name = string("normed_83_end_0"), val = tensor([1, 1, 2048])]; tensor normed_83_end_mask_0 = const()[name = string("normed_83_end_mask_0"), val = tensor([true, true, false])]; tensor normed_83_cast_fp16 = slice_by_index(begin = normed_83_begin_0, end = normed_83_end_0, end_mask = normed_83_end_mask_0, x = normed_81_cast_fp16)[name = string("normed_83_cast_fp16")]; tensor const_163_promoted_to_fp16 = const()[name = string("const_163_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477612800)))]; tensor hidden_states_83_cast_fp16 = mul(x = normed_83_cast_fp16, y = const_163_promoted_to_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor var_2483 = const()[name = string("op_2483"), val = tensor([0, 2, 1])]; tensor var_2485_axes_0 = const()[name = string("op_2485_axes_0"), val = tensor([2])]; tensor var_2484_cast_fp16 = transpose(perm = var_2483, x = hidden_states_83_cast_fp16)[name = string("transpose_23")]; tensor var_2485_cast_fp16 = expand_dims(axes = var_2485_axes_0, x = var_2484_cast_fp16)[name = string("op_2485_cast_fp16")]; string var_2492_pad_type_0 = const()[name = string("op_2492_pad_type_0"), val = string("valid")]; tensor var_2492_strides_0 = const()[name = string("op_2492_strides_0"), val = tensor([1, 1])]; tensor var_2492_pad_0 = const()[name = string("op_2492_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2492_dilations_0 = const()[name = string("op_2492_dilations_0"), val = tensor([1, 1])]; int32 var_2492_groups_0 = const()[name = string("op_2492_groups_0"), val = int32(1)]; tensor var_2492 = conv(dilations = var_2492_dilations_0, groups = var_2492_groups_0, pad = var_2492_pad_0, pad_type = var_2492_pad_type_0, strides = var_2492_strides_0, weight = model_model_layers_10_self_attn_q_proj_weight_palettized, x = var_2485_cast_fp16)[name = string("op_2492")]; tensor var_2493 = const()[name = string("op_2493"), val = tensor([1, 32, 1, 64])]; tensor var_2494 = reshape(shape = var_2493, x = var_2492)[name = string("op_2494")]; string var_2501_pad_type_0 = const()[name = string("op_2501_pad_type_0"), val = string("valid")]; tensor var_2501_strides_0 = const()[name = string("op_2501_strides_0"), val = tensor([1, 1])]; tensor var_2501_pad_0 = const()[name = string("op_2501_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2501_dilations_0 = const()[name = string("op_2501_dilations_0"), val = tensor([1, 1])]; int32 var_2501_groups_0 = const()[name = string("op_2501_groups_0"), val = int32(1)]; tensor var_2501 = conv(dilations = var_2501_dilations_0, groups = var_2501_groups_0, pad = var_2501_pad_0, pad_type = var_2501_pad_type_0, strides = var_2501_strides_0, weight = model_model_layers_10_self_attn_k_proj_weight_palettized, x = var_2485_cast_fp16)[name = string("op_2501")]; tensor var_2502 = const()[name = string("op_2502"), val = tensor([1, 8, 1, 64])]; tensor var_2503 = reshape(shape = var_2502, x = var_2501)[name = string("op_2503")]; string var_2510_pad_type_0 = const()[name = string("op_2510_pad_type_0"), val = string("valid")]; tensor var_2510_strides_0 = const()[name = string("op_2510_strides_0"), val = tensor([1, 1])]; tensor var_2510_pad_0 = const()[name = string("op_2510_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2510_dilations_0 = const()[name = string("op_2510_dilations_0"), val = tensor([1, 1])]; int32 var_2510_groups_0 = const()[name = string("op_2510_groups_0"), val = int32(1)]; tensor var_2510 = conv(dilations = var_2510_dilations_0, groups = var_2510_groups_0, pad = var_2510_pad_0, pad_type = var_2510_pad_type_0, strides = var_2510_strides_0, weight = model_model_layers_10_self_attn_v_proj_weight_palettized, x = var_2485_cast_fp16)[name = string("op_2510")]; tensor var_2511 = const()[name = string("op_2511"), val = tensor([1, 8, 1, 64])]; tensor var_2512 = reshape(shape = var_2511, x = var_2510)[name = string("op_2512")]; tensor x1_41_begin_0 = const()[name = string("x1_41_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_41_end_0 = const()[name = string("x1_41_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_41_end_mask_0 = const()[name = string("x1_41_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_41 = slice_by_index(begin = x1_41_begin_0, end = x1_41_end_0, end_mask = x1_41_end_mask_0, x = var_2494)[name = string("x1_41")]; tensor x2_41_begin_0 = const()[name = string("x2_41_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_41_end_0 = const()[name = string("x2_41_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_41_end_mask_0 = const()[name = string("x2_41_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_41 = slice_by_index(begin = x2_41_begin_0, end = x2_41_end_0, end_mask = x2_41_end_mask_0, x = var_2494)[name = string("x2_41")]; tensor var_2526_cast_fp16 = mul(x = x1_41, y = cos_3_cast_fp16)[name = string("op_2526_cast_fp16")]; tensor var_2527_cast_fp16 = mul(x = x2_41, y = sin_3_cast_fp16)[name = string("op_2527_cast_fp16")]; tensor var_2528_cast_fp16 = sub(x = var_2526_cast_fp16, y = var_2527_cast_fp16)[name = string("op_2528_cast_fp16")]; tensor var_2529_cast_fp16 = mul(x = x2_41, y = cos_3_cast_fp16)[name = string("op_2529_cast_fp16")]; tensor var_2530_cast_fp16 = mul(x = x1_41, y = sin_3_cast_fp16)[name = string("op_2530_cast_fp16")]; tensor var_2531_cast_fp16 = add(x = var_2529_cast_fp16, y = var_2530_cast_fp16)[name = string("op_2531_cast_fp16")]; bool rotated_41_interleave_0 = const()[name = string("rotated_41_interleave_0"), val = bool(false)]; tensor rotated_41_cast_fp16 = concat(axis = var_80, interleave = rotated_41_interleave_0, values = (var_2528_cast_fp16, var_2531_cast_fp16))[name = string("rotated_41_cast_fp16")]; tensor x1_43_begin_0 = const()[name = string("x1_43_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_43_end_0 = const()[name = string("x1_43_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_43_end_mask_0 = const()[name = string("x1_43_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_43 = slice_by_index(begin = x1_43_begin_0, end = x1_43_end_0, end_mask = x1_43_end_mask_0, x = var_2503)[name = string("x1_43")]; tensor x2_43_begin_0 = const()[name = string("x2_43_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_43_end_0 = const()[name = string("x2_43_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_43_end_mask_0 = const()[name = string("x2_43_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_43 = slice_by_index(begin = x2_43_begin_0, end = x2_43_end_0, end_mask = x2_43_end_mask_0, x = var_2503)[name = string("x2_43")]; tensor var_2547_cast_fp16 = mul(x = x1_43, y = cos_3_cast_fp16)[name = string("op_2547_cast_fp16")]; tensor var_2548_cast_fp16 = mul(x = x2_43, y = sin_3_cast_fp16)[name = string("op_2548_cast_fp16")]; tensor var_2549_cast_fp16 = sub(x = var_2547_cast_fp16, y = var_2548_cast_fp16)[name = string("op_2549_cast_fp16")]; tensor var_2550_cast_fp16 = mul(x = x2_43, y = cos_3_cast_fp16)[name = string("op_2550_cast_fp16")]; tensor var_2551_cast_fp16 = mul(x = x1_43, y = sin_3_cast_fp16)[name = string("op_2551_cast_fp16")]; tensor var_2552_cast_fp16 = add(x = var_2550_cast_fp16, y = var_2551_cast_fp16)[name = string("op_2552_cast_fp16")]; bool rotated_43_interleave_0 = const()[name = string("rotated_43_interleave_0"), val = bool(false)]; tensor rotated_43_cast_fp16 = concat(axis = var_80, interleave = rotated_43_interleave_0, values = (var_2549_cast_fp16, var_2552_cast_fp16))[name = string("rotated_43_cast_fp16")]; tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; int32 concat_82_axis_0 = const()[name = string("concat_82_axis_0"), val = int32(0)]; bool concat_82_interleave_0 = const()[name = string("concat_82_interleave_0"), val = bool(false)]; tensor concat_82 = concat(axis = concat_82_axis_0, interleave = concat_82_interleave_0, values = (expand_dims_120, expand_dims_121, current_pos, expand_dims_123))[name = string("concat_82")]; tensor concat_83_values1_0 = const()[name = string("concat_83_values1_0"), val = tensor([0])]; tensor concat_83_values3_0 = const()[name = string("concat_83_values3_0"), val = tensor([0])]; int32 concat_83_axis_0 = const()[name = string("concat_83_axis_0"), val = int32(0)]; bool concat_83_interleave_0 = const()[name = string("concat_83_interleave_0"), val = bool(false)]; tensor concat_83 = concat(axis = concat_83_axis_0, interleave = concat_83_interleave_0, values = (expand_dims_124, concat_83_values1_0, var_587, concat_83_values3_0))[name = string("concat_83")]; tensor model_model_kv_cache_0_internal_tensor_assign_21_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_82, begin_mask = model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0, end = concat_83, end_mask = model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_21_stride_0, update = rotated_43_cast_fp16, x = coreml_update_state_51)[name = string("model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_84_write_state")]; tensor coreml_update_state_52 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_84")]; tensor expand_dims_126 = const()[name = string("expand_dims_126"), val = tensor([26])]; tensor expand_dims_127 = const()[name = string("expand_dims_127"), val = tensor([0])]; tensor expand_dims_129 = const()[name = string("expand_dims_129"), val = tensor([0])]; tensor expand_dims_130 = const()[name = string("expand_dims_130"), val = tensor([27])]; int32 concat_86_axis_0 = const()[name = string("concat_86_axis_0"), val = int32(0)]; bool concat_86_interleave_0 = const()[name = string("concat_86_interleave_0"), val = bool(false)]; tensor concat_86 = concat(axis = concat_86_axis_0, interleave = concat_86_interleave_0, values = (expand_dims_126, expand_dims_127, current_pos, expand_dims_129))[name = string("concat_86")]; tensor concat_87_values1_0 = const()[name = string("concat_87_values1_0"), val = tensor([0])]; tensor concat_87_values3_0 = const()[name = string("concat_87_values3_0"), val = tensor([0])]; int32 concat_87_axis_0 = const()[name = string("concat_87_axis_0"), val = int32(0)]; bool concat_87_interleave_0 = const()[name = string("concat_87_interleave_0"), val = bool(false)]; tensor concat_87 = concat(axis = concat_87_axis_0, interleave = concat_87_interleave_0, values = (expand_dims_130, concat_87_values1_0, var_587, concat_87_values3_0))[name = string("concat_87")]; tensor model_model_kv_cache_0_internal_tensor_assign_22_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_86, begin_mask = model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0, end = concat_87, end_mask = model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_22_stride_0, update = var_2512, x = coreml_update_state_52)[name = string("model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_85_write_state")]; tensor coreml_update_state_53 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_85")]; tensor var_2572_begin_0 = const()[name = string("op_2572_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_2572_end_0 = const()[name = string("op_2572_end_0"), val = tensor([11, 8, 4096, 64])]; tensor var_2572_end_mask_0 = const()[name = string("op_2572_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2572_cast_fp16 = slice_by_index(begin = var_2572_begin_0, end = var_2572_end_0, end_mask = var_2572_end_mask_0, x = coreml_update_state_53)[name = string("op_2572_cast_fp16")]; tensor K_layer_cache_21_axes_0 = const()[name = string("K_layer_cache_21_axes_0"), val = tensor([0])]; tensor K_layer_cache_21_cast_fp16 = squeeze(axes = K_layer_cache_21_axes_0, x = var_2572_cast_fp16)[name = string("K_layer_cache_21_cast_fp16")]; tensor var_2574_begin_0 = const()[name = string("op_2574_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_2574_end_0 = const()[name = string("op_2574_end_0"), val = tensor([27, 8, 4096, 64])]; tensor var_2574_end_mask_0 = const()[name = string("op_2574_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2574_cast_fp16 = slice_by_index(begin = var_2574_begin_0, end = var_2574_end_0, end_mask = var_2574_end_mask_0, x = coreml_update_state_53)[name = string("op_2574_cast_fp16")]; tensor V_layer_cache_21_axes_0 = const()[name = string("V_layer_cache_21_axes_0"), val = tensor([0])]; tensor V_layer_cache_21_cast_fp16 = squeeze(axes = V_layer_cache_21_axes_0, x = var_2574_cast_fp16)[name = string("V_layer_cache_21_cast_fp16")]; tensor x_291_axes_0 = const()[name = string("x_291_axes_0"), val = tensor([1])]; tensor x_291_cast_fp16 = expand_dims(axes = x_291_axes_0, x = K_layer_cache_21_cast_fp16)[name = string("x_291_cast_fp16")]; tensor var_2583 = const()[name = string("op_2583"), val = tensor([1, 4, 1, 1])]; tensor x_293_cast_fp16 = tile(reps = var_2583, x = x_291_cast_fp16)[name = string("x_293_cast_fp16")]; tensor var_2587 = const()[name = string("op_2587"), val = tensor([1, -1, 4096, 64])]; tensor key_states_43_cast_fp16 = reshape(shape = var_2587, x = x_293_cast_fp16)[name = string("key_states_43_cast_fp16")]; tensor x_297_axes_0 = const()[name = string("x_297_axes_0"), val = tensor([1])]; tensor x_297_cast_fp16 = expand_dims(axes = x_297_axes_0, x = V_layer_cache_21_cast_fp16)[name = string("x_297_cast_fp16")]; tensor var_2590 = const()[name = string("op_2590"), val = tensor([1, 4, 1, 1])]; tensor x_299_cast_fp16 = tile(reps = var_2590, x = x_297_cast_fp16)[name = string("x_299_cast_fp16")]; tensor var_2594 = const()[name = string("op_2594"), val = tensor([1, -1, 4096, 64])]; tensor value_states_43_cast_fp16 = reshape(shape = var_2594, x = x_299_cast_fp16)[name = string("value_states_43_cast_fp16")]; bool var_2597_transpose_x_1 = const()[name = string("op_2597_transpose_x_1"), val = bool(false)]; bool var_2597_transpose_y_1 = const()[name = string("op_2597_transpose_y_1"), val = bool(true)]; tensor var_2597_cast_fp16 = matmul(transpose_x = var_2597_transpose_x_1, transpose_y = var_2597_transpose_y_1, x = rotated_41_cast_fp16, y = key_states_43_cast_fp16)[name = string("op_2597_cast_fp16")]; fp16 var_2598_to_fp16 = const()[name = string("op_2598_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_41_cast_fp16 = mul(x = var_2597_cast_fp16, y = var_2598_to_fp16)[name = string("attn_weights_41_cast_fp16")]; tensor x_301_cast_fp16 = add(x = attn_weights_41_cast_fp16, y = causal_mask)[name = string("x_301_cast_fp16")]; tensor reduce_max_10_axes_0 = const()[name = string("reduce_max_10_axes_0"), val = tensor([-1])]; bool reduce_max_10_keep_dims_0 = const()[name = string("reduce_max_10_keep_dims_0"), val = bool(true)]; tensor reduce_max_10_cast_fp16 = reduce_max(axes = reduce_max_10_axes_0, keep_dims = reduce_max_10_keep_dims_0, x = x_301_cast_fp16)[name = string("reduce_max_10_cast_fp16")]; tensor x_303_cast_fp16 = sub(x = x_301_cast_fp16, y = reduce_max_10_cast_fp16)[name = string("x_303_cast_fp16")]; tensor exp_x_21_cast_fp16 = exp(x = x_303_cast_fp16)[name = string("exp_x_21_cast_fp16")]; tensor var_2609_axes_0 = const()[name = string("op_2609_axes_0"), val = tensor([-1])]; bool var_2609_keep_dims_0 = const()[name = string("op_2609_keep_dims_0"), val = bool(true)]; tensor var_2609_cast_fp16 = reduce_sum(axes = var_2609_axes_0, keep_dims = var_2609_keep_dims_0, x = exp_x_21_cast_fp16)[name = string("op_2609_cast_fp16")]; tensor attn_weights_43_cast_fp16 = real_div(x = exp_x_21_cast_fp16, y = var_2609_cast_fp16)[name = string("attn_weights_43_cast_fp16")]; bool attn_output_61_transpose_x_0 = const()[name = string("attn_output_61_transpose_x_0"), val = bool(false)]; bool attn_output_61_transpose_y_0 = const()[name = string("attn_output_61_transpose_y_0"), val = bool(false)]; tensor attn_output_61_cast_fp16 = matmul(transpose_x = attn_output_61_transpose_x_0, transpose_y = attn_output_61_transpose_y_0, x = attn_weights_43_cast_fp16, y = value_states_43_cast_fp16)[name = string("attn_output_61_cast_fp16")]; tensor var_2612_perm_0 = const()[name = string("op_2612_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2614 = const()[name = string("op_2614"), val = tensor([1, 1, 2048])]; tensor var_2612_cast_fp16 = transpose(perm = var_2612_perm_0, x = attn_output_61_cast_fp16)[name = string("transpose_22")]; tensor input_145_cast_fp16 = reshape(shape = var_2614, x = var_2612_cast_fp16)[name = string("input_145_cast_fp16")]; tensor model_model_layers_10_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477616960))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479714176))))[name = string("model_model_layers_10_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_10_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_10_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_145_cast_fp16)[name = string("linear_10_cast_fp16")]; tensor hidden_states_85_cast_fp16 = add(x = hidden_states_81_cast_fp16, y = linear_10_cast_fp16)[name = string("hidden_states_85_cast_fp16")]; fp16 const_172_promoted_to_fp16 = const()[name = string("const_172_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2620_cast_fp16 = mul(x = hidden_states_85_cast_fp16, y = const_172_promoted_to_fp16)[name = string("op_2620_cast_fp16")]; bool input_147_interleave_0 = const()[name = string("input_147_interleave_0"), val = bool(false)]; tensor input_147_cast_fp16 = concat(axis = var_80, interleave = input_147_interleave_0, values = (hidden_states_85_cast_fp16, var_2620_cast_fp16))[name = string("input_147_cast_fp16")]; tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_74_to_fp16, x = input_147_cast_fp16)[name = string("normed_85_cast_fp16")]; tensor normed_87_begin_0 = const()[name = string("normed_87_begin_0"), val = tensor([0, 0, 0])]; tensor normed_87_end_0 = const()[name = string("normed_87_end_0"), val = tensor([1, 1, 2048])]; tensor normed_87_end_mask_0 = const()[name = string("normed_87_end_mask_0"), val = tensor([true, true, false])]; tensor normed_87_cast_fp16 = slice_by_index(begin = normed_87_begin_0, end = normed_87_end_0, end_mask = normed_87_end_mask_0, x = normed_85_cast_fp16)[name = string("normed_87_cast_fp16")]; tensor const_175_promoted_to_fp16 = const()[name = string("const_175_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479722432)))]; tensor x_305_cast_fp16 = mul(x = normed_87_cast_fp16, y = const_175_promoted_to_fp16)[name = string("x_305_cast_fp16")]; tensor var_2638 = const()[name = string("op_2638"), val = tensor([0, 2, 1])]; tensor input_149_axes_0 = const()[name = string("input_149_axes_0"), val = tensor([2])]; tensor var_2639 = transpose(perm = var_2638, x = x_305_cast_fp16)[name = string("transpose_21")]; tensor input_149 = expand_dims(axes = input_149_axes_0, x = var_2639)[name = string("input_149")]; string input_151_pad_type_0 = const()[name = string("input_151_pad_type_0"), val = string("valid")]; tensor input_151_strides_0 = const()[name = string("input_151_strides_0"), val = tensor([1, 1])]; tensor input_151_pad_0 = const()[name = string("input_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_151_dilations_0 = const()[name = string("input_151_dilations_0"), val = tensor([1, 1])]; int32 input_151_groups_0 = const()[name = string("input_151_groups_0"), val = int32(1)]; tensor input_151 = conv(dilations = input_151_dilations_0, groups = input_151_groups_0, pad = input_151_pad_0, pad_type = input_151_pad_type_0, strides = input_151_strides_0, weight = model_model_layers_10_mlp_gate_proj_weight_palettized, x = input_149)[name = string("input_151")]; string up_states_21_pad_type_0 = const()[name = string("up_states_21_pad_type_0"), val = string("valid")]; tensor up_states_21_strides_0 = const()[name = string("up_states_21_strides_0"), val = tensor([1, 1])]; tensor up_states_21_pad_0 = const()[name = string("up_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_21_dilations_0 = const()[name = string("up_states_21_dilations_0"), val = tensor([1, 1])]; int32 up_states_21_groups_0 = const()[name = string("up_states_21_groups_0"), val = int32(1)]; tensor up_states_21 = conv(dilations = up_states_21_dilations_0, groups = up_states_21_groups_0, pad = up_states_21_pad_0, pad_type = up_states_21_pad_type_0, strides = up_states_21_strides_0, weight = model_model_layers_10_mlp_up_proj_weight_palettized, x = input_149)[name = string("up_states_21")]; tensor gate_states_21 = silu(x = input_151)[name = string("gate_states_21")]; tensor input_153 = mul(x = gate_states_21, y = up_states_21)[name = string("input_153")]; string hidden_states_87_pad_type_0 = const()[name = string("hidden_states_87_pad_type_0"), val = string("valid")]; tensor hidden_states_87_strides_0 = const()[name = string("hidden_states_87_strides_0"), val = tensor([1, 1])]; tensor hidden_states_87_pad_0 = const()[name = string("hidden_states_87_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_87_dilations_0 = const()[name = string("hidden_states_87_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_87_groups_0 = const()[name = string("hidden_states_87_groups_0"), val = int32(1)]; tensor hidden_states_87 = conv(dilations = hidden_states_87_dilations_0, groups = hidden_states_87_groups_0, pad = hidden_states_87_pad_0, pad_type = hidden_states_87_pad_type_0, strides = hidden_states_87_strides_0, weight = model_model_layers_10_mlp_down_proj_weight_palettized, x = input_153)[name = string("hidden_states_87")]; tensor var_2661_axes_0 = const()[name = string("op_2661_axes_0"), val = tensor([2])]; tensor var_2661 = squeeze(axes = var_2661_axes_0, x = hidden_states_87)[name = string("op_2661")]; tensor var_2662 = const()[name = string("op_2662"), val = tensor([0, 2, 1])]; tensor var_2663 = transpose(perm = var_2662, x = var_2661)[name = string("transpose_20")]; tensor hidden_states_89_cast_fp16 = add(x = hidden_states_85_cast_fp16, y = var_2663)[name = string("hidden_states_89_cast_fp16")]; fp16 const_176_promoted_to_fp16 = const()[name = string("const_176_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2666_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_176_promoted_to_fp16)[name = string("op_2666_cast_fp16")]; bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; tensor input_155_cast_fp16 = concat(axis = var_80, interleave = input_155_interleave_0, values = (hidden_states_89_cast_fp16, var_2666_cast_fp16))[name = string("input_155_cast_fp16")]; tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_74_to_fp16, x = input_155_cast_fp16)[name = string("normed_89_cast_fp16")]; tensor normed_91_begin_0 = const()[name = string("normed_91_begin_0"), val = tensor([0, 0, 0])]; tensor normed_91_end_0 = const()[name = string("normed_91_end_0"), val = tensor([1, 1, 2048])]; tensor normed_91_end_mask_0 = const()[name = string("normed_91_end_mask_0"), val = tensor([true, true, false])]; tensor normed_91_cast_fp16 = slice_by_index(begin = normed_91_begin_0, end = normed_91_end_0, end_mask = normed_91_end_mask_0, x = normed_89_cast_fp16)[name = string("normed_91_cast_fp16")]; tensor const_179_promoted_to_fp16 = const()[name = string("const_179_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479726592)))]; tensor hidden_states_91_cast_fp16 = mul(x = normed_91_cast_fp16, y = const_179_promoted_to_fp16)[name = string("hidden_states_91_cast_fp16")]; tensor var_2680 = const()[name = string("op_2680"), val = tensor([0, 2, 1])]; tensor var_2682_axes_0 = const()[name = string("op_2682_axes_0"), val = tensor([2])]; tensor var_2681_cast_fp16 = transpose(perm = var_2680, x = hidden_states_91_cast_fp16)[name = string("transpose_19")]; tensor var_2682_cast_fp16 = expand_dims(axes = var_2682_axes_0, x = var_2681_cast_fp16)[name = string("op_2682_cast_fp16")]; string var_2689_pad_type_0 = const()[name = string("op_2689_pad_type_0"), val = string("valid")]; tensor var_2689_strides_0 = const()[name = string("op_2689_strides_0"), val = tensor([1, 1])]; tensor var_2689_pad_0 = const()[name = string("op_2689_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2689_dilations_0 = const()[name = string("op_2689_dilations_0"), val = tensor([1, 1])]; int32 var_2689_groups_0 = const()[name = string("op_2689_groups_0"), val = int32(1)]; tensor var_2689 = conv(dilations = var_2689_dilations_0, groups = var_2689_groups_0, pad = var_2689_pad_0, pad_type = var_2689_pad_type_0, strides = var_2689_strides_0, weight = model_model_layers_11_self_attn_q_proj_weight_palettized, x = var_2682_cast_fp16)[name = string("op_2689")]; tensor var_2690 = const()[name = string("op_2690"), val = tensor([1, 32, 1, 64])]; tensor var_2691 = reshape(shape = var_2690, x = var_2689)[name = string("op_2691")]; string var_2698_pad_type_0 = const()[name = string("op_2698_pad_type_0"), val = string("valid")]; tensor var_2698_strides_0 = const()[name = string("op_2698_strides_0"), val = tensor([1, 1])]; tensor var_2698_pad_0 = const()[name = string("op_2698_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2698_dilations_0 = const()[name = string("op_2698_dilations_0"), val = tensor([1, 1])]; int32 var_2698_groups_0 = const()[name = string("op_2698_groups_0"), val = int32(1)]; tensor var_2698 = conv(dilations = var_2698_dilations_0, groups = var_2698_groups_0, pad = var_2698_pad_0, pad_type = var_2698_pad_type_0, strides = var_2698_strides_0, weight = model_model_layers_11_self_attn_k_proj_weight_palettized, x = var_2682_cast_fp16)[name = string("op_2698")]; tensor var_2699 = const()[name = string("op_2699"), val = tensor([1, 8, 1, 64])]; tensor var_2700 = reshape(shape = var_2699, x = var_2698)[name = string("op_2700")]; string var_2707_pad_type_0 = const()[name = string("op_2707_pad_type_0"), val = string("valid")]; tensor var_2707_strides_0 = const()[name = string("op_2707_strides_0"), val = tensor([1, 1])]; tensor var_2707_pad_0 = const()[name = string("op_2707_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2707_dilations_0 = const()[name = string("op_2707_dilations_0"), val = tensor([1, 1])]; int32 var_2707_groups_0 = const()[name = string("op_2707_groups_0"), val = int32(1)]; tensor var_2707 = conv(dilations = var_2707_dilations_0, groups = var_2707_groups_0, pad = var_2707_pad_0, pad_type = var_2707_pad_type_0, strides = var_2707_strides_0, weight = model_model_layers_11_self_attn_v_proj_weight_palettized, x = var_2682_cast_fp16)[name = string("op_2707")]; tensor var_2708 = const()[name = string("op_2708"), val = tensor([1, 8, 1, 64])]; tensor var_2709 = reshape(shape = var_2708, x = var_2707)[name = string("op_2709")]; tensor x1_45_begin_0 = const()[name = string("x1_45_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_45_end_0 = const()[name = string("x1_45_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_45_end_mask_0 = const()[name = string("x1_45_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_45 = slice_by_index(begin = x1_45_begin_0, end = x1_45_end_0, end_mask = x1_45_end_mask_0, x = var_2691)[name = string("x1_45")]; tensor x2_45_begin_0 = const()[name = string("x2_45_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_45_end_0 = const()[name = string("x2_45_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_45_end_mask_0 = const()[name = string("x2_45_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_45 = slice_by_index(begin = x2_45_begin_0, end = x2_45_end_0, end_mask = x2_45_end_mask_0, x = var_2691)[name = string("x2_45")]; tensor var_2723_cast_fp16 = mul(x = x1_45, y = cos_3_cast_fp16)[name = string("op_2723_cast_fp16")]; tensor var_2724_cast_fp16 = mul(x = x2_45, y = sin_3_cast_fp16)[name = string("op_2724_cast_fp16")]; tensor var_2725_cast_fp16 = sub(x = var_2723_cast_fp16, y = var_2724_cast_fp16)[name = string("op_2725_cast_fp16")]; tensor var_2726_cast_fp16 = mul(x = x2_45, y = cos_3_cast_fp16)[name = string("op_2726_cast_fp16")]; tensor var_2727_cast_fp16 = mul(x = x1_45, y = sin_3_cast_fp16)[name = string("op_2727_cast_fp16")]; tensor var_2728_cast_fp16 = add(x = var_2726_cast_fp16, y = var_2727_cast_fp16)[name = string("op_2728_cast_fp16")]; bool rotated_45_interleave_0 = const()[name = string("rotated_45_interleave_0"), val = bool(false)]; tensor rotated_45_cast_fp16 = concat(axis = var_80, interleave = rotated_45_interleave_0, values = (var_2725_cast_fp16, var_2728_cast_fp16))[name = string("rotated_45_cast_fp16")]; tensor x1_47_begin_0 = const()[name = string("x1_47_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_47_end_0 = const()[name = string("x1_47_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_47_end_mask_0 = const()[name = string("x1_47_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_47 = slice_by_index(begin = x1_47_begin_0, end = x1_47_end_0, end_mask = x1_47_end_mask_0, x = var_2700)[name = string("x1_47")]; tensor x2_47_begin_0 = const()[name = string("x2_47_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_47_end_0 = const()[name = string("x2_47_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_47_end_mask_0 = const()[name = string("x2_47_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_47 = slice_by_index(begin = x2_47_begin_0, end = x2_47_end_0, end_mask = x2_47_end_mask_0, x = var_2700)[name = string("x2_47")]; tensor var_2744_cast_fp16 = mul(x = x1_47, y = cos_3_cast_fp16)[name = string("op_2744_cast_fp16")]; tensor var_2745_cast_fp16 = mul(x = x2_47, y = sin_3_cast_fp16)[name = string("op_2745_cast_fp16")]; tensor var_2746_cast_fp16 = sub(x = var_2744_cast_fp16, y = var_2745_cast_fp16)[name = string("op_2746_cast_fp16")]; tensor var_2747_cast_fp16 = mul(x = x2_47, y = cos_3_cast_fp16)[name = string("op_2747_cast_fp16")]; tensor var_2748_cast_fp16 = mul(x = x1_47, y = sin_3_cast_fp16)[name = string("op_2748_cast_fp16")]; tensor var_2749_cast_fp16 = add(x = var_2747_cast_fp16, y = var_2748_cast_fp16)[name = string("op_2749_cast_fp16")]; bool rotated_47_interleave_0 = const()[name = string("rotated_47_interleave_0"), val = bool(false)]; tensor rotated_47_cast_fp16 = concat(axis = var_80, interleave = rotated_47_interleave_0, values = (var_2746_cast_fp16, var_2749_cast_fp16))[name = string("rotated_47_cast_fp16")]; tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_132, expand_dims_133, current_pos, expand_dims_135))[name = string("concat_90")]; tensor concat_91_values1_0 = const()[name = string("concat_91_values1_0"), val = tensor([0])]; tensor concat_91_values3_0 = const()[name = string("concat_91_values3_0"), val = tensor([0])]; int32 concat_91_axis_0 = const()[name = string("concat_91_axis_0"), val = int32(0)]; bool concat_91_interleave_0 = const()[name = string("concat_91_interleave_0"), val = bool(false)]; tensor concat_91 = concat(axis = concat_91_axis_0, interleave = concat_91_interleave_0, values = (expand_dims_136, concat_91_values1_0, var_587, concat_91_values3_0))[name = string("concat_91")]; tensor model_model_kv_cache_0_internal_tensor_assign_23_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_90, begin_mask = model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0, end = concat_91, end_mask = model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_23_stride_0, update = rotated_47_cast_fp16, x = coreml_update_state_53)[name = string("model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_86_write_state")]; tensor coreml_update_state_54 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_86")]; tensor expand_dims_138 = const()[name = string("expand_dims_138"), val = tensor([27])]; tensor expand_dims_139 = const()[name = string("expand_dims_139"), val = tensor([0])]; tensor expand_dims_141 = const()[name = string("expand_dims_141"), val = tensor([0])]; tensor expand_dims_142 = const()[name = string("expand_dims_142"), val = tensor([28])]; int32 concat_94_axis_0 = const()[name = string("concat_94_axis_0"), val = int32(0)]; bool concat_94_interleave_0 = const()[name = string("concat_94_interleave_0"), val = bool(false)]; tensor concat_94 = concat(axis = concat_94_axis_0, interleave = concat_94_interleave_0, values = (expand_dims_138, expand_dims_139, current_pos, expand_dims_141))[name = string("concat_94")]; tensor concat_95_values1_0 = const()[name = string("concat_95_values1_0"), val = tensor([0])]; tensor concat_95_values3_0 = const()[name = string("concat_95_values3_0"), val = tensor([0])]; int32 concat_95_axis_0 = const()[name = string("concat_95_axis_0"), val = int32(0)]; bool concat_95_interleave_0 = const()[name = string("concat_95_interleave_0"), val = bool(false)]; tensor concat_95 = concat(axis = concat_95_axis_0, interleave = concat_95_interleave_0, values = (expand_dims_142, concat_95_values1_0, var_587, concat_95_values3_0))[name = string("concat_95")]; tensor model_model_kv_cache_0_internal_tensor_assign_24_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_94, begin_mask = model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0, end = concat_95, end_mask = model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_24_stride_0, update = var_2709, x = coreml_update_state_54)[name = string("model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_87_write_state")]; tensor coreml_update_state_55 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_87")]; tensor var_2769_begin_0 = const()[name = string("op_2769_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_2769_end_0 = const()[name = string("op_2769_end_0"), val = tensor([12, 8, 4096, 64])]; tensor var_2769_end_mask_0 = const()[name = string("op_2769_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2769_cast_fp16 = slice_by_index(begin = var_2769_begin_0, end = var_2769_end_0, end_mask = var_2769_end_mask_0, x = coreml_update_state_55)[name = string("op_2769_cast_fp16")]; tensor K_layer_cache_23_axes_0 = const()[name = string("K_layer_cache_23_axes_0"), val = tensor([0])]; tensor K_layer_cache_23_cast_fp16 = squeeze(axes = K_layer_cache_23_axes_0, x = var_2769_cast_fp16)[name = string("K_layer_cache_23_cast_fp16")]; tensor var_2771_begin_0 = const()[name = string("op_2771_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_2771_end_0 = const()[name = string("op_2771_end_0"), val = tensor([28, 8, 4096, 64])]; tensor var_2771_end_mask_0 = const()[name = string("op_2771_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2771_cast_fp16 = slice_by_index(begin = var_2771_begin_0, end = var_2771_end_0, end_mask = var_2771_end_mask_0, x = coreml_update_state_55)[name = string("op_2771_cast_fp16")]; tensor V_layer_cache_23_axes_0 = const()[name = string("V_layer_cache_23_axes_0"), val = tensor([0])]; tensor V_layer_cache_23_cast_fp16 = squeeze(axes = V_layer_cache_23_axes_0, x = var_2771_cast_fp16)[name = string("V_layer_cache_23_cast_fp16")]; tensor x_319_axes_0 = const()[name = string("x_319_axes_0"), val = tensor([1])]; tensor x_319_cast_fp16 = expand_dims(axes = x_319_axes_0, x = K_layer_cache_23_cast_fp16)[name = string("x_319_cast_fp16")]; tensor var_2780 = const()[name = string("op_2780"), val = tensor([1, 4, 1, 1])]; tensor x_321_cast_fp16 = tile(reps = var_2780, x = x_319_cast_fp16)[name = string("x_321_cast_fp16")]; tensor var_2784 = const()[name = string("op_2784"), val = tensor([1, -1, 4096, 64])]; tensor key_states_47_cast_fp16 = reshape(shape = var_2784, x = x_321_cast_fp16)[name = string("key_states_47_cast_fp16")]; tensor x_325_axes_0 = const()[name = string("x_325_axes_0"), val = tensor([1])]; tensor x_325_cast_fp16 = expand_dims(axes = x_325_axes_0, x = V_layer_cache_23_cast_fp16)[name = string("x_325_cast_fp16")]; tensor var_2787 = const()[name = string("op_2787"), val = tensor([1, 4, 1, 1])]; tensor x_327_cast_fp16 = tile(reps = var_2787, x = x_325_cast_fp16)[name = string("x_327_cast_fp16")]; tensor var_2791 = const()[name = string("op_2791"), val = tensor([1, -1, 4096, 64])]; tensor value_states_47_cast_fp16 = reshape(shape = var_2791, x = x_327_cast_fp16)[name = string("value_states_47_cast_fp16")]; bool var_2794_transpose_x_1 = const()[name = string("op_2794_transpose_x_1"), val = bool(false)]; bool var_2794_transpose_y_1 = const()[name = string("op_2794_transpose_y_1"), val = bool(true)]; tensor var_2794_cast_fp16 = matmul(transpose_x = var_2794_transpose_x_1, transpose_y = var_2794_transpose_y_1, x = rotated_45_cast_fp16, y = key_states_47_cast_fp16)[name = string("op_2794_cast_fp16")]; fp16 var_2795_to_fp16 = const()[name = string("op_2795_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_45_cast_fp16 = mul(x = var_2794_cast_fp16, y = var_2795_to_fp16)[name = string("attn_weights_45_cast_fp16")]; tensor x_329_cast_fp16 = add(x = attn_weights_45_cast_fp16, y = causal_mask)[name = string("x_329_cast_fp16")]; tensor reduce_max_11_axes_0 = const()[name = string("reduce_max_11_axes_0"), val = tensor([-1])]; bool reduce_max_11_keep_dims_0 = const()[name = string("reduce_max_11_keep_dims_0"), val = bool(true)]; tensor reduce_max_11_cast_fp16 = reduce_max(axes = reduce_max_11_axes_0, keep_dims = reduce_max_11_keep_dims_0, x = x_329_cast_fp16)[name = string("reduce_max_11_cast_fp16")]; tensor x_331_cast_fp16 = sub(x = x_329_cast_fp16, y = reduce_max_11_cast_fp16)[name = string("x_331_cast_fp16")]; tensor exp_x_23_cast_fp16 = exp(x = x_331_cast_fp16)[name = string("exp_x_23_cast_fp16")]; tensor var_2806_axes_0 = const()[name = string("op_2806_axes_0"), val = tensor([-1])]; bool var_2806_keep_dims_0 = const()[name = string("op_2806_keep_dims_0"), val = bool(true)]; tensor var_2806_cast_fp16 = reduce_sum(axes = var_2806_axes_0, keep_dims = var_2806_keep_dims_0, x = exp_x_23_cast_fp16)[name = string("op_2806_cast_fp16")]; tensor attn_weights_47_cast_fp16 = real_div(x = exp_x_23_cast_fp16, y = var_2806_cast_fp16)[name = string("attn_weights_47_cast_fp16")]; bool attn_output_67_transpose_x_0 = const()[name = string("attn_output_67_transpose_x_0"), val = bool(false)]; bool attn_output_67_transpose_y_0 = const()[name = string("attn_output_67_transpose_y_0"), val = bool(false)]; tensor attn_output_67_cast_fp16 = matmul(transpose_x = attn_output_67_transpose_x_0, transpose_y = attn_output_67_transpose_y_0, x = attn_weights_47_cast_fp16, y = value_states_47_cast_fp16)[name = string("attn_output_67_cast_fp16")]; tensor var_2809_perm_0 = const()[name = string("op_2809_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2811 = const()[name = string("op_2811"), val = tensor([1, 1, 2048])]; tensor var_2809_cast_fp16 = transpose(perm = var_2809_perm_0, x = attn_output_67_cast_fp16)[name = string("transpose_18")]; tensor input_159_cast_fp16 = reshape(shape = var_2811, x = var_2809_cast_fp16)[name = string("input_159_cast_fp16")]; tensor model_model_layers_11_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479730752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481827968))))[name = string("model_model_layers_11_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_11_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_11_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_159_cast_fp16)[name = string("linear_11_cast_fp16")]; tensor hidden_states_93_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = linear_11_cast_fp16)[name = string("hidden_states_93_cast_fp16")]; fp16 const_188_promoted_to_fp16 = const()[name = string("const_188_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2817_cast_fp16 = mul(x = hidden_states_93_cast_fp16, y = const_188_promoted_to_fp16)[name = string("op_2817_cast_fp16")]; bool input_161_interleave_0 = const()[name = string("input_161_interleave_0"), val = bool(false)]; tensor input_161_cast_fp16 = concat(axis = var_80, interleave = input_161_interleave_0, values = (hidden_states_93_cast_fp16, var_2817_cast_fp16))[name = string("input_161_cast_fp16")]; tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_74_to_fp16, x = input_161_cast_fp16)[name = string("normed_93_cast_fp16")]; tensor normed_95_begin_0 = const()[name = string("normed_95_begin_0"), val = tensor([0, 0, 0])]; tensor normed_95_end_0 = const()[name = string("normed_95_end_0"), val = tensor([1, 1, 2048])]; tensor normed_95_end_mask_0 = const()[name = string("normed_95_end_mask_0"), val = tensor([true, true, false])]; tensor normed_95_cast_fp16 = slice_by_index(begin = normed_95_begin_0, end = normed_95_end_0, end_mask = normed_95_end_mask_0, x = normed_93_cast_fp16)[name = string("normed_95_cast_fp16")]; tensor const_191_promoted_to_fp16 = const()[name = string("const_191_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481836224)))]; tensor x_333_cast_fp16 = mul(x = normed_95_cast_fp16, y = const_191_promoted_to_fp16)[name = string("x_333_cast_fp16")]; tensor var_2835 = const()[name = string("op_2835"), val = tensor([0, 2, 1])]; tensor input_163_axes_0 = const()[name = string("input_163_axes_0"), val = tensor([2])]; tensor var_2836 = transpose(perm = var_2835, x = x_333_cast_fp16)[name = string("transpose_17")]; tensor input_163 = expand_dims(axes = input_163_axes_0, x = var_2836)[name = string("input_163")]; string input_165_pad_type_0 = const()[name = string("input_165_pad_type_0"), val = string("valid")]; tensor input_165_strides_0 = const()[name = string("input_165_strides_0"), val = tensor([1, 1])]; tensor input_165_pad_0 = const()[name = string("input_165_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_165_dilations_0 = const()[name = string("input_165_dilations_0"), val = tensor([1, 1])]; int32 input_165_groups_0 = const()[name = string("input_165_groups_0"), val = int32(1)]; tensor input_165 = conv(dilations = input_165_dilations_0, groups = input_165_groups_0, pad = input_165_pad_0, pad_type = input_165_pad_type_0, strides = input_165_strides_0, weight = model_model_layers_11_mlp_gate_proj_weight_palettized, x = input_163)[name = string("input_165")]; string up_states_23_pad_type_0 = const()[name = string("up_states_23_pad_type_0"), val = string("valid")]; tensor up_states_23_strides_0 = const()[name = string("up_states_23_strides_0"), val = tensor([1, 1])]; tensor up_states_23_pad_0 = const()[name = string("up_states_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_23_dilations_0 = const()[name = string("up_states_23_dilations_0"), val = tensor([1, 1])]; int32 up_states_23_groups_0 = const()[name = string("up_states_23_groups_0"), val = int32(1)]; tensor up_states_23 = conv(dilations = up_states_23_dilations_0, groups = up_states_23_groups_0, pad = up_states_23_pad_0, pad_type = up_states_23_pad_type_0, strides = up_states_23_strides_0, weight = model_model_layers_11_mlp_up_proj_weight_palettized, x = input_163)[name = string("up_states_23")]; tensor gate_states_23 = silu(x = input_165)[name = string("gate_states_23")]; tensor input_167 = mul(x = gate_states_23, y = up_states_23)[name = string("input_167")]; string hidden_states_95_pad_type_0 = const()[name = string("hidden_states_95_pad_type_0"), val = string("valid")]; tensor hidden_states_95_strides_0 = const()[name = string("hidden_states_95_strides_0"), val = tensor([1, 1])]; tensor hidden_states_95_pad_0 = const()[name = string("hidden_states_95_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_95_dilations_0 = const()[name = string("hidden_states_95_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_95_groups_0 = const()[name = string("hidden_states_95_groups_0"), val = int32(1)]; tensor hidden_states_95 = conv(dilations = hidden_states_95_dilations_0, groups = hidden_states_95_groups_0, pad = hidden_states_95_pad_0, pad_type = hidden_states_95_pad_type_0, strides = hidden_states_95_strides_0, weight = model_model_layers_11_mlp_down_proj_weight_palettized, x = input_167)[name = string("hidden_states_95")]; tensor var_2858_axes_0 = const()[name = string("op_2858_axes_0"), val = tensor([2])]; tensor var_2858 = squeeze(axes = var_2858_axes_0, x = hidden_states_95)[name = string("op_2858")]; tensor var_2859 = const()[name = string("op_2859"), val = tensor([0, 2, 1])]; tensor var_2860 = transpose(perm = var_2859, x = var_2858)[name = string("transpose_16")]; tensor hidden_states_97_cast_fp16 = add(x = hidden_states_93_cast_fp16, y = var_2860)[name = string("hidden_states_97_cast_fp16")]; fp16 const_192_promoted_to_fp16 = const()[name = string("const_192_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2863_cast_fp16 = mul(x = hidden_states_97_cast_fp16, y = const_192_promoted_to_fp16)[name = string("op_2863_cast_fp16")]; bool input_169_interleave_0 = const()[name = string("input_169_interleave_0"), val = bool(false)]; tensor input_169_cast_fp16 = concat(axis = var_80, interleave = input_169_interleave_0, values = (hidden_states_97_cast_fp16, var_2863_cast_fp16))[name = string("input_169_cast_fp16")]; tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_74_to_fp16, x = input_169_cast_fp16)[name = string("normed_97_cast_fp16")]; tensor normed_99_begin_0 = const()[name = string("normed_99_begin_0"), val = tensor([0, 0, 0])]; tensor normed_99_end_0 = const()[name = string("normed_99_end_0"), val = tensor([1, 1, 2048])]; tensor normed_99_end_mask_0 = const()[name = string("normed_99_end_mask_0"), val = tensor([true, true, false])]; tensor normed_99_cast_fp16 = slice_by_index(begin = normed_99_begin_0, end = normed_99_end_0, end_mask = normed_99_end_mask_0, x = normed_97_cast_fp16)[name = string("normed_99_cast_fp16")]; tensor const_195_promoted_to_fp16 = const()[name = string("const_195_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481840384)))]; tensor hidden_states_99_cast_fp16 = mul(x = normed_99_cast_fp16, y = const_195_promoted_to_fp16)[name = string("hidden_states_99_cast_fp16")]; tensor var_2877 = const()[name = string("op_2877"), val = tensor([0, 2, 1])]; tensor var_2879_axes_0 = const()[name = string("op_2879_axes_0"), val = tensor([2])]; tensor var_2878_cast_fp16 = transpose(perm = var_2877, x = hidden_states_99_cast_fp16)[name = string("transpose_15")]; tensor var_2879_cast_fp16 = expand_dims(axes = var_2879_axes_0, x = var_2878_cast_fp16)[name = string("op_2879_cast_fp16")]; string var_2886_pad_type_0 = const()[name = string("op_2886_pad_type_0"), val = string("valid")]; tensor var_2886_strides_0 = const()[name = string("op_2886_strides_0"), val = tensor([1, 1])]; tensor var_2886_pad_0 = const()[name = string("op_2886_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2886_dilations_0 = const()[name = string("op_2886_dilations_0"), val = tensor([1, 1])]; int32 var_2886_groups_0 = const()[name = string("op_2886_groups_0"), val = int32(1)]; tensor var_2886 = conv(dilations = var_2886_dilations_0, groups = var_2886_groups_0, pad = var_2886_pad_0, pad_type = var_2886_pad_type_0, strides = var_2886_strides_0, weight = model_model_layers_12_self_attn_q_proj_weight_palettized, x = var_2879_cast_fp16)[name = string("op_2886")]; tensor var_2887 = const()[name = string("op_2887"), val = tensor([1, 32, 1, 64])]; tensor var_2888 = reshape(shape = var_2887, x = var_2886)[name = string("op_2888")]; string var_2895_pad_type_0 = const()[name = string("op_2895_pad_type_0"), val = string("valid")]; tensor var_2895_strides_0 = const()[name = string("op_2895_strides_0"), val = tensor([1, 1])]; tensor var_2895_pad_0 = const()[name = string("op_2895_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2895_dilations_0 = const()[name = string("op_2895_dilations_0"), val = tensor([1, 1])]; int32 var_2895_groups_0 = const()[name = string("op_2895_groups_0"), val = int32(1)]; tensor var_2895 = conv(dilations = var_2895_dilations_0, groups = var_2895_groups_0, pad = var_2895_pad_0, pad_type = var_2895_pad_type_0, strides = var_2895_strides_0, weight = model_model_layers_12_self_attn_k_proj_weight_palettized, x = var_2879_cast_fp16)[name = string("op_2895")]; tensor var_2896 = const()[name = string("op_2896"), val = tensor([1, 8, 1, 64])]; tensor var_2897 = reshape(shape = var_2896, x = var_2895)[name = string("op_2897")]; string var_2904_pad_type_0 = const()[name = string("op_2904_pad_type_0"), val = string("valid")]; tensor var_2904_strides_0 = const()[name = string("op_2904_strides_0"), val = tensor([1, 1])]; tensor var_2904_pad_0 = const()[name = string("op_2904_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2904_dilations_0 = const()[name = string("op_2904_dilations_0"), val = tensor([1, 1])]; int32 var_2904_groups_0 = const()[name = string("op_2904_groups_0"), val = int32(1)]; tensor var_2904 = conv(dilations = var_2904_dilations_0, groups = var_2904_groups_0, pad = var_2904_pad_0, pad_type = var_2904_pad_type_0, strides = var_2904_strides_0, weight = model_model_layers_12_self_attn_v_proj_weight_palettized, x = var_2879_cast_fp16)[name = string("op_2904")]; tensor var_2905 = const()[name = string("op_2905"), val = tensor([1, 8, 1, 64])]; tensor var_2906 = reshape(shape = var_2905, x = var_2904)[name = string("op_2906")]; tensor x1_49_begin_0 = const()[name = string("x1_49_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_49_end_0 = const()[name = string("x1_49_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_49_end_mask_0 = const()[name = string("x1_49_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_49 = slice_by_index(begin = x1_49_begin_0, end = x1_49_end_0, end_mask = x1_49_end_mask_0, x = var_2888)[name = string("x1_49")]; tensor x2_49_begin_0 = const()[name = string("x2_49_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_49_end_0 = const()[name = string("x2_49_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_49_end_mask_0 = const()[name = string("x2_49_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_49 = slice_by_index(begin = x2_49_begin_0, end = x2_49_end_0, end_mask = x2_49_end_mask_0, x = var_2888)[name = string("x2_49")]; tensor var_2920_cast_fp16 = mul(x = x1_49, y = cos_3_cast_fp16)[name = string("op_2920_cast_fp16")]; tensor var_2921_cast_fp16 = mul(x = x2_49, y = sin_3_cast_fp16)[name = string("op_2921_cast_fp16")]; tensor var_2922_cast_fp16 = sub(x = var_2920_cast_fp16, y = var_2921_cast_fp16)[name = string("op_2922_cast_fp16")]; tensor var_2923_cast_fp16 = mul(x = x2_49, y = cos_3_cast_fp16)[name = string("op_2923_cast_fp16")]; tensor var_2924_cast_fp16 = mul(x = x1_49, y = sin_3_cast_fp16)[name = string("op_2924_cast_fp16")]; tensor var_2925_cast_fp16 = add(x = var_2923_cast_fp16, y = var_2924_cast_fp16)[name = string("op_2925_cast_fp16")]; bool rotated_49_interleave_0 = const()[name = string("rotated_49_interleave_0"), val = bool(false)]; tensor rotated_49_cast_fp16 = concat(axis = var_80, interleave = rotated_49_interleave_0, values = (var_2922_cast_fp16, var_2925_cast_fp16))[name = string("rotated_49_cast_fp16")]; tensor x1_51_begin_0 = const()[name = string("x1_51_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_51_end_0 = const()[name = string("x1_51_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_51_end_mask_0 = const()[name = string("x1_51_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_51 = slice_by_index(begin = x1_51_begin_0, end = x1_51_end_0, end_mask = x1_51_end_mask_0, x = var_2897)[name = string("x1_51")]; tensor x2_51_begin_0 = const()[name = string("x2_51_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_51_end_0 = const()[name = string("x2_51_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_51_end_mask_0 = const()[name = string("x2_51_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_51 = slice_by_index(begin = x2_51_begin_0, end = x2_51_end_0, end_mask = x2_51_end_mask_0, x = var_2897)[name = string("x2_51")]; tensor var_2941_cast_fp16 = mul(x = x1_51, y = cos_3_cast_fp16)[name = string("op_2941_cast_fp16")]; tensor var_2942_cast_fp16 = mul(x = x2_51, y = sin_3_cast_fp16)[name = string("op_2942_cast_fp16")]; tensor var_2943_cast_fp16 = sub(x = var_2941_cast_fp16, y = var_2942_cast_fp16)[name = string("op_2943_cast_fp16")]; tensor var_2944_cast_fp16 = mul(x = x2_51, y = cos_3_cast_fp16)[name = string("op_2944_cast_fp16")]; tensor var_2945_cast_fp16 = mul(x = x1_51, y = sin_3_cast_fp16)[name = string("op_2945_cast_fp16")]; tensor var_2946_cast_fp16 = add(x = var_2944_cast_fp16, y = var_2945_cast_fp16)[name = string("op_2946_cast_fp16")]; bool rotated_51_interleave_0 = const()[name = string("rotated_51_interleave_0"), val = bool(false)]; tensor rotated_51_cast_fp16 = concat(axis = var_80, interleave = rotated_51_interleave_0, values = (var_2943_cast_fp16, var_2946_cast_fp16))[name = string("rotated_51_cast_fp16")]; tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; int32 concat_98_axis_0 = const()[name = string("concat_98_axis_0"), val = int32(0)]; bool concat_98_interleave_0 = const()[name = string("concat_98_interleave_0"), val = bool(false)]; tensor concat_98 = concat(axis = concat_98_axis_0, interleave = concat_98_interleave_0, values = (expand_dims_144, expand_dims_145, current_pos, expand_dims_147))[name = string("concat_98")]; tensor concat_99_values1_0 = const()[name = string("concat_99_values1_0"), val = tensor([0])]; tensor concat_99_values3_0 = const()[name = string("concat_99_values3_0"), val = tensor([0])]; int32 concat_99_axis_0 = const()[name = string("concat_99_axis_0"), val = int32(0)]; bool concat_99_interleave_0 = const()[name = string("concat_99_interleave_0"), val = bool(false)]; tensor concat_99 = concat(axis = concat_99_axis_0, interleave = concat_99_interleave_0, values = (expand_dims_148, concat_99_values1_0, var_587, concat_99_values3_0))[name = string("concat_99")]; tensor model_model_kv_cache_0_internal_tensor_assign_25_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_98, begin_mask = model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0, end = concat_99, end_mask = model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_25_stride_0, update = rotated_51_cast_fp16, x = coreml_update_state_55)[name = string("model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_88_write_state")]; tensor coreml_update_state_56 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_88")]; tensor expand_dims_150 = const()[name = string("expand_dims_150"), val = tensor([28])]; tensor expand_dims_151 = const()[name = string("expand_dims_151"), val = tensor([0])]; tensor expand_dims_153 = const()[name = string("expand_dims_153"), val = tensor([0])]; tensor expand_dims_154 = const()[name = string("expand_dims_154"), val = tensor([29])]; int32 concat_102_axis_0 = const()[name = string("concat_102_axis_0"), val = int32(0)]; bool concat_102_interleave_0 = const()[name = string("concat_102_interleave_0"), val = bool(false)]; tensor concat_102 = concat(axis = concat_102_axis_0, interleave = concat_102_interleave_0, values = (expand_dims_150, expand_dims_151, current_pos, expand_dims_153))[name = string("concat_102")]; tensor concat_103_values1_0 = const()[name = string("concat_103_values1_0"), val = tensor([0])]; tensor concat_103_values3_0 = const()[name = string("concat_103_values3_0"), val = tensor([0])]; int32 concat_103_axis_0 = const()[name = string("concat_103_axis_0"), val = int32(0)]; bool concat_103_interleave_0 = const()[name = string("concat_103_interleave_0"), val = bool(false)]; tensor concat_103 = concat(axis = concat_103_axis_0, interleave = concat_103_interleave_0, values = (expand_dims_154, concat_103_values1_0, var_587, concat_103_values3_0))[name = string("concat_103")]; tensor model_model_kv_cache_0_internal_tensor_assign_26_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_102, begin_mask = model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0, end = concat_103, end_mask = model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_26_stride_0, update = var_2906, x = coreml_update_state_56)[name = string("model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_89_write_state")]; tensor coreml_update_state_57 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_89")]; tensor var_2966_begin_0 = const()[name = string("op_2966_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_2966_end_0 = const()[name = string("op_2966_end_0"), val = tensor([13, 8, 4096, 64])]; tensor var_2966_end_mask_0 = const()[name = string("op_2966_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2966_cast_fp16 = slice_by_index(begin = var_2966_begin_0, end = var_2966_end_0, end_mask = var_2966_end_mask_0, x = coreml_update_state_57)[name = string("op_2966_cast_fp16")]; tensor K_layer_cache_25_axes_0 = const()[name = string("K_layer_cache_25_axes_0"), val = tensor([0])]; tensor K_layer_cache_25_cast_fp16 = squeeze(axes = K_layer_cache_25_axes_0, x = var_2966_cast_fp16)[name = string("K_layer_cache_25_cast_fp16")]; tensor var_2968_begin_0 = const()[name = string("op_2968_begin_0"), val = tensor([28, 0, 0, 0])]; tensor var_2968_end_0 = const()[name = string("op_2968_end_0"), val = tensor([29, 8, 4096, 64])]; tensor var_2968_end_mask_0 = const()[name = string("op_2968_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2968_cast_fp16 = slice_by_index(begin = var_2968_begin_0, end = var_2968_end_0, end_mask = var_2968_end_mask_0, x = coreml_update_state_57)[name = string("op_2968_cast_fp16")]; tensor V_layer_cache_25_axes_0 = const()[name = string("V_layer_cache_25_axes_0"), val = tensor([0])]; tensor V_layer_cache_25_cast_fp16 = squeeze(axes = V_layer_cache_25_axes_0, x = var_2968_cast_fp16)[name = string("V_layer_cache_25_cast_fp16")]; tensor x_347_axes_0 = const()[name = string("x_347_axes_0"), val = tensor([1])]; tensor x_347_cast_fp16 = expand_dims(axes = x_347_axes_0, x = K_layer_cache_25_cast_fp16)[name = string("x_347_cast_fp16")]; tensor var_2977 = const()[name = string("op_2977"), val = tensor([1, 4, 1, 1])]; tensor x_349_cast_fp16 = tile(reps = var_2977, x = x_347_cast_fp16)[name = string("x_349_cast_fp16")]; tensor var_2981 = const()[name = string("op_2981"), val = tensor([1, -1, 4096, 64])]; tensor key_states_51_cast_fp16 = reshape(shape = var_2981, x = x_349_cast_fp16)[name = string("key_states_51_cast_fp16")]; tensor x_353_axes_0 = const()[name = string("x_353_axes_0"), val = tensor([1])]; tensor x_353_cast_fp16 = expand_dims(axes = x_353_axes_0, x = V_layer_cache_25_cast_fp16)[name = string("x_353_cast_fp16")]; tensor var_2984 = const()[name = string("op_2984"), val = tensor([1, 4, 1, 1])]; tensor x_355_cast_fp16 = tile(reps = var_2984, x = x_353_cast_fp16)[name = string("x_355_cast_fp16")]; tensor var_2988 = const()[name = string("op_2988"), val = tensor([1, -1, 4096, 64])]; tensor value_states_51_cast_fp16 = reshape(shape = var_2988, x = x_355_cast_fp16)[name = string("value_states_51_cast_fp16")]; bool var_2991_transpose_x_1 = const()[name = string("op_2991_transpose_x_1"), val = bool(false)]; bool var_2991_transpose_y_1 = const()[name = string("op_2991_transpose_y_1"), val = bool(true)]; tensor var_2991_cast_fp16 = matmul(transpose_x = var_2991_transpose_x_1, transpose_y = var_2991_transpose_y_1, x = rotated_49_cast_fp16, y = key_states_51_cast_fp16)[name = string("op_2991_cast_fp16")]; fp16 var_2992_to_fp16 = const()[name = string("op_2992_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_49_cast_fp16 = mul(x = var_2991_cast_fp16, y = var_2992_to_fp16)[name = string("attn_weights_49_cast_fp16")]; tensor x_357_cast_fp16 = add(x = attn_weights_49_cast_fp16, y = causal_mask)[name = string("x_357_cast_fp16")]; tensor reduce_max_12_axes_0 = const()[name = string("reduce_max_12_axes_0"), val = tensor([-1])]; bool reduce_max_12_keep_dims_0 = const()[name = string("reduce_max_12_keep_dims_0"), val = bool(true)]; tensor reduce_max_12_cast_fp16 = reduce_max(axes = reduce_max_12_axes_0, keep_dims = reduce_max_12_keep_dims_0, x = x_357_cast_fp16)[name = string("reduce_max_12_cast_fp16")]; tensor x_359_cast_fp16 = sub(x = x_357_cast_fp16, y = reduce_max_12_cast_fp16)[name = string("x_359_cast_fp16")]; tensor exp_x_25_cast_fp16 = exp(x = x_359_cast_fp16)[name = string("exp_x_25_cast_fp16")]; tensor var_3003_axes_0 = const()[name = string("op_3003_axes_0"), val = tensor([-1])]; bool var_3003_keep_dims_0 = const()[name = string("op_3003_keep_dims_0"), val = bool(true)]; tensor var_3003_cast_fp16 = reduce_sum(axes = var_3003_axes_0, keep_dims = var_3003_keep_dims_0, x = exp_x_25_cast_fp16)[name = string("op_3003_cast_fp16")]; tensor attn_weights_51_cast_fp16 = real_div(x = exp_x_25_cast_fp16, y = var_3003_cast_fp16)[name = string("attn_weights_51_cast_fp16")]; bool attn_output_73_transpose_x_0 = const()[name = string("attn_output_73_transpose_x_0"), val = bool(false)]; bool attn_output_73_transpose_y_0 = const()[name = string("attn_output_73_transpose_y_0"), val = bool(false)]; tensor attn_output_73_cast_fp16 = matmul(transpose_x = attn_output_73_transpose_x_0, transpose_y = attn_output_73_transpose_y_0, x = attn_weights_51_cast_fp16, y = value_states_51_cast_fp16)[name = string("attn_output_73_cast_fp16")]; tensor var_3006_perm_0 = const()[name = string("op_3006_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3008 = const()[name = string("op_3008"), val = tensor([1, 1, 2048])]; tensor var_3006_cast_fp16 = transpose(perm = var_3006_perm_0, x = attn_output_73_cast_fp16)[name = string("transpose_14")]; tensor input_173_cast_fp16 = reshape(shape = var_3008, x = var_3006_cast_fp16)[name = string("input_173_cast_fp16")]; tensor model_model_layers_12_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481844544))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483941760))))[name = string("model_model_layers_12_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_12_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_12_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_173_cast_fp16)[name = string("linear_12_cast_fp16")]; tensor hidden_states_101_cast_fp16 = add(x = hidden_states_97_cast_fp16, y = linear_12_cast_fp16)[name = string("hidden_states_101_cast_fp16")]; fp16 const_204_promoted_to_fp16 = const()[name = string("const_204_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3014_cast_fp16 = mul(x = hidden_states_101_cast_fp16, y = const_204_promoted_to_fp16)[name = string("op_3014_cast_fp16")]; bool input_175_interleave_0 = const()[name = string("input_175_interleave_0"), val = bool(false)]; tensor input_175_cast_fp16 = concat(axis = var_80, interleave = input_175_interleave_0, values = (hidden_states_101_cast_fp16, var_3014_cast_fp16))[name = string("input_175_cast_fp16")]; tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_74_to_fp16, x = input_175_cast_fp16)[name = string("normed_101_cast_fp16")]; tensor normed_103_begin_0 = const()[name = string("normed_103_begin_0"), val = tensor([0, 0, 0])]; tensor normed_103_end_0 = const()[name = string("normed_103_end_0"), val = tensor([1, 1, 2048])]; tensor normed_103_end_mask_0 = const()[name = string("normed_103_end_mask_0"), val = tensor([true, true, false])]; tensor normed_103_cast_fp16 = slice_by_index(begin = normed_103_begin_0, end = normed_103_end_0, end_mask = normed_103_end_mask_0, x = normed_101_cast_fp16)[name = string("normed_103_cast_fp16")]; tensor const_207_promoted_to_fp16 = const()[name = string("const_207_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483950016)))]; tensor x_361_cast_fp16 = mul(x = normed_103_cast_fp16, y = const_207_promoted_to_fp16)[name = string("x_361_cast_fp16")]; tensor var_3032 = const()[name = string("op_3032"), val = tensor([0, 2, 1])]; tensor input_177_axes_0 = const()[name = string("input_177_axes_0"), val = tensor([2])]; tensor var_3033 = transpose(perm = var_3032, x = x_361_cast_fp16)[name = string("transpose_13")]; tensor input_177 = expand_dims(axes = input_177_axes_0, x = var_3033)[name = string("input_177")]; string input_179_pad_type_0 = const()[name = string("input_179_pad_type_0"), val = string("valid")]; tensor input_179_strides_0 = const()[name = string("input_179_strides_0"), val = tensor([1, 1])]; tensor input_179_pad_0 = const()[name = string("input_179_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_179_dilations_0 = const()[name = string("input_179_dilations_0"), val = tensor([1, 1])]; int32 input_179_groups_0 = const()[name = string("input_179_groups_0"), val = int32(1)]; tensor input_179 = conv(dilations = input_179_dilations_0, groups = input_179_groups_0, pad = input_179_pad_0, pad_type = input_179_pad_type_0, strides = input_179_strides_0, weight = model_model_layers_12_mlp_gate_proj_weight_palettized, x = input_177)[name = string("input_179")]; string up_states_25_pad_type_0 = const()[name = string("up_states_25_pad_type_0"), val = string("valid")]; tensor up_states_25_strides_0 = const()[name = string("up_states_25_strides_0"), val = tensor([1, 1])]; tensor up_states_25_pad_0 = const()[name = string("up_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_25_dilations_0 = const()[name = string("up_states_25_dilations_0"), val = tensor([1, 1])]; int32 up_states_25_groups_0 = const()[name = string("up_states_25_groups_0"), val = int32(1)]; tensor up_states_25 = conv(dilations = up_states_25_dilations_0, groups = up_states_25_groups_0, pad = up_states_25_pad_0, pad_type = up_states_25_pad_type_0, strides = up_states_25_strides_0, weight = model_model_layers_12_mlp_up_proj_weight_palettized, x = input_177)[name = string("up_states_25")]; tensor gate_states_25 = silu(x = input_179)[name = string("gate_states_25")]; tensor input_181 = mul(x = gate_states_25, y = up_states_25)[name = string("input_181")]; string hidden_states_103_pad_type_0 = const()[name = string("hidden_states_103_pad_type_0"), val = string("valid")]; tensor hidden_states_103_strides_0 = const()[name = string("hidden_states_103_strides_0"), val = tensor([1, 1])]; tensor hidden_states_103_pad_0 = const()[name = string("hidden_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_103_dilations_0 = const()[name = string("hidden_states_103_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_103_groups_0 = const()[name = string("hidden_states_103_groups_0"), val = int32(1)]; tensor hidden_states_103 = conv(dilations = hidden_states_103_dilations_0, groups = hidden_states_103_groups_0, pad = hidden_states_103_pad_0, pad_type = hidden_states_103_pad_type_0, strides = hidden_states_103_strides_0, weight = model_model_layers_12_mlp_down_proj_weight_palettized, x = input_181)[name = string("hidden_states_103")]; tensor var_3055_axes_0 = const()[name = string("op_3055_axes_0"), val = tensor([2])]; tensor var_3055 = squeeze(axes = var_3055_axes_0, x = hidden_states_103)[name = string("op_3055")]; tensor var_3056 = const()[name = string("op_3056"), val = tensor([0, 2, 1])]; tensor var_3057 = transpose(perm = var_3056, x = var_3055)[name = string("transpose_12")]; tensor hidden_states_105_cast_fp16 = add(x = hidden_states_101_cast_fp16, y = var_3057)[name = string("hidden_states_105_cast_fp16")]; fp16 const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3060_cast_fp16 = mul(x = hidden_states_105_cast_fp16, y = const_208_promoted_to_fp16)[name = string("op_3060_cast_fp16")]; bool input_183_interleave_0 = const()[name = string("input_183_interleave_0"), val = bool(false)]; tensor input_183_cast_fp16 = concat(axis = var_80, interleave = input_183_interleave_0, values = (hidden_states_105_cast_fp16, var_3060_cast_fp16))[name = string("input_183_cast_fp16")]; tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_74_to_fp16, x = input_183_cast_fp16)[name = string("normed_105_cast_fp16")]; tensor normed_107_begin_0 = const()[name = string("normed_107_begin_0"), val = tensor([0, 0, 0])]; tensor normed_107_end_0 = const()[name = string("normed_107_end_0"), val = tensor([1, 1, 2048])]; tensor normed_107_end_mask_0 = const()[name = string("normed_107_end_mask_0"), val = tensor([true, true, false])]; tensor normed_107_cast_fp16 = slice_by_index(begin = normed_107_begin_0, end = normed_107_end_0, end_mask = normed_107_end_mask_0, x = normed_105_cast_fp16)[name = string("normed_107_cast_fp16")]; tensor const_211_promoted_to_fp16 = const()[name = string("const_211_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483954176)))]; tensor hidden_states_107_cast_fp16 = mul(x = normed_107_cast_fp16, y = const_211_promoted_to_fp16)[name = string("hidden_states_107_cast_fp16")]; tensor var_3074 = const()[name = string("op_3074"), val = tensor([0, 2, 1])]; tensor var_3076_axes_0 = const()[name = string("op_3076_axes_0"), val = tensor([2])]; tensor var_3075_cast_fp16 = transpose(perm = var_3074, x = hidden_states_107_cast_fp16)[name = string("transpose_11")]; tensor var_3076_cast_fp16 = expand_dims(axes = var_3076_axes_0, x = var_3075_cast_fp16)[name = string("op_3076_cast_fp16")]; string var_3083_pad_type_0 = const()[name = string("op_3083_pad_type_0"), val = string("valid")]; tensor var_3083_strides_0 = const()[name = string("op_3083_strides_0"), val = tensor([1, 1])]; tensor var_3083_pad_0 = const()[name = string("op_3083_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3083_dilations_0 = const()[name = string("op_3083_dilations_0"), val = tensor([1, 1])]; int32 var_3083_groups_0 = const()[name = string("op_3083_groups_0"), val = int32(1)]; tensor var_3083 = conv(dilations = var_3083_dilations_0, groups = var_3083_groups_0, pad = var_3083_pad_0, pad_type = var_3083_pad_type_0, strides = var_3083_strides_0, weight = model_model_layers_13_self_attn_q_proj_weight_palettized, x = var_3076_cast_fp16)[name = string("op_3083")]; tensor var_3084 = const()[name = string("op_3084"), val = tensor([1, 32, 1, 64])]; tensor var_3085 = reshape(shape = var_3084, x = var_3083)[name = string("op_3085")]; string var_3092_pad_type_0 = const()[name = string("op_3092_pad_type_0"), val = string("valid")]; tensor var_3092_strides_0 = const()[name = string("op_3092_strides_0"), val = tensor([1, 1])]; tensor var_3092_pad_0 = const()[name = string("op_3092_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3092_dilations_0 = const()[name = string("op_3092_dilations_0"), val = tensor([1, 1])]; int32 var_3092_groups_0 = const()[name = string("op_3092_groups_0"), val = int32(1)]; tensor var_3092 = conv(dilations = var_3092_dilations_0, groups = var_3092_groups_0, pad = var_3092_pad_0, pad_type = var_3092_pad_type_0, strides = var_3092_strides_0, weight = model_model_layers_13_self_attn_k_proj_weight_palettized, x = var_3076_cast_fp16)[name = string("op_3092")]; tensor var_3093 = const()[name = string("op_3093"), val = tensor([1, 8, 1, 64])]; tensor var_3094 = reshape(shape = var_3093, x = var_3092)[name = string("op_3094")]; string var_3101_pad_type_0 = const()[name = string("op_3101_pad_type_0"), val = string("valid")]; tensor var_3101_strides_0 = const()[name = string("op_3101_strides_0"), val = tensor([1, 1])]; tensor var_3101_pad_0 = const()[name = string("op_3101_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3101_dilations_0 = const()[name = string("op_3101_dilations_0"), val = tensor([1, 1])]; int32 var_3101_groups_0 = const()[name = string("op_3101_groups_0"), val = int32(1)]; tensor var_3101 = conv(dilations = var_3101_dilations_0, groups = var_3101_groups_0, pad = var_3101_pad_0, pad_type = var_3101_pad_type_0, strides = var_3101_strides_0, weight = model_model_layers_13_self_attn_v_proj_weight_palettized, x = var_3076_cast_fp16)[name = string("op_3101")]; tensor var_3102 = const()[name = string("op_3102"), val = tensor([1, 8, 1, 64])]; tensor var_3103 = reshape(shape = var_3102, x = var_3101)[name = string("op_3103")]; tensor x1_53_begin_0 = const()[name = string("x1_53_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_53_end_0 = const()[name = string("x1_53_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_53_end_mask_0 = const()[name = string("x1_53_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_53 = slice_by_index(begin = x1_53_begin_0, end = x1_53_end_0, end_mask = x1_53_end_mask_0, x = var_3085)[name = string("x1_53")]; tensor x2_53_begin_0 = const()[name = string("x2_53_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_53_end_0 = const()[name = string("x2_53_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_53_end_mask_0 = const()[name = string("x2_53_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_53 = slice_by_index(begin = x2_53_begin_0, end = x2_53_end_0, end_mask = x2_53_end_mask_0, x = var_3085)[name = string("x2_53")]; tensor var_3117_cast_fp16 = mul(x = x1_53, y = cos_3_cast_fp16)[name = string("op_3117_cast_fp16")]; tensor var_3118_cast_fp16 = mul(x = x2_53, y = sin_3_cast_fp16)[name = string("op_3118_cast_fp16")]; tensor var_3119_cast_fp16 = sub(x = var_3117_cast_fp16, y = var_3118_cast_fp16)[name = string("op_3119_cast_fp16")]; tensor var_3120_cast_fp16 = mul(x = x2_53, y = cos_3_cast_fp16)[name = string("op_3120_cast_fp16")]; tensor var_3121_cast_fp16 = mul(x = x1_53, y = sin_3_cast_fp16)[name = string("op_3121_cast_fp16")]; tensor var_3122_cast_fp16 = add(x = var_3120_cast_fp16, y = var_3121_cast_fp16)[name = string("op_3122_cast_fp16")]; bool rotated_53_interleave_0 = const()[name = string("rotated_53_interleave_0"), val = bool(false)]; tensor rotated_53_cast_fp16 = concat(axis = var_80, interleave = rotated_53_interleave_0, values = (var_3119_cast_fp16, var_3122_cast_fp16))[name = string("rotated_53_cast_fp16")]; tensor x1_55_begin_0 = const()[name = string("x1_55_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_55_end_0 = const()[name = string("x1_55_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_55_end_mask_0 = const()[name = string("x1_55_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_55 = slice_by_index(begin = x1_55_begin_0, end = x1_55_end_0, end_mask = x1_55_end_mask_0, x = var_3094)[name = string("x1_55")]; tensor x2_55_begin_0 = const()[name = string("x2_55_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_55_end_0 = const()[name = string("x2_55_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_55_end_mask_0 = const()[name = string("x2_55_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_55 = slice_by_index(begin = x2_55_begin_0, end = x2_55_end_0, end_mask = x2_55_end_mask_0, x = var_3094)[name = string("x2_55")]; tensor var_3138_cast_fp16 = mul(x = x1_55, y = cos_3_cast_fp16)[name = string("op_3138_cast_fp16")]; tensor var_3139_cast_fp16 = mul(x = x2_55, y = sin_3_cast_fp16)[name = string("op_3139_cast_fp16")]; tensor var_3140_cast_fp16 = sub(x = var_3138_cast_fp16, y = var_3139_cast_fp16)[name = string("op_3140_cast_fp16")]; tensor var_3141_cast_fp16 = mul(x = x2_55, y = cos_3_cast_fp16)[name = string("op_3141_cast_fp16")]; tensor var_3142_cast_fp16 = mul(x = x1_55, y = sin_3_cast_fp16)[name = string("op_3142_cast_fp16")]; tensor var_3143_cast_fp16 = add(x = var_3141_cast_fp16, y = var_3142_cast_fp16)[name = string("op_3143_cast_fp16")]; bool rotated_55_interleave_0 = const()[name = string("rotated_55_interleave_0"), val = bool(false)]; tensor rotated_55_cast_fp16 = concat(axis = var_80, interleave = rotated_55_interleave_0, values = (var_3140_cast_fp16, var_3143_cast_fp16))[name = string("rotated_55_cast_fp16")]; tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; int32 concat_106_axis_0 = const()[name = string("concat_106_axis_0"), val = int32(0)]; bool concat_106_interleave_0 = const()[name = string("concat_106_interleave_0"), val = bool(false)]; tensor concat_106 = concat(axis = concat_106_axis_0, interleave = concat_106_interleave_0, values = (expand_dims_156, expand_dims_157, current_pos, expand_dims_159))[name = string("concat_106")]; tensor concat_107_values1_0 = const()[name = string("concat_107_values1_0"), val = tensor([0])]; tensor concat_107_values3_0 = const()[name = string("concat_107_values3_0"), val = tensor([0])]; int32 concat_107_axis_0 = const()[name = string("concat_107_axis_0"), val = int32(0)]; bool concat_107_interleave_0 = const()[name = string("concat_107_interleave_0"), val = bool(false)]; tensor concat_107 = concat(axis = concat_107_axis_0, interleave = concat_107_interleave_0, values = (expand_dims_160, concat_107_values1_0, var_587, concat_107_values3_0))[name = string("concat_107")]; tensor model_model_kv_cache_0_internal_tensor_assign_27_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_106, begin_mask = model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0, end = concat_107, end_mask = model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_27_stride_0, update = rotated_55_cast_fp16, x = coreml_update_state_57)[name = string("model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_90_write_state")]; tensor coreml_update_state_58 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_90")]; tensor expand_dims_162 = const()[name = string("expand_dims_162"), val = tensor([29])]; tensor expand_dims_163 = const()[name = string("expand_dims_163"), val = tensor([0])]; tensor expand_dims_165 = const()[name = string("expand_dims_165"), val = tensor([0])]; tensor expand_dims_166 = const()[name = string("expand_dims_166"), val = tensor([30])]; int32 concat_110_axis_0 = const()[name = string("concat_110_axis_0"), val = int32(0)]; bool concat_110_interleave_0 = const()[name = string("concat_110_interleave_0"), val = bool(false)]; tensor concat_110 = concat(axis = concat_110_axis_0, interleave = concat_110_interleave_0, values = (expand_dims_162, expand_dims_163, current_pos, expand_dims_165))[name = string("concat_110")]; tensor concat_111_values1_0 = const()[name = string("concat_111_values1_0"), val = tensor([0])]; tensor concat_111_values3_0 = const()[name = string("concat_111_values3_0"), val = tensor([0])]; int32 concat_111_axis_0 = const()[name = string("concat_111_axis_0"), val = int32(0)]; bool concat_111_interleave_0 = const()[name = string("concat_111_interleave_0"), val = bool(false)]; tensor concat_111 = concat(axis = concat_111_axis_0, interleave = concat_111_interleave_0, values = (expand_dims_166, concat_111_values1_0, var_587, concat_111_values3_0))[name = string("concat_111")]; tensor model_model_kv_cache_0_internal_tensor_assign_28_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_110, begin_mask = model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0, end = concat_111, end_mask = model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_28_stride_0, update = var_3103, x = coreml_update_state_58)[name = string("model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_91_write_state")]; tensor coreml_update_state_59 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_91")]; tensor var_3163_begin_0 = const()[name = string("op_3163_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_3163_end_0 = const()[name = string("op_3163_end_0"), val = tensor([14, 8, 4096, 64])]; tensor var_3163_end_mask_0 = const()[name = string("op_3163_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3163_cast_fp16 = slice_by_index(begin = var_3163_begin_0, end = var_3163_end_0, end_mask = var_3163_end_mask_0, x = coreml_update_state_59)[name = string("op_3163_cast_fp16")]; tensor K_layer_cache_27_axes_0 = const()[name = string("K_layer_cache_27_axes_0"), val = tensor([0])]; tensor K_layer_cache_27_cast_fp16 = squeeze(axes = K_layer_cache_27_axes_0, x = var_3163_cast_fp16)[name = string("K_layer_cache_27_cast_fp16")]; tensor var_3165_begin_0 = const()[name = string("op_3165_begin_0"), val = tensor([29, 0, 0, 0])]; tensor var_3165_end_0 = const()[name = string("op_3165_end_0"), val = tensor([30, 8, 4096, 64])]; tensor var_3165_end_mask_0 = const()[name = string("op_3165_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3165_cast_fp16 = slice_by_index(begin = var_3165_begin_0, end = var_3165_end_0, end_mask = var_3165_end_mask_0, x = coreml_update_state_59)[name = string("op_3165_cast_fp16")]; tensor V_layer_cache_27_axes_0 = const()[name = string("V_layer_cache_27_axes_0"), val = tensor([0])]; tensor V_layer_cache_27_cast_fp16 = squeeze(axes = V_layer_cache_27_axes_0, x = var_3165_cast_fp16)[name = string("V_layer_cache_27_cast_fp16")]; tensor x_375_axes_0 = const()[name = string("x_375_axes_0"), val = tensor([1])]; tensor x_375_cast_fp16 = expand_dims(axes = x_375_axes_0, x = K_layer_cache_27_cast_fp16)[name = string("x_375_cast_fp16")]; tensor var_3174 = const()[name = string("op_3174"), val = tensor([1, 4, 1, 1])]; tensor x_377_cast_fp16 = tile(reps = var_3174, x = x_375_cast_fp16)[name = string("x_377_cast_fp16")]; tensor var_3178 = const()[name = string("op_3178"), val = tensor([1, -1, 4096, 64])]; tensor key_states_55_cast_fp16 = reshape(shape = var_3178, x = x_377_cast_fp16)[name = string("key_states_55_cast_fp16")]; tensor x_381_axes_0 = const()[name = string("x_381_axes_0"), val = tensor([1])]; tensor x_381_cast_fp16 = expand_dims(axes = x_381_axes_0, x = V_layer_cache_27_cast_fp16)[name = string("x_381_cast_fp16")]; tensor var_3181 = const()[name = string("op_3181"), val = tensor([1, 4, 1, 1])]; tensor x_383_cast_fp16 = tile(reps = var_3181, x = x_381_cast_fp16)[name = string("x_383_cast_fp16")]; tensor var_3185 = const()[name = string("op_3185"), val = tensor([1, -1, 4096, 64])]; tensor value_states_55_cast_fp16 = reshape(shape = var_3185, x = x_383_cast_fp16)[name = string("value_states_55_cast_fp16")]; bool var_3188_transpose_x_1 = const()[name = string("op_3188_transpose_x_1"), val = bool(false)]; bool var_3188_transpose_y_1 = const()[name = string("op_3188_transpose_y_1"), val = bool(true)]; tensor var_3188_cast_fp16 = matmul(transpose_x = var_3188_transpose_x_1, transpose_y = var_3188_transpose_y_1, x = rotated_53_cast_fp16, y = key_states_55_cast_fp16)[name = string("op_3188_cast_fp16")]; fp16 var_3189_to_fp16 = const()[name = string("op_3189_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_53_cast_fp16 = mul(x = var_3188_cast_fp16, y = var_3189_to_fp16)[name = string("attn_weights_53_cast_fp16")]; tensor x_385_cast_fp16 = add(x = attn_weights_53_cast_fp16, y = causal_mask)[name = string("x_385_cast_fp16")]; tensor reduce_max_13_axes_0 = const()[name = string("reduce_max_13_axes_0"), val = tensor([-1])]; bool reduce_max_13_keep_dims_0 = const()[name = string("reduce_max_13_keep_dims_0"), val = bool(true)]; tensor reduce_max_13_cast_fp16 = reduce_max(axes = reduce_max_13_axes_0, keep_dims = reduce_max_13_keep_dims_0, x = x_385_cast_fp16)[name = string("reduce_max_13_cast_fp16")]; tensor x_387_cast_fp16 = sub(x = x_385_cast_fp16, y = reduce_max_13_cast_fp16)[name = string("x_387_cast_fp16")]; tensor exp_x_27_cast_fp16 = exp(x = x_387_cast_fp16)[name = string("exp_x_27_cast_fp16")]; tensor var_3200_axes_0 = const()[name = string("op_3200_axes_0"), val = tensor([-1])]; bool var_3200_keep_dims_0 = const()[name = string("op_3200_keep_dims_0"), val = bool(true)]; tensor var_3200_cast_fp16 = reduce_sum(axes = var_3200_axes_0, keep_dims = var_3200_keep_dims_0, x = exp_x_27_cast_fp16)[name = string("op_3200_cast_fp16")]; tensor attn_weights_55_cast_fp16 = real_div(x = exp_x_27_cast_fp16, y = var_3200_cast_fp16)[name = string("attn_weights_55_cast_fp16")]; bool attn_output_79_transpose_x_0 = const()[name = string("attn_output_79_transpose_x_0"), val = bool(false)]; bool attn_output_79_transpose_y_0 = const()[name = string("attn_output_79_transpose_y_0"), val = bool(false)]; tensor attn_output_79_cast_fp16 = matmul(transpose_x = attn_output_79_transpose_x_0, transpose_y = attn_output_79_transpose_y_0, x = attn_weights_55_cast_fp16, y = value_states_55_cast_fp16)[name = string("attn_output_79_cast_fp16")]; tensor var_3203_perm_0 = const()[name = string("op_3203_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3205 = const()[name = string("op_3205"), val = tensor([1, 1, 2048])]; tensor var_3203_cast_fp16 = transpose(perm = var_3203_perm_0, x = attn_output_79_cast_fp16)[name = string("transpose_10")]; tensor input_187_cast_fp16 = reshape(shape = var_3205, x = var_3203_cast_fp16)[name = string("input_187_cast_fp16")]; tensor model_model_layers_13_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483958336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(486055552))))[name = string("model_model_layers_13_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_13_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_13_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_187_cast_fp16)[name = string("linear_13_cast_fp16")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = linear_13_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; fp16 const_220_promoted_to_fp16 = const()[name = string("const_220_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3211_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_220_promoted_to_fp16)[name = string("op_3211_cast_fp16")]; bool input_189_interleave_0 = const()[name = string("input_189_interleave_0"), val = bool(false)]; tensor input_189_cast_fp16 = concat(axis = var_80, interleave = input_189_interleave_0, values = (hidden_states_109_cast_fp16, var_3211_cast_fp16))[name = string("input_189_cast_fp16")]; tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_74_to_fp16, x = input_189_cast_fp16)[name = string("normed_109_cast_fp16")]; tensor normed_111_begin_0 = const()[name = string("normed_111_begin_0"), val = tensor([0, 0, 0])]; tensor normed_111_end_0 = const()[name = string("normed_111_end_0"), val = tensor([1, 1, 2048])]; tensor normed_111_end_mask_0 = const()[name = string("normed_111_end_mask_0"), val = tensor([true, true, false])]; tensor normed_111_cast_fp16 = slice_by_index(begin = normed_111_begin_0, end = normed_111_end_0, end_mask = normed_111_end_mask_0, x = normed_109_cast_fp16)[name = string("normed_111_cast_fp16")]; tensor const_223_promoted_to_fp16 = const()[name = string("const_223_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(486063808)))]; tensor x_389_cast_fp16 = mul(x = normed_111_cast_fp16, y = const_223_promoted_to_fp16)[name = string("x_389_cast_fp16")]; tensor var_3229 = const()[name = string("op_3229"), val = tensor([0, 2, 1])]; tensor input_191_axes_0 = const()[name = string("input_191_axes_0"), val = tensor([2])]; tensor var_3230 = transpose(perm = var_3229, x = x_389_cast_fp16)[name = string("transpose_9")]; tensor input_191 = expand_dims(axes = input_191_axes_0, x = var_3230)[name = string("input_191")]; string input_193_pad_type_0 = const()[name = string("input_193_pad_type_0"), val = string("valid")]; tensor input_193_strides_0 = const()[name = string("input_193_strides_0"), val = tensor([1, 1])]; tensor input_193_pad_0 = const()[name = string("input_193_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_193_dilations_0 = const()[name = string("input_193_dilations_0"), val = tensor([1, 1])]; int32 input_193_groups_0 = const()[name = string("input_193_groups_0"), val = int32(1)]; tensor input_193 = conv(dilations = input_193_dilations_0, groups = input_193_groups_0, pad = input_193_pad_0, pad_type = input_193_pad_type_0, strides = input_193_strides_0, weight = model_model_layers_13_mlp_gate_proj_weight_palettized, x = input_191)[name = string("input_193")]; string up_states_27_pad_type_0 = const()[name = string("up_states_27_pad_type_0"), val = string("valid")]; tensor up_states_27_strides_0 = const()[name = string("up_states_27_strides_0"), val = tensor([1, 1])]; tensor up_states_27_pad_0 = const()[name = string("up_states_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_27_dilations_0 = const()[name = string("up_states_27_dilations_0"), val = tensor([1, 1])]; int32 up_states_27_groups_0 = const()[name = string("up_states_27_groups_0"), val = int32(1)]; tensor up_states_27 = conv(dilations = up_states_27_dilations_0, groups = up_states_27_groups_0, pad = up_states_27_pad_0, pad_type = up_states_27_pad_type_0, strides = up_states_27_strides_0, weight = model_model_layers_13_mlp_up_proj_weight_palettized, x = input_191)[name = string("up_states_27")]; tensor gate_states_27 = silu(x = input_193)[name = string("gate_states_27")]; tensor input_195 = mul(x = gate_states_27, y = up_states_27)[name = string("input_195")]; string hidden_states_111_pad_type_0 = const()[name = string("hidden_states_111_pad_type_0"), val = string("valid")]; tensor hidden_states_111_strides_0 = const()[name = string("hidden_states_111_strides_0"), val = tensor([1, 1])]; tensor hidden_states_111_pad_0 = const()[name = string("hidden_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_111_dilations_0 = const()[name = string("hidden_states_111_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_111_groups_0 = const()[name = string("hidden_states_111_groups_0"), val = int32(1)]; tensor hidden_states_111 = conv(dilations = hidden_states_111_dilations_0, groups = hidden_states_111_groups_0, pad = hidden_states_111_pad_0, pad_type = hidden_states_111_pad_type_0, strides = hidden_states_111_strides_0, weight = model_model_layers_13_mlp_down_proj_weight_palettized, x = input_195)[name = string("hidden_states_111")]; tensor var_3252_axes_0 = const()[name = string("op_3252_axes_0"), val = tensor([2])]; tensor var_3252 = squeeze(axes = var_3252_axes_0, x = hidden_states_111)[name = string("op_3252")]; tensor var_3253 = const()[name = string("op_3253"), val = tensor([0, 2, 1])]; tensor var_3254 = transpose(perm = var_3253, x = var_3252)[name = string("transpose_8")]; tensor hidden_states_113_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = var_3254)[name = string("hidden_states_113_cast_fp16")]; fp16 const_224_promoted_to_fp16 = const()[name = string("const_224_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3257_cast_fp16 = mul(x = hidden_states_113_cast_fp16, y = const_224_promoted_to_fp16)[name = string("op_3257_cast_fp16")]; bool input_197_interleave_0 = const()[name = string("input_197_interleave_0"), val = bool(false)]; tensor input_197_cast_fp16 = concat(axis = var_80, interleave = input_197_interleave_0, values = (hidden_states_113_cast_fp16, var_3257_cast_fp16))[name = string("input_197_cast_fp16")]; tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_74_to_fp16, x = input_197_cast_fp16)[name = string("normed_113_cast_fp16")]; tensor normed_115_begin_0 = const()[name = string("normed_115_begin_0"), val = tensor([0, 0, 0])]; tensor normed_115_end_0 = const()[name = string("normed_115_end_0"), val = tensor([1, 1, 2048])]; tensor normed_115_end_mask_0 = const()[name = string("normed_115_end_mask_0"), val = tensor([true, true, false])]; tensor normed_115_cast_fp16 = slice_by_index(begin = normed_115_begin_0, end = normed_115_end_0, end_mask = normed_115_end_mask_0, x = normed_113_cast_fp16)[name = string("normed_115_cast_fp16")]; tensor const_227_promoted_to_fp16 = const()[name = string("const_227_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(486067968)))]; tensor hidden_states_115_cast_fp16 = mul(x = normed_115_cast_fp16, y = const_227_promoted_to_fp16)[name = string("hidden_states_115_cast_fp16")]; tensor var_3271 = const()[name = string("op_3271"), val = tensor([0, 2, 1])]; tensor var_3273_axes_0 = const()[name = string("op_3273_axes_0"), val = tensor([2])]; tensor var_3272_cast_fp16 = transpose(perm = var_3271, x = hidden_states_115_cast_fp16)[name = string("transpose_7")]; tensor var_3273_cast_fp16 = expand_dims(axes = var_3273_axes_0, x = var_3272_cast_fp16)[name = string("op_3273_cast_fp16")]; string var_3280_pad_type_0 = const()[name = string("op_3280_pad_type_0"), val = string("valid")]; tensor var_3280_strides_0 = const()[name = string("op_3280_strides_0"), val = tensor([1, 1])]; tensor var_3280_pad_0 = const()[name = string("op_3280_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3280_dilations_0 = const()[name = string("op_3280_dilations_0"), val = tensor([1, 1])]; int32 var_3280_groups_0 = const()[name = string("op_3280_groups_0"), val = int32(1)]; tensor var_3280 = conv(dilations = var_3280_dilations_0, groups = var_3280_groups_0, pad = var_3280_pad_0, pad_type = var_3280_pad_type_0, strides = var_3280_strides_0, weight = model_model_layers_14_self_attn_q_proj_weight_palettized, x = var_3273_cast_fp16)[name = string("op_3280")]; tensor var_3281 = const()[name = string("op_3281"), val = tensor([1, 32, 1, 64])]; tensor var_3282 = reshape(shape = var_3281, x = var_3280)[name = string("op_3282")]; string var_3289_pad_type_0 = const()[name = string("op_3289_pad_type_0"), val = string("valid")]; tensor var_3289_strides_0 = const()[name = string("op_3289_strides_0"), val = tensor([1, 1])]; tensor var_3289_pad_0 = const()[name = string("op_3289_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3289_dilations_0 = const()[name = string("op_3289_dilations_0"), val = tensor([1, 1])]; int32 var_3289_groups_0 = const()[name = string("op_3289_groups_0"), val = int32(1)]; tensor var_3289 = conv(dilations = var_3289_dilations_0, groups = var_3289_groups_0, pad = var_3289_pad_0, pad_type = var_3289_pad_type_0, strides = var_3289_strides_0, weight = model_model_layers_14_self_attn_k_proj_weight_palettized, x = var_3273_cast_fp16)[name = string("op_3289")]; tensor var_3290 = const()[name = string("op_3290"), val = tensor([1, 8, 1, 64])]; tensor var_3291 = reshape(shape = var_3290, x = var_3289)[name = string("op_3291")]; string var_3298_pad_type_0 = const()[name = string("op_3298_pad_type_0"), val = string("valid")]; tensor var_3298_strides_0 = const()[name = string("op_3298_strides_0"), val = tensor([1, 1])]; tensor var_3298_pad_0 = const()[name = string("op_3298_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3298_dilations_0 = const()[name = string("op_3298_dilations_0"), val = tensor([1, 1])]; int32 var_3298_groups_0 = const()[name = string("op_3298_groups_0"), val = int32(1)]; tensor var_3298 = conv(dilations = var_3298_dilations_0, groups = var_3298_groups_0, pad = var_3298_pad_0, pad_type = var_3298_pad_type_0, strides = var_3298_strides_0, weight = model_model_layers_14_self_attn_v_proj_weight_palettized, x = var_3273_cast_fp16)[name = string("op_3298")]; tensor var_3299 = const()[name = string("op_3299"), val = tensor([1, 8, 1, 64])]; tensor var_3300 = reshape(shape = var_3299, x = var_3298)[name = string("op_3300")]; tensor x1_57_begin_0 = const()[name = string("x1_57_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_57_end_0 = const()[name = string("x1_57_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_57_end_mask_0 = const()[name = string("x1_57_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_57 = slice_by_index(begin = x1_57_begin_0, end = x1_57_end_0, end_mask = x1_57_end_mask_0, x = var_3282)[name = string("x1_57")]; tensor x2_57_begin_0 = const()[name = string("x2_57_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_57_end_0 = const()[name = string("x2_57_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_57_end_mask_0 = const()[name = string("x2_57_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_57 = slice_by_index(begin = x2_57_begin_0, end = x2_57_end_0, end_mask = x2_57_end_mask_0, x = var_3282)[name = string("x2_57")]; tensor var_3314_cast_fp16 = mul(x = x1_57, y = cos_3_cast_fp16)[name = string("op_3314_cast_fp16")]; tensor var_3315_cast_fp16 = mul(x = x2_57, y = sin_3_cast_fp16)[name = string("op_3315_cast_fp16")]; tensor var_3316_cast_fp16 = sub(x = var_3314_cast_fp16, y = var_3315_cast_fp16)[name = string("op_3316_cast_fp16")]; tensor var_3317_cast_fp16 = mul(x = x2_57, y = cos_3_cast_fp16)[name = string("op_3317_cast_fp16")]; tensor var_3318_cast_fp16 = mul(x = x1_57, y = sin_3_cast_fp16)[name = string("op_3318_cast_fp16")]; tensor var_3319_cast_fp16 = add(x = var_3317_cast_fp16, y = var_3318_cast_fp16)[name = string("op_3319_cast_fp16")]; bool rotated_57_interleave_0 = const()[name = string("rotated_57_interleave_0"), val = bool(false)]; tensor rotated_57_cast_fp16 = concat(axis = var_80, interleave = rotated_57_interleave_0, values = (var_3316_cast_fp16, var_3319_cast_fp16))[name = string("rotated_57_cast_fp16")]; tensor x1_59_begin_0 = const()[name = string("x1_59_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_59_end_0 = const()[name = string("x1_59_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_59_end_mask_0 = const()[name = string("x1_59_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_59 = slice_by_index(begin = x1_59_begin_0, end = x1_59_end_0, end_mask = x1_59_end_mask_0, x = var_3291)[name = string("x1_59")]; tensor x2_59_begin_0 = const()[name = string("x2_59_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_59_end_0 = const()[name = string("x2_59_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_59_end_mask_0 = const()[name = string("x2_59_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_59 = slice_by_index(begin = x2_59_begin_0, end = x2_59_end_0, end_mask = x2_59_end_mask_0, x = var_3291)[name = string("x2_59")]; tensor var_3335_cast_fp16 = mul(x = x1_59, y = cos_3_cast_fp16)[name = string("op_3335_cast_fp16")]; tensor var_3336_cast_fp16 = mul(x = x2_59, y = sin_3_cast_fp16)[name = string("op_3336_cast_fp16")]; tensor var_3337_cast_fp16 = sub(x = var_3335_cast_fp16, y = var_3336_cast_fp16)[name = string("op_3337_cast_fp16")]; tensor var_3338_cast_fp16 = mul(x = x2_59, y = cos_3_cast_fp16)[name = string("op_3338_cast_fp16")]; tensor var_3339_cast_fp16 = mul(x = x1_59, y = sin_3_cast_fp16)[name = string("op_3339_cast_fp16")]; tensor var_3340_cast_fp16 = add(x = var_3338_cast_fp16, y = var_3339_cast_fp16)[name = string("op_3340_cast_fp16")]; bool rotated_59_interleave_0 = const()[name = string("rotated_59_interleave_0"), val = bool(false)]; tensor rotated_59_cast_fp16 = concat(axis = var_80, interleave = rotated_59_interleave_0, values = (var_3337_cast_fp16, var_3340_cast_fp16))[name = string("rotated_59_cast_fp16")]; tensor expand_dims_168 = const()[name = string("expand_dims_168"), val = tensor([14])]; tensor expand_dims_169 = const()[name = string("expand_dims_169"), val = tensor([0])]; tensor expand_dims_171 = const()[name = string("expand_dims_171"), val = tensor([0])]; tensor expand_dims_172 = const()[name = string("expand_dims_172"), val = tensor([15])]; int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_168, expand_dims_169, current_pos, expand_dims_171))[name = string("concat_114")]; tensor concat_115_values1_0 = const()[name = string("concat_115_values1_0"), val = tensor([0])]; tensor concat_115_values3_0 = const()[name = string("concat_115_values3_0"), val = tensor([0])]; int32 concat_115_axis_0 = const()[name = string("concat_115_axis_0"), val = int32(0)]; bool concat_115_interleave_0 = const()[name = string("concat_115_interleave_0"), val = bool(false)]; tensor concat_115 = concat(axis = concat_115_axis_0, interleave = concat_115_interleave_0, values = (expand_dims_172, concat_115_values1_0, var_587, concat_115_values3_0))[name = string("concat_115")]; tensor model_model_kv_cache_0_internal_tensor_assign_29_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_cast_fp16 = slice_update(begin = concat_114, begin_mask = model_model_kv_cache_0_internal_tensor_assign_29_begin_mask_0, end = concat_115, end_mask = model_model_kv_cache_0_internal_tensor_assign_29_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_29_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_29_stride_0, update = rotated_59_cast_fp16, x = coreml_update_state_59)[name = string("model_model_kv_cache_0_internal_tensor_assign_29_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_29_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_92_write_state")]; tensor coreml_update_state_60 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_92")]; tensor expand_dims_174 = const()[name = string("expand_dims_174"), val = tensor([30])]; tensor expand_dims_175 = const()[name = string("expand_dims_175"), val = tensor([0])]; tensor expand_dims_177 = const()[name = string("expand_dims_177"), val = tensor([0])]; tensor expand_dims_178 = const()[name = string("expand_dims_178"), val = tensor([31])]; int32 concat_118_axis_0 = const()[name = string("concat_118_axis_0"), val = int32(0)]; bool concat_118_interleave_0 = const()[name = string("concat_118_interleave_0"), val = bool(false)]; tensor concat_118 = concat(axis = concat_118_axis_0, interleave = concat_118_interleave_0, values = (expand_dims_174, expand_dims_175, current_pos, expand_dims_177))[name = string("concat_118")]; tensor concat_119_values1_0 = const()[name = string("concat_119_values1_0"), val = tensor([0])]; tensor concat_119_values3_0 = const()[name = string("concat_119_values3_0"), val = tensor([0])]; int32 concat_119_axis_0 = const()[name = string("concat_119_axis_0"), val = int32(0)]; bool concat_119_interleave_0 = const()[name = string("concat_119_interleave_0"), val = bool(false)]; tensor concat_119 = concat(axis = concat_119_axis_0, interleave = concat_119_interleave_0, values = (expand_dims_178, concat_119_values1_0, var_587, concat_119_values3_0))[name = string("concat_119")]; tensor model_model_kv_cache_0_internal_tensor_assign_30_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_cast_fp16 = slice_update(begin = concat_118, begin_mask = model_model_kv_cache_0_internal_tensor_assign_30_begin_mask_0, end = concat_119, end_mask = model_model_kv_cache_0_internal_tensor_assign_30_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_30_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_30_stride_0, update = var_3300, x = coreml_update_state_60)[name = string("model_model_kv_cache_0_internal_tensor_assign_30_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_30_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_93_write_state")]; tensor coreml_update_state_61 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_93")]; tensor var_3360_begin_0 = const()[name = string("op_3360_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_3360_end_0 = const()[name = string("op_3360_end_0"), val = tensor([15, 8, 4096, 64])]; tensor var_3360_end_mask_0 = const()[name = string("op_3360_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3360_cast_fp16 = slice_by_index(begin = var_3360_begin_0, end = var_3360_end_0, end_mask = var_3360_end_mask_0, x = coreml_update_state_61)[name = string("op_3360_cast_fp16")]; tensor K_layer_cache_29_axes_0 = const()[name = string("K_layer_cache_29_axes_0"), val = tensor([0])]; tensor K_layer_cache_29_cast_fp16 = squeeze(axes = K_layer_cache_29_axes_0, x = var_3360_cast_fp16)[name = string("K_layer_cache_29_cast_fp16")]; tensor var_3362_begin_0 = const()[name = string("op_3362_begin_0"), val = tensor([30, 0, 0, 0])]; tensor var_3362_end_0 = const()[name = string("op_3362_end_0"), val = tensor([31, 8, 4096, 64])]; tensor var_3362_end_mask_0 = const()[name = string("op_3362_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3362_cast_fp16 = slice_by_index(begin = var_3362_begin_0, end = var_3362_end_0, end_mask = var_3362_end_mask_0, x = coreml_update_state_61)[name = string("op_3362_cast_fp16")]; tensor V_layer_cache_29_axes_0 = const()[name = string("V_layer_cache_29_axes_0"), val = tensor([0])]; tensor V_layer_cache_29_cast_fp16 = squeeze(axes = V_layer_cache_29_axes_0, x = var_3362_cast_fp16)[name = string("V_layer_cache_29_cast_fp16")]; tensor x_403_axes_0 = const()[name = string("x_403_axes_0"), val = tensor([1])]; tensor x_403_cast_fp16 = expand_dims(axes = x_403_axes_0, x = K_layer_cache_29_cast_fp16)[name = string("x_403_cast_fp16")]; tensor var_3371 = const()[name = string("op_3371"), val = tensor([1, 4, 1, 1])]; tensor x_405_cast_fp16 = tile(reps = var_3371, x = x_403_cast_fp16)[name = string("x_405_cast_fp16")]; tensor var_3375 = const()[name = string("op_3375"), val = tensor([1, -1, 4096, 64])]; tensor key_states_59_cast_fp16 = reshape(shape = var_3375, x = x_405_cast_fp16)[name = string("key_states_59_cast_fp16")]; tensor x_409_axes_0 = const()[name = string("x_409_axes_0"), val = tensor([1])]; tensor x_409_cast_fp16 = expand_dims(axes = x_409_axes_0, x = V_layer_cache_29_cast_fp16)[name = string("x_409_cast_fp16")]; tensor var_3378 = const()[name = string("op_3378"), val = tensor([1, 4, 1, 1])]; tensor x_411_cast_fp16 = tile(reps = var_3378, x = x_409_cast_fp16)[name = string("x_411_cast_fp16")]; tensor var_3382 = const()[name = string("op_3382"), val = tensor([1, -1, 4096, 64])]; tensor value_states_59_cast_fp16 = reshape(shape = var_3382, x = x_411_cast_fp16)[name = string("value_states_59_cast_fp16")]; bool var_3385_transpose_x_1 = const()[name = string("op_3385_transpose_x_1"), val = bool(false)]; bool var_3385_transpose_y_1 = const()[name = string("op_3385_transpose_y_1"), val = bool(true)]; tensor var_3385_cast_fp16 = matmul(transpose_x = var_3385_transpose_x_1, transpose_y = var_3385_transpose_y_1, x = rotated_57_cast_fp16, y = key_states_59_cast_fp16)[name = string("op_3385_cast_fp16")]; fp16 var_3386_to_fp16 = const()[name = string("op_3386_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_57_cast_fp16 = mul(x = var_3385_cast_fp16, y = var_3386_to_fp16)[name = string("attn_weights_57_cast_fp16")]; tensor x_413_cast_fp16 = add(x = attn_weights_57_cast_fp16, y = causal_mask)[name = string("x_413_cast_fp16")]; tensor reduce_max_14_axes_0 = const()[name = string("reduce_max_14_axes_0"), val = tensor([-1])]; bool reduce_max_14_keep_dims_0 = const()[name = string("reduce_max_14_keep_dims_0"), val = bool(true)]; tensor reduce_max_14_cast_fp16 = reduce_max(axes = reduce_max_14_axes_0, keep_dims = reduce_max_14_keep_dims_0, x = x_413_cast_fp16)[name = string("reduce_max_14_cast_fp16")]; tensor x_415_cast_fp16 = sub(x = x_413_cast_fp16, y = reduce_max_14_cast_fp16)[name = string("x_415_cast_fp16")]; tensor exp_x_29_cast_fp16 = exp(x = x_415_cast_fp16)[name = string("exp_x_29_cast_fp16")]; tensor var_3397_axes_0 = const()[name = string("op_3397_axes_0"), val = tensor([-1])]; bool var_3397_keep_dims_0 = const()[name = string("op_3397_keep_dims_0"), val = bool(true)]; tensor var_3397_cast_fp16 = reduce_sum(axes = var_3397_axes_0, keep_dims = var_3397_keep_dims_0, x = exp_x_29_cast_fp16)[name = string("op_3397_cast_fp16")]; tensor attn_weights_59_cast_fp16 = real_div(x = exp_x_29_cast_fp16, y = var_3397_cast_fp16)[name = string("attn_weights_59_cast_fp16")]; bool attn_output_85_transpose_x_0 = const()[name = string("attn_output_85_transpose_x_0"), val = bool(false)]; bool attn_output_85_transpose_y_0 = const()[name = string("attn_output_85_transpose_y_0"), val = bool(false)]; tensor attn_output_85_cast_fp16 = matmul(transpose_x = attn_output_85_transpose_x_0, transpose_y = attn_output_85_transpose_y_0, x = attn_weights_59_cast_fp16, y = value_states_59_cast_fp16)[name = string("attn_output_85_cast_fp16")]; tensor var_3400_perm_0 = const()[name = string("op_3400_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3402 = const()[name = string("op_3402"), val = tensor([1, 1, 2048])]; tensor var_3400_cast_fp16 = transpose(perm = var_3400_perm_0, x = attn_output_85_cast_fp16)[name = string("transpose_6")]; tensor input_201_cast_fp16 = reshape(shape = var_3402, x = var_3400_cast_fp16)[name = string("input_201_cast_fp16")]; tensor model_model_layers_14_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(486072128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488169344))))[name = string("model_model_layers_14_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_14_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_14_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_201_cast_fp16)[name = string("linear_14_cast_fp16")]; tensor hidden_states_117_cast_fp16 = add(x = hidden_states_113_cast_fp16, y = linear_14_cast_fp16)[name = string("hidden_states_117_cast_fp16")]; fp16 const_236_promoted_to_fp16 = const()[name = string("const_236_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3408_cast_fp16 = mul(x = hidden_states_117_cast_fp16, y = const_236_promoted_to_fp16)[name = string("op_3408_cast_fp16")]; bool input_203_interleave_0 = const()[name = string("input_203_interleave_0"), val = bool(false)]; tensor input_203_cast_fp16 = concat(axis = var_80, interleave = input_203_interleave_0, values = (hidden_states_117_cast_fp16, var_3408_cast_fp16))[name = string("input_203_cast_fp16")]; tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_74_to_fp16, x = input_203_cast_fp16)[name = string("normed_117_cast_fp16")]; tensor normed_119_begin_0 = const()[name = string("normed_119_begin_0"), val = tensor([0, 0, 0])]; tensor normed_119_end_0 = const()[name = string("normed_119_end_0"), val = tensor([1, 1, 2048])]; tensor normed_119_end_mask_0 = const()[name = string("normed_119_end_mask_0"), val = tensor([true, true, false])]; tensor normed_119_cast_fp16 = slice_by_index(begin = normed_119_begin_0, end = normed_119_end_0, end_mask = normed_119_end_mask_0, x = normed_117_cast_fp16)[name = string("normed_119_cast_fp16")]; tensor const_239_promoted_to_fp16 = const()[name = string("const_239_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488177600)))]; tensor x_417_cast_fp16 = mul(x = normed_119_cast_fp16, y = const_239_promoted_to_fp16)[name = string("x_417_cast_fp16")]; tensor var_3426 = const()[name = string("op_3426"), val = tensor([0, 2, 1])]; tensor input_205_axes_0 = const()[name = string("input_205_axes_0"), val = tensor([2])]; tensor var_3427 = transpose(perm = var_3426, x = x_417_cast_fp16)[name = string("transpose_5")]; tensor input_205 = expand_dims(axes = input_205_axes_0, x = var_3427)[name = string("input_205")]; string input_207_pad_type_0 = const()[name = string("input_207_pad_type_0"), val = string("valid")]; tensor input_207_strides_0 = const()[name = string("input_207_strides_0"), val = tensor([1, 1])]; tensor input_207_pad_0 = const()[name = string("input_207_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_207_dilations_0 = const()[name = string("input_207_dilations_0"), val = tensor([1, 1])]; int32 input_207_groups_0 = const()[name = string("input_207_groups_0"), val = int32(1)]; tensor input_207 = conv(dilations = input_207_dilations_0, groups = input_207_groups_0, pad = input_207_pad_0, pad_type = input_207_pad_type_0, strides = input_207_strides_0, weight = model_model_layers_14_mlp_gate_proj_weight_palettized, x = input_205)[name = string("input_207")]; string up_states_29_pad_type_0 = const()[name = string("up_states_29_pad_type_0"), val = string("valid")]; tensor up_states_29_strides_0 = const()[name = string("up_states_29_strides_0"), val = tensor([1, 1])]; tensor up_states_29_pad_0 = const()[name = string("up_states_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_29_dilations_0 = const()[name = string("up_states_29_dilations_0"), val = tensor([1, 1])]; int32 up_states_29_groups_0 = const()[name = string("up_states_29_groups_0"), val = int32(1)]; tensor up_states_29 = conv(dilations = up_states_29_dilations_0, groups = up_states_29_groups_0, pad = up_states_29_pad_0, pad_type = up_states_29_pad_type_0, strides = up_states_29_strides_0, weight = model_model_layers_14_mlp_up_proj_weight_palettized, x = input_205)[name = string("up_states_29")]; tensor gate_states_29 = silu(x = input_207)[name = string("gate_states_29")]; tensor input_209 = mul(x = gate_states_29, y = up_states_29)[name = string("input_209")]; string hidden_states_119_pad_type_0 = const()[name = string("hidden_states_119_pad_type_0"), val = string("valid")]; tensor hidden_states_119_strides_0 = const()[name = string("hidden_states_119_strides_0"), val = tensor([1, 1])]; tensor hidden_states_119_pad_0 = const()[name = string("hidden_states_119_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_119_dilations_0 = const()[name = string("hidden_states_119_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_119_groups_0 = const()[name = string("hidden_states_119_groups_0"), val = int32(1)]; tensor hidden_states_119 = conv(dilations = hidden_states_119_dilations_0, groups = hidden_states_119_groups_0, pad = hidden_states_119_pad_0, pad_type = hidden_states_119_pad_type_0, strides = hidden_states_119_strides_0, weight = model_model_layers_14_mlp_down_proj_weight_palettized, x = input_209)[name = string("hidden_states_119")]; tensor var_3449_axes_0 = const()[name = string("op_3449_axes_0"), val = tensor([2])]; tensor var_3449 = squeeze(axes = var_3449_axes_0, x = hidden_states_119)[name = string("op_3449")]; tensor var_3450 = const()[name = string("op_3450"), val = tensor([0, 2, 1])]; tensor var_3451 = transpose(perm = var_3450, x = var_3449)[name = string("transpose_4")]; tensor hidden_states_121_cast_fp16 = add(x = hidden_states_117_cast_fp16, y = var_3451)[name = string("hidden_states_121_cast_fp16")]; fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3454_cast_fp16 = mul(x = hidden_states_121_cast_fp16, y = const_240_promoted_to_fp16)[name = string("op_3454_cast_fp16")]; bool input_211_interleave_0 = const()[name = string("input_211_interleave_0"), val = bool(false)]; tensor input_211_cast_fp16 = concat(axis = var_80, interleave = input_211_interleave_0, values = (hidden_states_121_cast_fp16, var_3454_cast_fp16))[name = string("input_211_cast_fp16")]; tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_74_to_fp16, x = input_211_cast_fp16)[name = string("normed_121_cast_fp16")]; tensor normed_123_begin_0 = const()[name = string("normed_123_begin_0"), val = tensor([0, 0, 0])]; tensor normed_123_end_0 = const()[name = string("normed_123_end_0"), val = tensor([1, 1, 2048])]; tensor normed_123_end_mask_0 = const()[name = string("normed_123_end_mask_0"), val = tensor([true, true, false])]; tensor normed_123_cast_fp16 = slice_by_index(begin = normed_123_begin_0, end = normed_123_end_0, end_mask = normed_123_end_mask_0, x = normed_121_cast_fp16)[name = string("normed_123_cast_fp16")]; tensor const_243_promoted_to_fp16 = const()[name = string("const_243_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488181760)))]; tensor hidden_states_123_cast_fp16 = mul(x = normed_123_cast_fp16, y = const_243_promoted_to_fp16)[name = string("hidden_states_123_cast_fp16")]; tensor var_3468 = const()[name = string("op_3468"), val = tensor([0, 2, 1])]; tensor var_3470_axes_0 = const()[name = string("op_3470_axes_0"), val = tensor([2])]; tensor var_3469_cast_fp16 = transpose(perm = var_3468, x = hidden_states_123_cast_fp16)[name = string("transpose_3")]; tensor var_3470_cast_fp16 = expand_dims(axes = var_3470_axes_0, x = var_3469_cast_fp16)[name = string("op_3470_cast_fp16")]; string var_3477_pad_type_0 = const()[name = string("op_3477_pad_type_0"), val = string("valid")]; tensor var_3477_strides_0 = const()[name = string("op_3477_strides_0"), val = tensor([1, 1])]; tensor var_3477_pad_0 = const()[name = string("op_3477_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3477_dilations_0 = const()[name = string("op_3477_dilations_0"), val = tensor([1, 1])]; int32 var_3477_groups_0 = const()[name = string("op_3477_groups_0"), val = int32(1)]; tensor var_3477 = conv(dilations = var_3477_dilations_0, groups = var_3477_groups_0, pad = var_3477_pad_0, pad_type = var_3477_pad_type_0, strides = var_3477_strides_0, weight = model_model_layers_15_self_attn_q_proj_weight_palettized, x = var_3470_cast_fp16)[name = string("op_3477")]; tensor var_3478 = const()[name = string("op_3478"), val = tensor([1, 32, 1, 64])]; tensor var_3479 = reshape(shape = var_3478, x = var_3477)[name = string("op_3479")]; string var_3486_pad_type_0 = const()[name = string("op_3486_pad_type_0"), val = string("valid")]; tensor var_3486_strides_0 = const()[name = string("op_3486_strides_0"), val = tensor([1, 1])]; tensor var_3486_pad_0 = const()[name = string("op_3486_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3486_dilations_0 = const()[name = string("op_3486_dilations_0"), val = tensor([1, 1])]; int32 var_3486_groups_0 = const()[name = string("op_3486_groups_0"), val = int32(1)]; tensor var_3486 = conv(dilations = var_3486_dilations_0, groups = var_3486_groups_0, pad = var_3486_pad_0, pad_type = var_3486_pad_type_0, strides = var_3486_strides_0, weight = model_model_layers_15_self_attn_k_proj_weight_palettized, x = var_3470_cast_fp16)[name = string("op_3486")]; tensor var_3487 = const()[name = string("op_3487"), val = tensor([1, 8, 1, 64])]; tensor var_3488 = reshape(shape = var_3487, x = var_3486)[name = string("op_3488")]; string var_3495_pad_type_0 = const()[name = string("op_3495_pad_type_0"), val = string("valid")]; tensor var_3495_strides_0 = const()[name = string("op_3495_strides_0"), val = tensor([1, 1])]; tensor var_3495_pad_0 = const()[name = string("op_3495_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3495_dilations_0 = const()[name = string("op_3495_dilations_0"), val = tensor([1, 1])]; int32 var_3495_groups_0 = const()[name = string("op_3495_groups_0"), val = int32(1)]; tensor var_3495 = conv(dilations = var_3495_dilations_0, groups = var_3495_groups_0, pad = var_3495_pad_0, pad_type = var_3495_pad_type_0, strides = var_3495_strides_0, weight = model_model_layers_15_self_attn_v_proj_weight_palettized, x = var_3470_cast_fp16)[name = string("op_3495")]; tensor var_3496 = const()[name = string("op_3496"), val = tensor([1, 8, 1, 64])]; tensor var_3497 = reshape(shape = var_3496, x = var_3495)[name = string("op_3497")]; tensor x1_61_begin_0 = const()[name = string("x1_61_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_61_end_0 = const()[name = string("x1_61_end_0"), val = tensor([1, 32, 1, 32])]; tensor x1_61_end_mask_0 = const()[name = string("x1_61_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1_61 = slice_by_index(begin = x1_61_begin_0, end = x1_61_end_0, end_mask = x1_61_end_mask_0, x = var_3479)[name = string("x1_61")]; tensor x2_61_begin_0 = const()[name = string("x2_61_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_61_end_0 = const()[name = string("x2_61_end_0"), val = tensor([1, 32, 1, 64])]; tensor x2_61_end_mask_0 = const()[name = string("x2_61_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_61 = slice_by_index(begin = x2_61_begin_0, end = x2_61_end_0, end_mask = x2_61_end_mask_0, x = var_3479)[name = string("x2_61")]; tensor var_3511_cast_fp16 = mul(x = x1_61, y = cos_3_cast_fp16)[name = string("op_3511_cast_fp16")]; tensor var_3512_cast_fp16 = mul(x = x2_61, y = sin_3_cast_fp16)[name = string("op_3512_cast_fp16")]; tensor var_3513_cast_fp16 = sub(x = var_3511_cast_fp16, y = var_3512_cast_fp16)[name = string("op_3513_cast_fp16")]; tensor var_3514_cast_fp16 = mul(x = x2_61, y = cos_3_cast_fp16)[name = string("op_3514_cast_fp16")]; tensor var_3515_cast_fp16 = mul(x = x1_61, y = sin_3_cast_fp16)[name = string("op_3515_cast_fp16")]; tensor var_3516_cast_fp16 = add(x = var_3514_cast_fp16, y = var_3515_cast_fp16)[name = string("op_3516_cast_fp16")]; bool rotated_61_interleave_0 = const()[name = string("rotated_61_interleave_0"), val = bool(false)]; tensor rotated_61_cast_fp16 = concat(axis = var_80, interleave = rotated_61_interleave_0, values = (var_3513_cast_fp16, var_3516_cast_fp16))[name = string("rotated_61_cast_fp16")]; tensor x1_begin_0 = const()[name = string("x1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_end_0 = const()[name = string("x1_end_0"), val = tensor([1, 8, 1, 32])]; tensor x1_end_mask_0 = const()[name = string("x1_end_mask_0"), val = tensor([true, true, true, false])]; tensor x1 = slice_by_index(begin = x1_begin_0, end = x1_end_0, end_mask = x1_end_mask_0, x = var_3488)[name = string("x1")]; tensor x2_begin_0 = const()[name = string("x2_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_end_0 = const()[name = string("x2_end_0"), val = tensor([1, 8, 1, 64])]; tensor x2_end_mask_0 = const()[name = string("x2_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2 = slice_by_index(begin = x2_begin_0, end = x2_end_0, end_mask = x2_end_mask_0, x = var_3488)[name = string("x2")]; tensor var_3532_cast_fp16 = mul(x = x1, y = cos_3_cast_fp16)[name = string("op_3532_cast_fp16")]; tensor var_3533_cast_fp16 = mul(x = x2, y = sin_3_cast_fp16)[name = string("op_3533_cast_fp16")]; tensor var_3534_cast_fp16 = sub(x = var_3532_cast_fp16, y = var_3533_cast_fp16)[name = string("op_3534_cast_fp16")]; tensor var_3535_cast_fp16 = mul(x = x2, y = cos_3_cast_fp16)[name = string("op_3535_cast_fp16")]; tensor var_3536_cast_fp16 = mul(x = x1, y = sin_3_cast_fp16)[name = string("op_3536_cast_fp16")]; tensor var_3537_cast_fp16 = add(x = var_3535_cast_fp16, y = var_3536_cast_fp16)[name = string("op_3537_cast_fp16")]; bool rotated_interleave_0 = const()[name = string("rotated_interleave_0"), val = bool(false)]; tensor rotated_cast_fp16 = concat(axis = var_80, interleave = rotated_interleave_0, values = (var_3534_cast_fp16, var_3537_cast_fp16))[name = string("rotated_cast_fp16")]; tensor expand_dims_180 = const()[name = string("expand_dims_180"), val = tensor([15])]; tensor expand_dims_181 = const()[name = string("expand_dims_181"), val = tensor([0])]; tensor expand_dims_183 = const()[name = string("expand_dims_183"), val = tensor([0])]; tensor expand_dims_184 = const()[name = string("expand_dims_184"), val = tensor([16])]; int32 concat_122_axis_0 = const()[name = string("concat_122_axis_0"), val = int32(0)]; bool concat_122_interleave_0 = const()[name = string("concat_122_interleave_0"), val = bool(false)]; tensor concat_122 = concat(axis = concat_122_axis_0, interleave = concat_122_interleave_0, values = (expand_dims_180, expand_dims_181, current_pos, expand_dims_183))[name = string("concat_122")]; tensor concat_123_values1_0 = const()[name = string("concat_123_values1_0"), val = tensor([0])]; tensor concat_123_values3_0 = const()[name = string("concat_123_values3_0"), val = tensor([0])]; int32 concat_123_axis_0 = const()[name = string("concat_123_axis_0"), val = int32(0)]; bool concat_123_interleave_0 = const()[name = string("concat_123_interleave_0"), val = bool(false)]; tensor concat_123 = concat(axis = concat_123_axis_0, interleave = concat_123_interleave_0, values = (expand_dims_184, concat_123_values1_0, var_587, concat_123_values3_0))[name = string("concat_123")]; tensor model_model_kv_cache_0_internal_tensor_assign_31_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_cast_fp16 = slice_update(begin = concat_122, begin_mask = model_model_kv_cache_0_internal_tensor_assign_31_begin_mask_0, end = concat_123, end_mask = model_model_kv_cache_0_internal_tensor_assign_31_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_31_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_31_stride_0, update = rotated_cast_fp16, x = coreml_update_state_61)[name = string("model_model_kv_cache_0_internal_tensor_assign_31_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_31_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_94_write_state")]; tensor coreml_update_state_62 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_94")]; tensor expand_dims_186 = const()[name = string("expand_dims_186"), val = tensor([31])]; tensor expand_dims_187 = const()[name = string("expand_dims_187"), val = tensor([0])]; tensor expand_dims_189 = const()[name = string("expand_dims_189"), val = tensor([0])]; tensor expand_dims_190 = const()[name = string("expand_dims_190"), val = tensor([32])]; int32 concat_126_axis_0 = const()[name = string("concat_126_axis_0"), val = int32(0)]; bool concat_126_interleave_0 = const()[name = string("concat_126_interleave_0"), val = bool(false)]; tensor concat_126 = concat(axis = concat_126_axis_0, interleave = concat_126_interleave_0, values = (expand_dims_186, expand_dims_187, current_pos, expand_dims_189))[name = string("concat_126")]; tensor concat_127_values1_0 = const()[name = string("concat_127_values1_0"), val = tensor([0])]; tensor concat_127_values3_0 = const()[name = string("concat_127_values3_0"), val = tensor([0])]; int32 concat_127_axis_0 = const()[name = string("concat_127_axis_0"), val = int32(0)]; bool concat_127_interleave_0 = const()[name = string("concat_127_interleave_0"), val = bool(false)]; tensor concat_127 = concat(axis = concat_127_axis_0, interleave = concat_127_interleave_0, values = (expand_dims_190, concat_127_values1_0, var_587, concat_127_values3_0))[name = string("concat_127")]; tensor model_model_kv_cache_0_internal_tensor_assign_32_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_cast_fp16 = slice_update(begin = concat_126, begin_mask = model_model_kv_cache_0_internal_tensor_assign_32_begin_mask_0, end = concat_127, end_mask = model_model_kv_cache_0_internal_tensor_assign_32_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_32_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_32_stride_0, update = var_3497, x = coreml_update_state_62)[name = string("model_model_kv_cache_0_internal_tensor_assign_32_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_32_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_95_write_state")]; tensor coreml_update_state_63 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_95")]; tensor var_3557_begin_0 = const()[name = string("op_3557_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_3557_end_0 = const()[name = string("op_3557_end_0"), val = tensor([16, 8, 4096, 64])]; tensor var_3557_end_mask_0 = const()[name = string("op_3557_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3557_cast_fp16 = slice_by_index(begin = var_3557_begin_0, end = var_3557_end_0, end_mask = var_3557_end_mask_0, x = coreml_update_state_63)[name = string("op_3557_cast_fp16")]; tensor K_layer_cache_axes_0 = const()[name = string("K_layer_cache_axes_0"), val = tensor([0])]; tensor K_layer_cache_cast_fp16 = squeeze(axes = K_layer_cache_axes_0, x = var_3557_cast_fp16)[name = string("K_layer_cache_cast_fp16")]; tensor var_3559_begin_0 = const()[name = string("op_3559_begin_0"), val = tensor([31, 0, 0, 0])]; tensor var_3559_end_0 = const()[name = string("op_3559_end_0"), val = tensor([1, 8, 4096, 64])]; tensor var_3559_end_mask_0 = const()[name = string("op_3559_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3559_cast_fp16 = slice_by_index(begin = var_3559_begin_0, end = var_3559_end_0, end_mask = var_3559_end_mask_0, x = coreml_update_state_63)[name = string("op_3559_cast_fp16")]; tensor V_layer_cache_axes_0 = const()[name = string("V_layer_cache_axes_0"), val = tensor([0])]; tensor V_layer_cache_cast_fp16 = squeeze(axes = V_layer_cache_axes_0, x = var_3559_cast_fp16)[name = string("V_layer_cache_cast_fp16")]; tensor x_431_axes_0 = const()[name = string("x_431_axes_0"), val = tensor([1])]; tensor x_431_cast_fp16 = expand_dims(axes = x_431_axes_0, x = K_layer_cache_cast_fp16)[name = string("x_431_cast_fp16")]; tensor var_3568 = const()[name = string("op_3568"), val = tensor([1, 4, 1, 1])]; tensor x_433_cast_fp16 = tile(reps = var_3568, x = x_431_cast_fp16)[name = string("x_433_cast_fp16")]; tensor var_3572 = const()[name = string("op_3572"), val = tensor([1, -1, 4096, 64])]; tensor key_states_cast_fp16 = reshape(shape = var_3572, x = x_433_cast_fp16)[name = string("key_states_cast_fp16")]; tensor x_437_axes_0 = const()[name = string("x_437_axes_0"), val = tensor([1])]; tensor x_437_cast_fp16 = expand_dims(axes = x_437_axes_0, x = V_layer_cache_cast_fp16)[name = string("x_437_cast_fp16")]; tensor var_3575 = const()[name = string("op_3575"), val = tensor([1, 4, 1, 1])]; tensor x_439_cast_fp16 = tile(reps = var_3575, x = x_437_cast_fp16)[name = string("x_439_cast_fp16")]; tensor var_3579 = const()[name = string("op_3579"), val = tensor([1, -1, 4096, 64])]; tensor value_states_cast_fp16 = reshape(shape = var_3579, x = x_439_cast_fp16)[name = string("value_states_cast_fp16")]; bool var_3582_transpose_x_1 = const()[name = string("op_3582_transpose_x_1"), val = bool(false)]; bool var_3582_transpose_y_1 = const()[name = string("op_3582_transpose_y_1"), val = bool(true)]; tensor var_3582_cast_fp16 = matmul(transpose_x = var_3582_transpose_x_1, transpose_y = var_3582_transpose_y_1, x = rotated_61_cast_fp16, y = key_states_cast_fp16)[name = string("op_3582_cast_fp16")]; fp16 var_3583_to_fp16 = const()[name = string("op_3583_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_61_cast_fp16 = mul(x = var_3582_cast_fp16, y = var_3583_to_fp16)[name = string("attn_weights_61_cast_fp16")]; tensor x_441_cast_fp16 = add(x = attn_weights_61_cast_fp16, y = causal_mask)[name = string("x_441_cast_fp16")]; tensor reduce_max_15_axes_0 = const()[name = string("reduce_max_15_axes_0"), val = tensor([-1])]; bool reduce_max_15_keep_dims_0 = const()[name = string("reduce_max_15_keep_dims_0"), val = bool(true)]; tensor reduce_max_15_cast_fp16 = reduce_max(axes = reduce_max_15_axes_0, keep_dims = reduce_max_15_keep_dims_0, x = x_441_cast_fp16)[name = string("reduce_max_15_cast_fp16")]; tensor x_443_cast_fp16 = sub(x = x_441_cast_fp16, y = reduce_max_15_cast_fp16)[name = string("x_443_cast_fp16")]; tensor exp_x_cast_fp16 = exp(x = x_443_cast_fp16)[name = string("exp_x_cast_fp16")]; tensor var_3594_axes_0 = const()[name = string("op_3594_axes_0"), val = tensor([-1])]; bool var_3594_keep_dims_0 = const()[name = string("op_3594_keep_dims_0"), val = bool(true)]; tensor var_3594_cast_fp16 = reduce_sum(axes = var_3594_axes_0, keep_dims = var_3594_keep_dims_0, x = exp_x_cast_fp16)[name = string("op_3594_cast_fp16")]; tensor attn_weights_cast_fp16 = real_div(x = exp_x_cast_fp16, y = var_3594_cast_fp16)[name = string("attn_weights_cast_fp16")]; bool attn_output_91_transpose_x_0 = const()[name = string("attn_output_91_transpose_x_0"), val = bool(false)]; bool attn_output_91_transpose_y_0 = const()[name = string("attn_output_91_transpose_y_0"), val = bool(false)]; tensor attn_output_91_cast_fp16 = matmul(transpose_x = attn_output_91_transpose_x_0, transpose_y = attn_output_91_transpose_y_0, x = attn_weights_cast_fp16, y = value_states_cast_fp16)[name = string("attn_output_91_cast_fp16")]; tensor var_3597_perm_0 = const()[name = string("op_3597_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3599 = const()[name = string("op_3599"), val = tensor([1, 1, 2048])]; tensor var_3597_cast_fp16 = transpose(perm = var_3597_perm_0, x = attn_output_91_cast_fp16)[name = string("transpose_2")]; tensor input_215_cast_fp16 = reshape(shape = var_3599, x = var_3597_cast_fp16)[name = string("input_215_cast_fp16")]; tensor model_model_layers_15_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488185920))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(490283136))))[name = string("model_model_layers_15_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_15_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_15_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_215_cast_fp16)[name = string("linear_15_cast_fp16")]; tensor hidden_states_125_cast_fp16 = add(x = hidden_states_121_cast_fp16, y = linear_15_cast_fp16)[name = string("hidden_states_125_cast_fp16")]; fp16 const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3605_cast_fp16 = mul(x = hidden_states_125_cast_fp16, y = const_252_promoted_to_fp16)[name = string("op_3605_cast_fp16")]; bool input_217_interleave_0 = const()[name = string("input_217_interleave_0"), val = bool(false)]; tensor input_217_cast_fp16 = concat(axis = var_80, interleave = input_217_interleave_0, values = (hidden_states_125_cast_fp16, var_3605_cast_fp16))[name = string("input_217_cast_fp16")]; tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_74_to_fp16, x = input_217_cast_fp16)[name = string("normed_125_cast_fp16")]; tensor normed_127_begin_0 = const()[name = string("normed_127_begin_0"), val = tensor([0, 0, 0])]; tensor normed_127_end_0 = const()[name = string("normed_127_end_0"), val = tensor([1, 1, 2048])]; tensor normed_127_end_mask_0 = const()[name = string("normed_127_end_mask_0"), val = tensor([true, true, false])]; tensor normed_127_cast_fp16 = slice_by_index(begin = normed_127_begin_0, end = normed_127_end_0, end_mask = normed_127_end_mask_0, x = normed_125_cast_fp16)[name = string("normed_127_cast_fp16")]; tensor const_255_promoted_to_fp16 = const()[name = string("const_255_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(490291392)))]; tensor x_445_cast_fp16 = mul(x = normed_127_cast_fp16, y = const_255_promoted_to_fp16)[name = string("x_445_cast_fp16")]; tensor var_3623 = const()[name = string("op_3623"), val = tensor([0, 2, 1])]; tensor input_219_axes_0 = const()[name = string("input_219_axes_0"), val = tensor([2])]; tensor var_3624 = transpose(perm = var_3623, x = x_445_cast_fp16)[name = string("transpose_1")]; tensor input_219 = expand_dims(axes = input_219_axes_0, x = var_3624)[name = string("input_219")]; string input_221_pad_type_0 = const()[name = string("input_221_pad_type_0"), val = string("valid")]; tensor input_221_strides_0 = const()[name = string("input_221_strides_0"), val = tensor([1, 1])]; tensor input_221_pad_0 = const()[name = string("input_221_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_221_dilations_0 = const()[name = string("input_221_dilations_0"), val = tensor([1, 1])]; int32 input_221_groups_0 = const()[name = string("input_221_groups_0"), val = int32(1)]; tensor input_221 = conv(dilations = input_221_dilations_0, groups = input_221_groups_0, pad = input_221_pad_0, pad_type = input_221_pad_type_0, strides = input_221_strides_0, weight = model_model_layers_15_mlp_gate_proj_weight_palettized, x = input_219)[name = string("input_221")]; string up_states_pad_type_0 = const()[name = string("up_states_pad_type_0"), val = string("valid")]; tensor up_states_strides_0 = const()[name = string("up_states_strides_0"), val = tensor([1, 1])]; tensor up_states_pad_0 = const()[name = string("up_states_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_dilations_0 = const()[name = string("up_states_dilations_0"), val = tensor([1, 1])]; int32 up_states_groups_0 = const()[name = string("up_states_groups_0"), val = int32(1)]; tensor up_states = conv(dilations = up_states_dilations_0, groups = up_states_groups_0, pad = up_states_pad_0, pad_type = up_states_pad_type_0, strides = up_states_strides_0, weight = model_model_layers_15_mlp_up_proj_weight_palettized, x = input_219)[name = string("up_states")]; tensor gate_states = silu(x = input_221)[name = string("gate_states")]; tensor input_223 = mul(x = gate_states, y = up_states)[name = string("input_223")]; string hidden_states_127_pad_type_0 = const()[name = string("hidden_states_127_pad_type_0"), val = string("valid")]; tensor hidden_states_127_strides_0 = const()[name = string("hidden_states_127_strides_0"), val = tensor([1, 1])]; tensor hidden_states_127_pad_0 = const()[name = string("hidden_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_127_dilations_0 = const()[name = string("hidden_states_127_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_127_groups_0 = const()[name = string("hidden_states_127_groups_0"), val = int32(1)]; tensor hidden_states_127 = conv(dilations = hidden_states_127_dilations_0, groups = hidden_states_127_groups_0, pad = hidden_states_127_pad_0, pad_type = hidden_states_127_pad_type_0, strides = hidden_states_127_strides_0, weight = model_model_layers_15_mlp_down_proj_weight_palettized, x = input_223)[name = string("hidden_states_127")]; tensor var_3646_axes_0 = const()[name = string("op_3646_axes_0"), val = tensor([2])]; tensor var_3646 = squeeze(axes = var_3646_axes_0, x = hidden_states_127)[name = string("op_3646")]; tensor var_3647 = const()[name = string("op_3647"), val = tensor([0, 2, 1])]; tensor var_3648 = transpose(perm = var_3647, x = var_3646)[name = string("transpose_0")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_125_cast_fp16, y = var_3648)[name = string("hidden_states_cast_fp16")]; fp16 const_256_promoted_to_fp16 = const()[name = string("const_256_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3651_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_256_promoted_to_fp16)[name = string("op_3651_cast_fp16")]; bool input_interleave_0 = const()[name = string("input_interleave_0"), val = bool(false)]; tensor input_cast_fp16 = concat(axis = var_80, interleave = input_interleave_0, values = (hidden_states_cast_fp16, var_3651_cast_fp16))[name = string("input_cast_fp16")]; tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_74_to_fp16, x = input_cast_fp16)[name = string("normed_129_cast_fp16")]; tensor normed_begin_0 = const()[name = string("normed_begin_0"), val = tensor([0, 0, 0])]; tensor normed_end_0 = const()[name = string("normed_end_0"), val = tensor([1, 1, 2048])]; tensor normed_end_mask_0 = const()[name = string("normed_end_mask_0"), val = tensor([true, true, false])]; tensor normed_cast_fp16 = slice_by_index(begin = normed_begin_0, end = normed_end_0, end_mask = normed_end_mask_0, x = normed_129_cast_fp16)[name = string("normed_cast_fp16")]; tensor const_259_promoted_to_fp16 = const()[name = string("const_259_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(490295552)))]; tensor output_hidden_states = mul(x = normed_cast_fp16, y = const_259_promoted_to_fp16)[name = string("op_3664_cast_fp16")]; tensor position_ids_tmp = identity(x = position_ids)[name = string("position_ids_tmp")]; } -> (output_hidden_states); func prefill(tensor causal_mask, tensor current_pos, tensor hidden_states, state> model_model_kv_cache_0, tensor position_ids) { tensor model_model_layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2097280))))[name = string("model_model_layers_0_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2105536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2629888))))[name = string("model_model_layers_0_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(2632000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3156352))))[name = string("model_model_layers_0_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_0_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(3158464))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11547136))))[name = string("model_model_layers_0_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_0_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(11579968))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(19968640))))[name = string("model_model_layers_0_mlp_up_proj_weight_palettized")]; tensor model_model_layers_0_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(20001472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28390144))))[name = string("model_model_layers_0_mlp_down_proj_weight_palettized")]; tensor model_model_layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(28398400))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30495616))))[name = string("model_model_layers_1_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30503872))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31028224))))[name = string("model_model_layers_1_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31030336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31554688))))[name = string("model_model_layers_1_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_1_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(31556800))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39945472))))[name = string("model_model_layers_1_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_1_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39978304))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48366976))))[name = string("model_model_layers_1_mlp_up_proj_weight_palettized")]; tensor model_model_layers_1_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48399808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(56788480))))[name = string("model_model_layers_1_mlp_down_proj_weight_palettized")]; tensor model_model_layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(56796736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(58893952))))[name = string("model_model_layers_2_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(58902208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59426560))))[name = string("model_model_layers_2_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59428672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59953024))))[name = string("model_model_layers_2_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_2_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(59955136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68343808))))[name = string("model_model_layers_2_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_2_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68376640))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76765312))))[name = string("model_model_layers_2_mlp_up_proj_weight_palettized")]; tensor model_model_layers_2_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76798144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85186816))))[name = string("model_model_layers_2_mlp_down_proj_weight_palettized")]; tensor model_model_layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85195072))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87292288))))[name = string("model_model_layers_3_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87300544))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87824896))))[name = string("model_model_layers_3_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(87827008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(88351360))))[name = string("model_model_layers_3_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_3_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(88353472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(96742144))))[name = string("model_model_layers_3_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_3_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(96774976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105163648))))[name = string("model_model_layers_3_mlp_up_proj_weight_palettized")]; tensor model_model_layers_3_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(105196480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(113585152))))[name = string("model_model_layers_3_mlp_down_proj_weight_palettized")]; tensor model_model_layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(113593408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115690624))))[name = string("model_model_layers_4_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115698880))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116223232))))[name = string("model_model_layers_4_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116225344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116749696))))[name = string("model_model_layers_4_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_4_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116751808))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125140480))))[name = string("model_model_layers_4_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_4_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125173312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133561984))))[name = string("model_model_layers_4_mlp_up_proj_weight_palettized")]; tensor model_model_layers_4_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133594816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141983488))))[name = string("model_model_layers_4_mlp_down_proj_weight_palettized")]; tensor model_model_layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141991744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144088960))))[name = string("model_model_layers_5_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144097216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144621568))))[name = string("model_model_layers_5_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144623680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145148032))))[name = string("model_model_layers_5_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_5_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145150144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153538816))))[name = string("model_model_layers_5_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_5_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153571648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(161960320))))[name = string("model_model_layers_5_mlp_up_proj_weight_palettized")]; tensor model_model_layers_5_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(161993152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(170381824))))[name = string("model_model_layers_5_mlp_down_proj_weight_palettized")]; tensor model_model_layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(170390080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172487296))))[name = string("model_model_layers_6_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(172495552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(173019904))))[name = string("model_model_layers_6_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(173022016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(173546368))))[name = string("model_model_layers_6_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_6_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(173548480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(181937152))))[name = string("model_model_layers_6_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_6_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(181969984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(190358656))))[name = string("model_model_layers_6_mlp_up_proj_weight_palettized")]; tensor model_model_layers_6_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(190391488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(198780160))))[name = string("model_model_layers_6_mlp_down_proj_weight_palettized")]; tensor model_model_layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(198788416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200885632))))[name = string("model_model_layers_7_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(200893888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(201418240))))[name = string("model_model_layers_7_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(201420352))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(201944704))))[name = string("model_model_layers_7_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_7_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(201946816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210335488))))[name = string("model_model_layers_7_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_7_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210368320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218756992))))[name = string("model_model_layers_7_mlp_up_proj_weight_palettized")]; tensor model_model_layers_7_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218789824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(227178496))))[name = string("model_model_layers_7_mlp_down_proj_weight_palettized")]; tensor model_model_layers_8_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(227186752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(229283968))))[name = string("model_model_layers_8_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_8_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(229292224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(229816576))))[name = string("model_model_layers_8_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_8_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(229818688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(230343040))))[name = string("model_model_layers_8_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_8_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(230345152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(238733824))))[name = string("model_model_layers_8_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_8_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(238766656))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(247155328))))[name = string("model_model_layers_8_mlp_up_proj_weight_palettized")]; tensor model_model_layers_8_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(247188160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(255576832))))[name = string("model_model_layers_8_mlp_down_proj_weight_palettized")]; tensor model_model_layers_9_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(255585088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257682304))))[name = string("model_model_layers_9_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_9_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257690560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258214912))))[name = string("model_model_layers_9_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_9_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258217024))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258741376))))[name = string("model_model_layers_9_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_9_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258743488))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(267132160))))[name = string("model_model_layers_9_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_9_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(267164992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275553664))))[name = string("model_model_layers_9_mlp_up_proj_weight_palettized")]; tensor model_model_layers_9_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(275586496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283975168))))[name = string("model_model_layers_9_mlp_down_proj_weight_palettized")]; tensor model_model_layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283983424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286080640))))[name = string("model_model_layers_10_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286088896))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286613248))))[name = string("model_model_layers_10_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(286615360))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(287139712))))[name = string("model_model_layers_10_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_10_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(287141824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295530496))))[name = string("model_model_layers_10_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_10_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(295563328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303952000))))[name = string("model_model_layers_10_mlp_up_proj_weight_palettized")]; tensor model_model_layers_10_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303984832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312373504))))[name = string("model_model_layers_10_mlp_down_proj_weight_palettized")]; tensor model_model_layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312381760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(314478976))))[name = string("model_model_layers_11_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(314487232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315011584))))[name = string("model_model_layers_11_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315013696))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315538048))))[name = string("model_model_layers_11_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_11_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(315540160))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323928832))))[name = string("model_model_layers_11_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_11_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(323961664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332350336))))[name = string("model_model_layers_11_mlp_up_proj_weight_palettized")]; tensor model_model_layers_11_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(332383168))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(340771840))))[name = string("model_model_layers_11_mlp_down_proj_weight_palettized")]; tensor model_model_layers_12_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(340780096))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342877312))))[name = string("model_model_layers_12_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_12_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(342885568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(343409920))))[name = string("model_model_layers_12_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_12_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(343412032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(343936384))))[name = string("model_model_layers_12_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_12_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(343938496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352327168))))[name = string("model_model_layers_12_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_12_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352360000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360748672))))[name = string("model_model_layers_12_mlp_up_proj_weight_palettized")]; tensor model_model_layers_12_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360781504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(369170176))))[name = string("model_model_layers_12_mlp_down_proj_weight_palettized")]; tensor model_model_layers_13_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(369178432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(371275648))))[name = string("model_model_layers_13_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_13_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(371283904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(371808256))))[name = string("model_model_layers_13_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_13_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(371810368))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372334720))))[name = string("model_model_layers_13_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_13_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372336832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(380725504))))[name = string("model_model_layers_13_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_13_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(380758336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(389147008))))[name = string("model_model_layers_13_mlp_up_proj_weight_palettized")]; tensor model_model_layers_13_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(389179840))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397568512))))[name = string("model_model_layers_13_mlp_down_proj_weight_palettized")]; tensor model_model_layers_14_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397576768))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399673984))))[name = string("model_model_layers_14_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_14_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399682240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400206592))))[name = string("model_model_layers_14_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_14_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400208704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400733056))))[name = string("model_model_layers_14_self_attn_v_proj_weight_palettized")]; tensor model_model_layers_14_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400735168))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(409123840))))[name = string("model_model_layers_14_mlp_gate_proj_weight_palettized")]; tensor model_model_layers_14_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(409156672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(417545344))))[name = string("model_model_layers_14_mlp_up_proj_weight_palettized")]; tensor model_model_layers_14_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(417578176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(425966848))))[name = string("model_model_layers_14_mlp_down_proj_weight_palettized")]; tensor model_model_layers_15_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(425975104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(428072320))))[name = string("model_model_layers_15_self_attn_q_proj_weight_palettized")]; tensor model_model_layers_15_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(428080576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(428604928))))[name = string("model_model_layers_15_self_attn_k_proj_weight_palettized")]; tensor model_model_layers_15_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(428607040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(429131392))))[name = string("model_model_layers_15_self_attn_v_proj_weight_palettized")]; int32 var_73 = const()[name = string("op_73"), val = int32(-1)]; int32 var_486_batch_dims_0 = const()[name = string("op_486_batch_dims_0"), val = int32(0)]; bool var_486_validate_indices_0 = const()[name = string("op_486_validate_indices_0"), val = bool(false)]; tensor var_86_to_fp16 = const()[name = string("op_86_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(455422080)))]; string position_ids_to_int16_dtype_0 = const()[name = string("position_ids_to_int16_dtype_0"), val = string("int16")]; string cast_166_dtype_0 = const()[name = string("cast_166_dtype_0"), val = string("int32")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor position_ids_to_int16 = cast(dtype = position_ids_to_int16_dtype_0, x = position_ids)[name = string("cast_5")]; tensor cast_166 = cast(dtype = cast_166_dtype_0, x = position_ids_to_int16)[name = string("cast_4")]; tensor greater_equal_0 = greater_equal(x = cast_166, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_128 = const()[name = string("slice_by_index_128"), val = int32(8192)]; tensor add_0 = add(x = cast_166, y = slice_by_index_128)[name = string("add_0")]; tensor select_0 = select(a = cast_166, b = add_0, cond = greater_equal_0)[name = string("select_0")]; string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(8192)]; tensor add_0_1 = add(x = cast_0, y = slice_by_index_0)[name = string("add_0_1")]; tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; int32 op_486_cast_fp16_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_486_cast_fp16_cast_uint16_cast_uint16_axis_0"), val = int32(1)]; tensor op_486_cast_fp16_cast_uint16_cast_uint16 = gather(axis = op_486_cast_fp16_cast_uint16_cast_uint16_axis_0, batch_dims = var_486_batch_dims_0, indices = select_0_1, validate_indices = var_486_validate_indices_0, x = var_86_to_fp16)[name = string("op_486_cast_fp16_cast_uint16_cast_uint16")]; tensor var_487 = const()[name = string("op_487"), val = tensor([1, 64, 1, 64])]; tensor cos_1_cast_fp16 = reshape(shape = var_487, x = op_486_cast_fp16_cast_uint16_cast_uint16)[name = string("cos_1_cast_fp16")]; int32 var_491_axis_0 = const()[name = string("op_491_axis_0"), val = int32(1)]; int32 var_491_batch_dims_0 = const()[name = string("op_491_batch_dims_0"), val = int32(0)]; bool var_491_validate_indices_0 = const()[name = string("op_491_validate_indices_0"), val = bool(false)]; tensor var_81_to_fp16 = const()[name = string("op_81_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454373440)))]; string position_ids_to_uint16_dtype_0 = const()[name = string("position_ids_to_uint16_dtype_0"), val = string("uint16")]; tensor position_ids_to_uint16 = cast(dtype = position_ids_to_uint16_dtype_0, x = position_ids)[name = string("cast_1")]; tensor var_491_cast_fp16_cast_uint16 = gather(axis = var_491_axis_0, batch_dims = var_491_batch_dims_0, indices = position_ids_to_uint16, validate_indices = var_491_validate_indices_0, x = var_81_to_fp16)[name = string("op_491_cast_fp16_cast_uint16")]; tensor var_492 = const()[name = string("op_492"), val = tensor([1, 64, 1, 64])]; tensor sin_1_cast_fp16 = reshape(shape = var_492, x = var_491_cast_fp16_cast_uint16)[name = string("sin_1_cast_fp16")]; fp16 const_1_promoted_to_fp16 = const()[name = string("const_1_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_497_cast_fp16 = mul(x = hidden_states, y = const_1_promoted_to_fp16)[name = string("op_497_cast_fp16")]; bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; tensor input_1_cast_fp16 = concat(axis = var_73, interleave = input_1_interleave_0, values = (hidden_states, var_497_cast_fp16))[name = string("input_1_cast_fp16")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_76_to_fp16 = const()[name = string("op_76_to_fp16"), val = fp16(0x1.5p-17)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_76_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; tensor normed_3_begin_0 = const()[name = string("normed_3_begin_0"), val = tensor([0, 0, 0])]; tensor normed_3_end_0 = const()[name = string("normed_3_end_0"), val = tensor([1, 64, 2048])]; tensor normed_3_end_mask_0 = const()[name = string("normed_3_end_mask_0"), val = tensor([true, true, false])]; tensor normed_3_cast_fp16 = slice_by_index(begin = normed_3_begin_0, end = normed_3_end_0, end_mask = normed_3_end_mask_0, x = normed_1_cast_fp16)[name = string("normed_3_cast_fp16")]; tensor const_4_promoted_to_fp16 = const()[name = string("const_4_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(456470720)))]; tensor hidden_states_3_cast_fp16 = mul(x = normed_3_cast_fp16, y = const_4_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor var_512 = const()[name = string("op_512"), val = tensor([0, 2, 1])]; tensor var_514_axes_0 = const()[name = string("op_514_axes_0"), val = tensor([2])]; tensor var_513_cast_fp16 = transpose(perm = var_512, x = hidden_states_3_cast_fp16)[name = string("transpose_111")]; tensor var_514_cast_fp16 = expand_dims(axes = var_514_axes_0, x = var_513_cast_fp16)[name = string("op_514_cast_fp16")]; string query_states_1_pad_type_0 = const()[name = string("query_states_1_pad_type_0"), val = string("valid")]; tensor query_states_1_strides_0 = const()[name = string("query_states_1_strides_0"), val = tensor([1, 1])]; tensor query_states_1_pad_0 = const()[name = string("query_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_1_dilations_0 = const()[name = string("query_states_1_dilations_0"), val = tensor([1, 1])]; int32 query_states_1_groups_0 = const()[name = string("query_states_1_groups_0"), val = int32(1)]; tensor query_states_1 = conv(dilations = query_states_1_dilations_0, groups = query_states_1_groups_0, pad = query_states_1_pad_0, pad_type = query_states_1_pad_type_0, strides = query_states_1_strides_0, weight = model_model_layers_0_self_attn_q_proj_weight_palettized, x = var_514_cast_fp16)[name = string("query_states_1")]; string key_states_1_pad_type_0 = const()[name = string("key_states_1_pad_type_0"), val = string("valid")]; tensor key_states_1_strides_0 = const()[name = string("key_states_1_strides_0"), val = tensor([1, 1])]; tensor key_states_1_pad_0 = const()[name = string("key_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_1_dilations_0 = const()[name = string("key_states_1_dilations_0"), val = tensor([1, 1])]; int32 key_states_1_groups_0 = const()[name = string("key_states_1_groups_0"), val = int32(1)]; tensor key_states_1 = conv(dilations = key_states_1_dilations_0, groups = key_states_1_groups_0, pad = key_states_1_pad_0, pad_type = key_states_1_pad_type_0, strides = key_states_1_strides_0, weight = model_model_layers_0_self_attn_k_proj_weight_palettized, x = var_514_cast_fp16)[name = string("key_states_1")]; string value_states_1_pad_type_0 = const()[name = string("value_states_1_pad_type_0"), val = string("valid")]; tensor value_states_1_strides_0 = const()[name = string("value_states_1_strides_0"), val = tensor([1, 1])]; tensor value_states_1_pad_0 = const()[name = string("value_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_1_dilations_0 = const()[name = string("value_states_1_dilations_0"), val = tensor([1, 1])]; int32 value_states_1_groups_0 = const()[name = string("value_states_1_groups_0"), val = int32(1)]; tensor value_states_1 = conv(dilations = value_states_1_dilations_0, groups = value_states_1_groups_0, pad = value_states_1_pad_0, pad_type = value_states_1_pad_type_0, strides = value_states_1_strides_0, weight = model_model_layers_0_self_attn_v_proj_weight_palettized, x = var_514_cast_fp16)[name = string("value_states_1")]; tensor var_534 = const()[name = string("op_534"), val = tensor([1, 32, 64, 64])]; tensor var_535 = reshape(shape = var_534, x = query_states_1)[name = string("op_535")]; tensor var_536 = const()[name = string("op_536"), val = tensor([0, 1, 3, 2])]; tensor var_538 = const()[name = string("op_538"), val = tensor([1, 8, 64, 64])]; tensor var_539 = reshape(shape = var_538, x = key_states_1)[name = string("op_539")]; tensor var_540 = const()[name = string("op_540"), val = tensor([0, 1, 3, 2])]; tensor var_542 = const()[name = string("op_542"), val = tensor([1, 8, 64, 64])]; tensor var_543 = reshape(shape = var_542, x = value_states_1)[name = string("op_543")]; tensor var_544 = const()[name = string("op_544"), val = tensor([0, 1, 3, 2])]; tensor var_546 = const()[name = string("op_546"), val = tensor([0, 2, 1, 3])]; tensor var_548 = const()[name = string("op_548"), val = tensor([0, 2, 1, 3])]; tensor x1_1_begin_0 = const()[name = string("x1_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_1_end_0 = const()[name = string("x1_1_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_1_end_mask_0 = const()[name = string("x1_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_1 = transpose(perm = var_536, x = var_535)[name = string("transpose_110")]; tensor x1_1 = slice_by_index(begin = x1_1_begin_0, end = x1_1_end_0, end_mask = x1_1_end_mask_0, x = x_1)[name = string("x1_1")]; tensor x2_1_begin_0 = const()[name = string("x2_1_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_1_end_0 = const()[name = string("x2_1_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_1_end_mask_0 = const()[name = string("x2_1_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_1 = slice_by_index(begin = x2_1_begin_0, end = x2_1_end_0, end_mask = x2_1_end_mask_0, x = x_1)[name = string("x2_1")]; tensor cos_7_begin_0 = const()[name = string("cos_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor cos_7_end_0 = const()[name = string("cos_7_end_0"), val = tensor([1, 1, 64, 32])]; tensor cos_7_end_mask_0 = const()[name = string("cos_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor cos_5 = transpose(perm = var_546, x = cos_1_cast_fp16)[name = string("transpose_109")]; tensor cos_7 = slice_by_index(begin = cos_7_begin_0, end = cos_7_end_0, end_mask = cos_7_end_mask_0, x = cos_5)[name = string("cos_7")]; tensor sin_7_begin_0 = const()[name = string("sin_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor sin_7_end_0 = const()[name = string("sin_7_end_0"), val = tensor([1, 1, 64, 32])]; tensor sin_7_end_mask_0 = const()[name = string("sin_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor sin_5 = transpose(perm = var_548, x = sin_1_cast_fp16)[name = string("transpose_108")]; tensor sin_7 = slice_by_index(begin = sin_7_begin_0, end = sin_7_end_0, end_mask = sin_7_end_mask_0, x = sin_5)[name = string("sin_7")]; tensor var_562 = mul(x = x1_1, y = cos_7)[name = string("op_562")]; tensor var_563 = mul(x = x2_1, y = sin_7)[name = string("op_563")]; tensor var_564 = sub(x = var_562, y = var_563)[name = string("op_564")]; tensor var_565 = mul(x = x2_1, y = cos_7)[name = string("op_565")]; tensor var_566 = mul(x = x1_1, y = sin_7)[name = string("op_566")]; tensor var_567 = add(x = var_565, y = var_566)[name = string("op_567")]; bool rotated_1_interleave_0 = const()[name = string("rotated_1_interleave_0"), val = bool(false)]; tensor rotated_1 = concat(axis = var_73, interleave = rotated_1_interleave_0, values = (var_564, var_567))[name = string("rotated_1")]; tensor x1_3_begin_0 = const()[name = string("x1_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_3_end_0 = const()[name = string("x1_3_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_3_end_mask_0 = const()[name = string("x1_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_5 = transpose(perm = var_540, x = var_539)[name = string("transpose_107")]; tensor x1_3 = slice_by_index(begin = x1_3_begin_0, end = x1_3_end_0, end_mask = x1_3_end_mask_0, x = x_5)[name = string("x1_3")]; tensor x2_3_begin_0 = const()[name = string("x2_3_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_3_end_0 = const()[name = string("x2_3_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_3_end_mask_0 = const()[name = string("x2_3_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_3 = slice_by_index(begin = x2_3_begin_0, end = x2_3_end_0, end_mask = x2_3_end_mask_0, x = x_5)[name = string("x2_3")]; tensor var_583 = mul(x = x1_3, y = cos_7)[name = string("op_583")]; tensor var_584 = mul(x = x2_3, y = sin_7)[name = string("op_584")]; tensor var_585 = sub(x = var_583, y = var_584)[name = string("op_585")]; tensor var_586 = mul(x = x2_3, y = cos_7)[name = string("op_586")]; tensor var_587 = mul(x = x1_3, y = sin_7)[name = string("op_587")]; tensor var_588 = add(x = var_586, y = var_587)[name = string("op_588")]; bool rotated_3_interleave_0 = const()[name = string("rotated_3_interleave_0"), val = bool(false)]; tensor rotated_3 = concat(axis = var_73, interleave = rotated_3_interleave_0, values = (var_585, var_588))[name = string("rotated_3")]; tensor seq_length_1 = const()[name = string("seq_length_1"), val = tensor([64])]; tensor var_597 = add(x = current_pos, y = seq_length_1)[name = string("op_597")]; tensor read_state_0 = read_state(input = model_model_kv_cache_0)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; int32 concat_2_axis_0 = const()[name = string("concat_2_axis_0"), val = int32(0)]; bool concat_2_interleave_0 = const()[name = string("concat_2_interleave_0"), val = bool(false)]; tensor concat_2 = concat(axis = concat_2_axis_0, interleave = concat_2_interleave_0, values = (expand_dims_0, expand_dims_1, current_pos, expand_dims_3))[name = string("concat_2")]; tensor concat_3_values1_0 = const()[name = string("concat_3_values1_0"), val = tensor([0])]; tensor concat_3_values3_0 = const()[name = string("concat_3_values3_0"), val = tensor([0])]; int32 concat_3_axis_0 = const()[name = string("concat_3_axis_0"), val = int32(0)]; bool concat_3_interleave_0 = const()[name = string("concat_3_interleave_0"), val = bool(false)]; tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (expand_dims_4, concat_3_values1_0, var_597, concat_3_values3_0))[name = string("concat_3")]; tensor model_model_kv_cache_0_internal_tensor_assign_1_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_2, begin_mask = model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0, end = concat_3, end_mask = model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_1_stride_0, update = rotated_3, x = read_state_0)[name = string("model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_96_write_state")]; tensor coreml_update_state_32 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_96")]; tensor expand_dims_6 = const()[name = string("expand_dims_6"), val = tensor([16])]; tensor expand_dims_7 = const()[name = string("expand_dims_7"), val = tensor([0])]; tensor expand_dims_9 = const()[name = string("expand_dims_9"), val = tensor([0])]; tensor expand_dims_10 = const()[name = string("expand_dims_10"), val = tensor([17])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_6, expand_dims_7, current_pos, expand_dims_9))[name = string("concat_6")]; tensor concat_7_values1_0 = const()[name = string("concat_7_values1_0"), val = tensor([0])]; tensor concat_7_values3_0 = const()[name = string("concat_7_values3_0"), val = tensor([0])]; int32 concat_7_axis_0 = const()[name = string("concat_7_axis_0"), val = int32(0)]; bool concat_7_interleave_0 = const()[name = string("concat_7_interleave_0"), val = bool(false)]; tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (expand_dims_10, concat_7_values1_0, var_597, concat_7_values3_0))[name = string("concat_7")]; tensor model_model_kv_cache_0_internal_tensor_assign_2_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_3 = transpose(perm = var_544, x = var_543)[name = string("transpose_106")]; tensor model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_6, begin_mask = model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0, end = concat_7, end_mask = model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_2_stride_0, update = value_states_3, x = coreml_update_state_32)[name = string("model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_97_write_state")]; tensor coreml_update_state_33 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_97")]; tensor var_611_begin_0 = const()[name = string("op_611_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_611_end_0 = const()[name = string("op_611_end_0"), val = tensor([1, 8, 4096, 64])]; tensor var_611_end_mask_0 = const()[name = string("op_611_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_611_cast_fp16 = slice_by_index(begin = var_611_begin_0, end = var_611_end_0, end_mask = var_611_end_mask_0, x = coreml_update_state_33)[name = string("op_611_cast_fp16")]; tensor K_layer_cache_1_axes_0 = const()[name = string("K_layer_cache_1_axes_0"), val = tensor([0])]; tensor K_layer_cache_1_cast_fp16 = squeeze(axes = K_layer_cache_1_axes_0, x = var_611_cast_fp16)[name = string("K_layer_cache_1_cast_fp16")]; tensor var_613_begin_0 = const()[name = string("op_613_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_613_end_0 = const()[name = string("op_613_end_0"), val = tensor([17, 8, 4096, 64])]; tensor var_613_end_mask_0 = const()[name = string("op_613_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_613_cast_fp16 = slice_by_index(begin = var_613_begin_0, end = var_613_end_0, end_mask = var_613_end_mask_0, x = coreml_update_state_33)[name = string("op_613_cast_fp16")]; tensor V_layer_cache_1_axes_0 = const()[name = string("V_layer_cache_1_axes_0"), val = tensor([0])]; tensor V_layer_cache_1_cast_fp16 = squeeze(axes = V_layer_cache_1_axes_0, x = var_613_cast_fp16)[name = string("V_layer_cache_1_cast_fp16")]; tensor x_11_axes_0 = const()[name = string("x_11_axes_0"), val = tensor([1])]; tensor x_11_cast_fp16 = expand_dims(axes = x_11_axes_0, x = K_layer_cache_1_cast_fp16)[name = string("x_11_cast_fp16")]; tensor var_622 = const()[name = string("op_622"), val = tensor([1, 4, 1, 1])]; tensor x_13_cast_fp16 = tile(reps = var_622, x = x_11_cast_fp16)[name = string("x_13_cast_fp16")]; tensor var_626 = const()[name = string("op_626"), val = tensor([1, -1, 4096, 64])]; tensor var_627_cast_fp16 = reshape(shape = var_626, x = x_13_cast_fp16)[name = string("op_627_cast_fp16")]; tensor x_17_axes_0 = const()[name = string("x_17_axes_0"), val = tensor([1])]; tensor x_17_cast_fp16 = expand_dims(axes = x_17_axes_0, x = V_layer_cache_1_cast_fp16)[name = string("x_17_cast_fp16")]; tensor var_629 = const()[name = string("op_629"), val = tensor([1, 4, 1, 1])]; tensor x_19_cast_fp16 = tile(reps = var_629, x = x_17_cast_fp16)[name = string("x_19_cast_fp16")]; bool var_636_transpose_x_0 = const()[name = string("op_636_transpose_x_0"), val = bool(false)]; bool var_636_transpose_y_0 = const()[name = string("op_636_transpose_y_0"), val = bool(true)]; tensor var_636_cast_fp16 = matmul(transpose_x = var_636_transpose_x_0, transpose_y = var_636_transpose_y_0, x = rotated_1, y = var_627_cast_fp16)[name = string("op_636_cast_fp16")]; fp16 var_637_to_fp16 = const()[name = string("op_637_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_1_cast_fp16 = mul(x = var_636_cast_fp16, y = var_637_to_fp16)[name = string("attn_weights_1_cast_fp16")]; tensor x_21_cast_fp16 = add(x = attn_weights_1_cast_fp16, y = causal_mask)[name = string("x_21_cast_fp16")]; tensor reduce_max_0_axes_0 = const()[name = string("reduce_max_0_axes_0"), val = tensor([-1])]; bool reduce_max_0_keep_dims_0 = const()[name = string("reduce_max_0_keep_dims_0"), val = bool(true)]; tensor reduce_max_0_cast_fp16 = reduce_max(axes = reduce_max_0_axes_0, keep_dims = reduce_max_0_keep_dims_0, x = x_21_cast_fp16)[name = string("reduce_max_0_cast_fp16")]; tensor x_23_cast_fp16 = sub(x = x_21_cast_fp16, y = reduce_max_0_cast_fp16)[name = string("x_23_cast_fp16")]; tensor exp_x_1_cast_fp16 = exp(x = x_23_cast_fp16)[name = string("exp_x_1_cast_fp16")]; tensor var_648_axes_0 = const()[name = string("op_648_axes_0"), val = tensor([-1])]; bool var_648_keep_dims_0 = const()[name = string("op_648_keep_dims_0"), val = bool(true)]; tensor var_648_cast_fp16 = reduce_sum(axes = var_648_axes_0, keep_dims = var_648_keep_dims_0, x = exp_x_1_cast_fp16)[name = string("op_648_cast_fp16")]; tensor var_649_cast_fp16 = real_div(x = exp_x_1_cast_fp16, y = var_648_cast_fp16)[name = string("op_649_cast_fp16")]; tensor concat_12 = const()[name = string("concat_12"), val = tensor([32, 64, 4096])]; tensor reshape_0_cast_fp16 = reshape(shape = concat_12, x = var_649_cast_fp16)[name = string("reshape_0_cast_fp16")]; tensor concat_13 = const()[name = string("concat_13"), val = tensor([32, 4096, 64])]; tensor reshape_1_cast_fp16 = reshape(shape = concat_13, x = x_19_cast_fp16)[name = string("reshape_1_cast_fp16")]; bool matmul_0_transpose_x_0 = const()[name = string("matmul_0_transpose_x_0"), val = bool(false)]; bool matmul_0_transpose_y_0 = const()[name = string("matmul_0_transpose_y_0"), val = bool(false)]; tensor matmul_0_cast_fp16 = matmul(transpose_x = matmul_0_transpose_x_0, transpose_y = matmul_0_transpose_y_0, x = reshape_0_cast_fp16, y = reshape_1_cast_fp16)[name = string("matmul_0_cast_fp16")]; tensor concat_17 = const()[name = string("concat_17"), val = tensor([1, 32, 64, 64])]; tensor reshape_2_cast_fp16 = reshape(shape = concat_17, x = matmul_0_cast_fp16)[name = string("reshape_2_cast_fp16")]; tensor var_652_perm_0 = const()[name = string("op_652_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_654 = const()[name = string("op_654"), val = tensor([1, 64, 2048])]; tensor var_652_cast_fp16 = transpose(perm = var_652_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_105")]; tensor input_5_cast_fp16 = reshape(shape = var_654, x = var_652_cast_fp16)[name = string("input_5_cast_fp16")]; tensor model_model_layers_0_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(456474880))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(458572096))))[name = string("model_model_layers_0_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_0_bias_0_to_fp16 = const()[name = string("linear_0_bias_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(458580352)))]; tensor linear_0_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_0_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_5_cast_fp16)[name = string("linear_0_cast_fp16")]; tensor hidden_states_5_cast_fp16 = add(x = hidden_states, y = linear_0_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; fp16 const_15_promoted_to_fp16 = const()[name = string("const_15_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_660_cast_fp16 = mul(x = hidden_states_5_cast_fp16, y = const_15_promoted_to_fp16)[name = string("op_660_cast_fp16")]; bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; tensor input_7_cast_fp16 = concat(axis = var_73, interleave = input_7_interleave_0, values = (hidden_states_5_cast_fp16, var_660_cast_fp16))[name = string("input_7_cast_fp16")]; tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_76_to_fp16, x = input_7_cast_fp16)[name = string("normed_5_cast_fp16")]; tensor normed_7_begin_0 = const()[name = string("normed_7_begin_0"), val = tensor([0, 0, 0])]; tensor normed_7_end_0 = const()[name = string("normed_7_end_0"), val = tensor([1, 64, 2048])]; tensor normed_7_end_mask_0 = const()[name = string("normed_7_end_mask_0"), val = tensor([true, true, false])]; tensor normed_7_cast_fp16 = slice_by_index(begin = normed_7_begin_0, end = normed_7_end_0, end_mask = normed_7_end_mask_0, x = normed_5_cast_fp16)[name = string("normed_7_cast_fp16")]; tensor const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(458584512)))]; tensor x_25_cast_fp16 = mul(x = normed_7_cast_fp16, y = const_18_promoted_to_fp16)[name = string("x_25_cast_fp16")]; tensor var_678 = const()[name = string("op_678"), val = tensor([0, 2, 1])]; tensor input_9_axes_0 = const()[name = string("input_9_axes_0"), val = tensor([2])]; tensor var_679 = transpose(perm = var_678, x = x_25_cast_fp16)[name = string("transpose_104")]; tensor input_9 = expand_dims(axes = input_9_axes_0, x = var_679)[name = string("input_9")]; string input_11_pad_type_0 = const()[name = string("input_11_pad_type_0"), val = string("valid")]; tensor input_11_strides_0 = const()[name = string("input_11_strides_0"), val = tensor([1, 1])]; tensor input_11_pad_0 = const()[name = string("input_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_11_dilations_0 = const()[name = string("input_11_dilations_0"), val = tensor([1, 1])]; int32 input_11_groups_0 = const()[name = string("input_11_groups_0"), val = int32(1)]; tensor input_11 = conv(dilations = input_11_dilations_0, groups = input_11_groups_0, pad = input_11_pad_0, pad_type = input_11_pad_type_0, strides = input_11_strides_0, weight = model_model_layers_0_mlp_gate_proj_weight_palettized, x = input_9)[name = string("input_11")]; string up_states_1_pad_type_0 = const()[name = string("up_states_1_pad_type_0"), val = string("valid")]; tensor up_states_1_strides_0 = const()[name = string("up_states_1_strides_0"), val = tensor([1, 1])]; tensor up_states_1_pad_0 = const()[name = string("up_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_1_dilations_0 = const()[name = string("up_states_1_dilations_0"), val = tensor([1, 1])]; int32 up_states_1_groups_0 = const()[name = string("up_states_1_groups_0"), val = int32(1)]; tensor up_states_1 = conv(dilations = up_states_1_dilations_0, groups = up_states_1_groups_0, pad = up_states_1_pad_0, pad_type = up_states_1_pad_type_0, strides = up_states_1_strides_0, weight = model_model_layers_0_mlp_up_proj_weight_palettized, x = input_9)[name = string("up_states_1")]; tensor gate_states_1 = silu(x = input_11)[name = string("gate_states_1")]; tensor input_13 = mul(x = gate_states_1, y = up_states_1)[name = string("input_13")]; string hidden_states_7_pad_type_0 = const()[name = string("hidden_states_7_pad_type_0"), val = string("valid")]; tensor hidden_states_7_strides_0 = const()[name = string("hidden_states_7_strides_0"), val = tensor([1, 1])]; tensor hidden_states_7_pad_0 = const()[name = string("hidden_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_7_dilations_0 = const()[name = string("hidden_states_7_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_7_groups_0 = const()[name = string("hidden_states_7_groups_0"), val = int32(1)]; tensor hidden_states_7 = conv(dilations = hidden_states_7_dilations_0, groups = hidden_states_7_groups_0, pad = hidden_states_7_pad_0, pad_type = hidden_states_7_pad_type_0, strides = hidden_states_7_strides_0, weight = model_model_layers_0_mlp_down_proj_weight_palettized, x = input_13)[name = string("hidden_states_7")]; tensor var_701_axes_0 = const()[name = string("op_701_axes_0"), val = tensor([2])]; tensor var_701 = squeeze(axes = var_701_axes_0, x = hidden_states_7)[name = string("op_701")]; tensor var_702 = const()[name = string("op_702"), val = tensor([0, 2, 1])]; tensor var_703 = transpose(perm = var_702, x = var_701)[name = string("transpose_103")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = var_703)[name = string("hidden_states_9_cast_fp16")]; fp16 const_19_promoted_to_fp16 = const()[name = string("const_19_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_706_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_19_promoted_to_fp16)[name = string("op_706_cast_fp16")]; bool input_15_interleave_0 = const()[name = string("input_15_interleave_0"), val = bool(false)]; tensor input_15_cast_fp16 = concat(axis = var_73, interleave = input_15_interleave_0, values = (hidden_states_9_cast_fp16, var_706_cast_fp16))[name = string("input_15_cast_fp16")]; tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_76_to_fp16, x = input_15_cast_fp16)[name = string("normed_9_cast_fp16")]; tensor normed_11_begin_0 = const()[name = string("normed_11_begin_0"), val = tensor([0, 0, 0])]; tensor normed_11_end_0 = const()[name = string("normed_11_end_0"), val = tensor([1, 64, 2048])]; tensor normed_11_end_mask_0 = const()[name = string("normed_11_end_mask_0"), val = tensor([true, true, false])]; tensor normed_11_cast_fp16 = slice_by_index(begin = normed_11_begin_0, end = normed_11_end_0, end_mask = normed_11_end_mask_0, x = normed_9_cast_fp16)[name = string("normed_11_cast_fp16")]; tensor const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(458588672)))]; tensor hidden_states_11_cast_fp16 = mul(x = normed_11_cast_fp16, y = const_22_promoted_to_fp16)[name = string("hidden_states_11_cast_fp16")]; tensor var_721 = const()[name = string("op_721"), val = tensor([0, 2, 1])]; tensor var_723_axes_0 = const()[name = string("op_723_axes_0"), val = tensor([2])]; tensor var_722_cast_fp16 = transpose(perm = var_721, x = hidden_states_11_cast_fp16)[name = string("transpose_102")]; tensor var_723_cast_fp16 = expand_dims(axes = var_723_axes_0, x = var_722_cast_fp16)[name = string("op_723_cast_fp16")]; string query_states_5_pad_type_0 = const()[name = string("query_states_5_pad_type_0"), val = string("valid")]; tensor query_states_5_strides_0 = const()[name = string("query_states_5_strides_0"), val = tensor([1, 1])]; tensor query_states_5_pad_0 = const()[name = string("query_states_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_5_dilations_0 = const()[name = string("query_states_5_dilations_0"), val = tensor([1, 1])]; int32 query_states_5_groups_0 = const()[name = string("query_states_5_groups_0"), val = int32(1)]; tensor query_states_5 = conv(dilations = query_states_5_dilations_0, groups = query_states_5_groups_0, pad = query_states_5_pad_0, pad_type = query_states_5_pad_type_0, strides = query_states_5_strides_0, weight = model_model_layers_1_self_attn_q_proj_weight_palettized, x = var_723_cast_fp16)[name = string("query_states_5")]; string key_states_7_pad_type_0 = const()[name = string("key_states_7_pad_type_0"), val = string("valid")]; tensor key_states_7_strides_0 = const()[name = string("key_states_7_strides_0"), val = tensor([1, 1])]; tensor key_states_7_pad_0 = const()[name = string("key_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_7_dilations_0 = const()[name = string("key_states_7_dilations_0"), val = tensor([1, 1])]; int32 key_states_7_groups_0 = const()[name = string("key_states_7_groups_0"), val = int32(1)]; tensor key_states_7 = conv(dilations = key_states_7_dilations_0, groups = key_states_7_groups_0, pad = key_states_7_pad_0, pad_type = key_states_7_pad_type_0, strides = key_states_7_strides_0, weight = model_model_layers_1_self_attn_k_proj_weight_palettized, x = var_723_cast_fp16)[name = string("key_states_7")]; string value_states_7_pad_type_0 = const()[name = string("value_states_7_pad_type_0"), val = string("valid")]; tensor value_states_7_strides_0 = const()[name = string("value_states_7_strides_0"), val = tensor([1, 1])]; tensor value_states_7_pad_0 = const()[name = string("value_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_7_dilations_0 = const()[name = string("value_states_7_dilations_0"), val = tensor([1, 1])]; int32 value_states_7_groups_0 = const()[name = string("value_states_7_groups_0"), val = int32(1)]; tensor value_states_7 = conv(dilations = value_states_7_dilations_0, groups = value_states_7_groups_0, pad = value_states_7_pad_0, pad_type = value_states_7_pad_type_0, strides = value_states_7_strides_0, weight = model_model_layers_1_self_attn_v_proj_weight_palettized, x = var_723_cast_fp16)[name = string("value_states_7")]; tensor var_743 = const()[name = string("op_743"), val = tensor([1, 32, 64, 64])]; tensor var_744 = reshape(shape = var_743, x = query_states_5)[name = string("op_744")]; tensor var_745 = const()[name = string("op_745"), val = tensor([0, 1, 3, 2])]; tensor var_747 = const()[name = string("op_747"), val = tensor([1, 8, 64, 64])]; tensor var_748 = reshape(shape = var_747, x = key_states_7)[name = string("op_748")]; tensor var_749 = const()[name = string("op_749"), val = tensor([0, 1, 3, 2])]; tensor var_751 = const()[name = string("op_751"), val = tensor([1, 8, 64, 64])]; tensor var_752 = reshape(shape = var_751, x = value_states_7)[name = string("op_752")]; tensor var_753 = const()[name = string("op_753"), val = tensor([0, 1, 3, 2])]; tensor x1_5_begin_0 = const()[name = string("x1_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_5_end_0 = const()[name = string("x1_5_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_5_end_mask_0 = const()[name = string("x1_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_29 = transpose(perm = var_745, x = var_744)[name = string("transpose_101")]; tensor x1_5 = slice_by_index(begin = x1_5_begin_0, end = x1_5_end_0, end_mask = x1_5_end_mask_0, x = x_29)[name = string("x1_5")]; tensor x2_5_begin_0 = const()[name = string("x2_5_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_5_end_0 = const()[name = string("x2_5_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_5_end_mask_0 = const()[name = string("x2_5_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_5 = slice_by_index(begin = x2_5_begin_0, end = x2_5_end_0, end_mask = x2_5_end_mask_0, x = x_29)[name = string("x2_5")]; tensor var_771 = mul(x = x1_5, y = cos_7)[name = string("op_771")]; tensor var_772 = mul(x = x2_5, y = sin_7)[name = string("op_772")]; tensor var_773 = sub(x = var_771, y = var_772)[name = string("op_773")]; tensor var_774 = mul(x = x2_5, y = cos_7)[name = string("op_774")]; tensor var_775 = mul(x = x1_5, y = sin_7)[name = string("op_775")]; tensor var_776 = add(x = var_774, y = var_775)[name = string("op_776")]; bool rotated_5_interleave_0 = const()[name = string("rotated_5_interleave_0"), val = bool(false)]; tensor rotated_5 = concat(axis = var_73, interleave = rotated_5_interleave_0, values = (var_773, var_776))[name = string("rotated_5")]; tensor x1_7_begin_0 = const()[name = string("x1_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_7_end_0 = const()[name = string("x1_7_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_7_end_mask_0 = const()[name = string("x1_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_33 = transpose(perm = var_749, x = var_748)[name = string("transpose_100")]; tensor x1_7 = slice_by_index(begin = x1_7_begin_0, end = x1_7_end_0, end_mask = x1_7_end_mask_0, x = x_33)[name = string("x1_7")]; tensor x2_7_begin_0 = const()[name = string("x2_7_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_7_end_0 = const()[name = string("x2_7_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_7_end_mask_0 = const()[name = string("x2_7_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_7 = slice_by_index(begin = x2_7_begin_0, end = x2_7_end_0, end_mask = x2_7_end_mask_0, x = x_33)[name = string("x2_7")]; tensor var_792 = mul(x = x1_7, y = cos_7)[name = string("op_792")]; tensor var_793 = mul(x = x2_7, y = sin_7)[name = string("op_793")]; tensor var_794 = sub(x = var_792, y = var_793)[name = string("op_794")]; tensor var_795 = mul(x = x2_7, y = cos_7)[name = string("op_795")]; tensor var_796 = mul(x = x1_7, y = sin_7)[name = string("op_796")]; tensor var_797 = add(x = var_795, y = var_796)[name = string("op_797")]; bool rotated_7_interleave_0 = const()[name = string("rotated_7_interleave_0"), val = bool(false)]; tensor rotated_7 = concat(axis = var_73, interleave = rotated_7_interleave_0, values = (var_794, var_797))[name = string("rotated_7")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; int32 concat_20_axis_0 = const()[name = string("concat_20_axis_0"), val = int32(0)]; bool concat_20_interleave_0 = const()[name = string("concat_20_interleave_0"), val = bool(false)]; tensor concat_20 = concat(axis = concat_20_axis_0, interleave = concat_20_interleave_0, values = (expand_dims_12, expand_dims_13, current_pos, expand_dims_15))[name = string("concat_20")]; tensor concat_21_values1_0 = const()[name = string("concat_21_values1_0"), val = tensor([0])]; tensor concat_21_values3_0 = const()[name = string("concat_21_values3_0"), val = tensor([0])]; int32 concat_21_axis_0 = const()[name = string("concat_21_axis_0"), val = int32(0)]; bool concat_21_interleave_0 = const()[name = string("concat_21_interleave_0"), val = bool(false)]; tensor concat_21 = concat(axis = concat_21_axis_0, interleave = concat_21_interleave_0, values = (expand_dims_16, concat_21_values1_0, var_597, concat_21_values3_0))[name = string("concat_21")]; tensor model_model_kv_cache_0_internal_tensor_assign_3_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_20, begin_mask = model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0, end = concat_21, end_mask = model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_3_stride_0, update = rotated_7, x = coreml_update_state_33)[name = string("model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_98_write_state")]; tensor coreml_update_state_34 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_98")]; tensor expand_dims_18 = const()[name = string("expand_dims_18"), val = tensor([17])]; tensor expand_dims_19 = const()[name = string("expand_dims_19"), val = tensor([0])]; tensor expand_dims_21 = const()[name = string("expand_dims_21"), val = tensor([0])]; tensor expand_dims_22 = const()[name = string("expand_dims_22"), val = tensor([18])]; int32 concat_24_axis_0 = const()[name = string("concat_24_axis_0"), val = int32(0)]; bool concat_24_interleave_0 = const()[name = string("concat_24_interleave_0"), val = bool(false)]; tensor concat_24 = concat(axis = concat_24_axis_0, interleave = concat_24_interleave_0, values = (expand_dims_18, expand_dims_19, current_pos, expand_dims_21))[name = string("concat_24")]; tensor concat_25_values1_0 = const()[name = string("concat_25_values1_0"), val = tensor([0])]; tensor concat_25_values3_0 = const()[name = string("concat_25_values3_0"), val = tensor([0])]; int32 concat_25_axis_0 = const()[name = string("concat_25_axis_0"), val = int32(0)]; bool concat_25_interleave_0 = const()[name = string("concat_25_interleave_0"), val = bool(false)]; tensor concat_25 = concat(axis = concat_25_axis_0, interleave = concat_25_interleave_0, values = (expand_dims_22, concat_25_values1_0, var_597, concat_25_values3_0))[name = string("concat_25")]; tensor model_model_kv_cache_0_internal_tensor_assign_4_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_9 = transpose(perm = var_753, x = var_752)[name = string("transpose_99")]; tensor model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_24, begin_mask = model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0, end = concat_25, end_mask = model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_4_stride_0, update = value_states_9, x = coreml_update_state_34)[name = string("model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_99_write_state")]; tensor coreml_update_state_35 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_99")]; tensor var_820_begin_0 = const()[name = string("op_820_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_820_end_0 = const()[name = string("op_820_end_0"), val = tensor([2, 8, 4096, 64])]; tensor var_820_end_mask_0 = const()[name = string("op_820_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_820_cast_fp16 = slice_by_index(begin = var_820_begin_0, end = var_820_end_0, end_mask = var_820_end_mask_0, x = coreml_update_state_35)[name = string("op_820_cast_fp16")]; tensor K_layer_cache_3_axes_0 = const()[name = string("K_layer_cache_3_axes_0"), val = tensor([0])]; tensor K_layer_cache_3_cast_fp16 = squeeze(axes = K_layer_cache_3_axes_0, x = var_820_cast_fp16)[name = string("K_layer_cache_3_cast_fp16")]; tensor var_822_begin_0 = const()[name = string("op_822_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_822_end_0 = const()[name = string("op_822_end_0"), val = tensor([18, 8, 4096, 64])]; tensor var_822_end_mask_0 = const()[name = string("op_822_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_822_cast_fp16 = slice_by_index(begin = var_822_begin_0, end = var_822_end_0, end_mask = var_822_end_mask_0, x = coreml_update_state_35)[name = string("op_822_cast_fp16")]; tensor V_layer_cache_3_axes_0 = const()[name = string("V_layer_cache_3_axes_0"), val = tensor([0])]; tensor V_layer_cache_3_cast_fp16 = squeeze(axes = V_layer_cache_3_axes_0, x = var_822_cast_fp16)[name = string("V_layer_cache_3_cast_fp16")]; tensor x_39_axes_0 = const()[name = string("x_39_axes_0"), val = tensor([1])]; tensor x_39_cast_fp16 = expand_dims(axes = x_39_axes_0, x = K_layer_cache_3_cast_fp16)[name = string("x_39_cast_fp16")]; tensor var_831 = const()[name = string("op_831"), val = tensor([1, 4, 1, 1])]; tensor x_41_cast_fp16 = tile(reps = var_831, x = x_39_cast_fp16)[name = string("x_41_cast_fp16")]; tensor var_835 = const()[name = string("op_835"), val = tensor([1, -1, 4096, 64])]; tensor var_836_cast_fp16 = reshape(shape = var_835, x = x_41_cast_fp16)[name = string("op_836_cast_fp16")]; tensor x_45_axes_0 = const()[name = string("x_45_axes_0"), val = tensor([1])]; tensor x_45_cast_fp16 = expand_dims(axes = x_45_axes_0, x = V_layer_cache_3_cast_fp16)[name = string("x_45_cast_fp16")]; tensor var_838 = const()[name = string("op_838"), val = tensor([1, 4, 1, 1])]; tensor x_47_cast_fp16 = tile(reps = var_838, x = x_45_cast_fp16)[name = string("x_47_cast_fp16")]; bool var_845_transpose_x_0 = const()[name = string("op_845_transpose_x_0"), val = bool(false)]; bool var_845_transpose_y_0 = const()[name = string("op_845_transpose_y_0"), val = bool(true)]; tensor var_845_cast_fp16 = matmul(transpose_x = var_845_transpose_x_0, transpose_y = var_845_transpose_y_0, x = rotated_5, y = var_836_cast_fp16)[name = string("op_845_cast_fp16")]; fp16 var_846_to_fp16 = const()[name = string("op_846_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_3_cast_fp16 = mul(x = var_845_cast_fp16, y = var_846_to_fp16)[name = string("attn_weights_3_cast_fp16")]; tensor x_49_cast_fp16 = add(x = attn_weights_3_cast_fp16, y = causal_mask)[name = string("x_49_cast_fp16")]; tensor reduce_max_1_axes_0 = const()[name = string("reduce_max_1_axes_0"), val = tensor([-1])]; bool reduce_max_1_keep_dims_0 = const()[name = string("reduce_max_1_keep_dims_0"), val = bool(true)]; tensor reduce_max_1_cast_fp16 = reduce_max(axes = reduce_max_1_axes_0, keep_dims = reduce_max_1_keep_dims_0, x = x_49_cast_fp16)[name = string("reduce_max_1_cast_fp16")]; tensor x_51_cast_fp16 = sub(x = x_49_cast_fp16, y = reduce_max_1_cast_fp16)[name = string("x_51_cast_fp16")]; tensor exp_x_3_cast_fp16 = exp(x = x_51_cast_fp16)[name = string("exp_x_3_cast_fp16")]; tensor var_857_axes_0 = const()[name = string("op_857_axes_0"), val = tensor([-1])]; bool var_857_keep_dims_0 = const()[name = string("op_857_keep_dims_0"), val = bool(true)]; tensor var_857_cast_fp16 = reduce_sum(axes = var_857_axes_0, keep_dims = var_857_keep_dims_0, x = exp_x_3_cast_fp16)[name = string("op_857_cast_fp16")]; tensor var_858_cast_fp16 = real_div(x = exp_x_3_cast_fp16, y = var_857_cast_fp16)[name = string("op_858_cast_fp16")]; tensor concat_30 = const()[name = string("concat_30"), val = tensor([32, 64, 4096])]; tensor reshape_3_cast_fp16 = reshape(shape = concat_30, x = var_858_cast_fp16)[name = string("reshape_3_cast_fp16")]; tensor concat_31 = const()[name = string("concat_31"), val = tensor([32, 4096, 64])]; tensor reshape_4_cast_fp16 = reshape(shape = concat_31, x = x_47_cast_fp16)[name = string("reshape_4_cast_fp16")]; bool matmul_1_transpose_x_0 = const()[name = string("matmul_1_transpose_x_0"), val = bool(false)]; bool matmul_1_transpose_y_0 = const()[name = string("matmul_1_transpose_y_0"), val = bool(false)]; tensor matmul_1_cast_fp16 = matmul(transpose_x = matmul_1_transpose_x_0, transpose_y = matmul_1_transpose_y_0, x = reshape_3_cast_fp16, y = reshape_4_cast_fp16)[name = string("matmul_1_cast_fp16")]; tensor concat_35 = const()[name = string("concat_35"), val = tensor([1, 32, 64, 64])]; tensor reshape_5_cast_fp16 = reshape(shape = concat_35, x = matmul_1_cast_fp16)[name = string("reshape_5_cast_fp16")]; tensor var_861_perm_0 = const()[name = string("op_861_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_863 = const()[name = string("op_863"), val = tensor([1, 64, 2048])]; tensor var_861_cast_fp16 = transpose(perm = var_861_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_98")]; tensor input_19_cast_fp16 = reshape(shape = var_863, x = var_861_cast_fp16)[name = string("input_19_cast_fp16")]; tensor model_model_layers_1_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(458592832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460690048))))[name = string("model_model_layers_1_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_1_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_1_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_19_cast_fp16)[name = string("linear_1_cast_fp16")]; tensor hidden_states_13_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = linear_1_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_869_cast_fp16 = mul(x = hidden_states_13_cast_fp16, y = const_33_promoted_to_fp16)[name = string("op_869_cast_fp16")]; bool input_21_interleave_0 = const()[name = string("input_21_interleave_0"), val = bool(false)]; tensor input_21_cast_fp16 = concat(axis = var_73, interleave = input_21_interleave_0, values = (hidden_states_13_cast_fp16, var_869_cast_fp16))[name = string("input_21_cast_fp16")]; tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_76_to_fp16, x = input_21_cast_fp16)[name = string("normed_13_cast_fp16")]; tensor normed_15_begin_0 = const()[name = string("normed_15_begin_0"), val = tensor([0, 0, 0])]; tensor normed_15_end_0 = const()[name = string("normed_15_end_0"), val = tensor([1, 64, 2048])]; tensor normed_15_end_mask_0 = const()[name = string("normed_15_end_mask_0"), val = tensor([true, true, false])]; tensor normed_15_cast_fp16 = slice_by_index(begin = normed_15_begin_0, end = normed_15_end_0, end_mask = normed_15_end_mask_0, x = normed_13_cast_fp16)[name = string("normed_15_cast_fp16")]; tensor const_36_promoted_to_fp16 = const()[name = string("const_36_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460698304)))]; tensor x_53_cast_fp16 = mul(x = normed_15_cast_fp16, y = const_36_promoted_to_fp16)[name = string("x_53_cast_fp16")]; tensor var_887 = const()[name = string("op_887"), val = tensor([0, 2, 1])]; tensor input_23_axes_0 = const()[name = string("input_23_axes_0"), val = tensor([2])]; tensor var_888 = transpose(perm = var_887, x = x_53_cast_fp16)[name = string("transpose_97")]; tensor input_23 = expand_dims(axes = input_23_axes_0, x = var_888)[name = string("input_23")]; string input_25_pad_type_0 = const()[name = string("input_25_pad_type_0"), val = string("valid")]; tensor input_25_strides_0 = const()[name = string("input_25_strides_0"), val = tensor([1, 1])]; tensor input_25_pad_0 = const()[name = string("input_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_25_dilations_0 = const()[name = string("input_25_dilations_0"), val = tensor([1, 1])]; int32 input_25_groups_0 = const()[name = string("input_25_groups_0"), val = int32(1)]; tensor input_25 = conv(dilations = input_25_dilations_0, groups = input_25_groups_0, pad = input_25_pad_0, pad_type = input_25_pad_type_0, strides = input_25_strides_0, weight = model_model_layers_1_mlp_gate_proj_weight_palettized, x = input_23)[name = string("input_25")]; string up_states_3_pad_type_0 = const()[name = string("up_states_3_pad_type_0"), val = string("valid")]; tensor up_states_3_strides_0 = const()[name = string("up_states_3_strides_0"), val = tensor([1, 1])]; tensor up_states_3_pad_0 = const()[name = string("up_states_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_3_dilations_0 = const()[name = string("up_states_3_dilations_0"), val = tensor([1, 1])]; int32 up_states_3_groups_0 = const()[name = string("up_states_3_groups_0"), val = int32(1)]; tensor up_states_3 = conv(dilations = up_states_3_dilations_0, groups = up_states_3_groups_0, pad = up_states_3_pad_0, pad_type = up_states_3_pad_type_0, strides = up_states_3_strides_0, weight = model_model_layers_1_mlp_up_proj_weight_palettized, x = input_23)[name = string("up_states_3")]; tensor gate_states_3 = silu(x = input_25)[name = string("gate_states_3")]; tensor input_27 = mul(x = gate_states_3, y = up_states_3)[name = string("input_27")]; string hidden_states_15_pad_type_0 = const()[name = string("hidden_states_15_pad_type_0"), val = string("valid")]; tensor hidden_states_15_strides_0 = const()[name = string("hidden_states_15_strides_0"), val = tensor([1, 1])]; tensor hidden_states_15_pad_0 = const()[name = string("hidden_states_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_15_dilations_0 = const()[name = string("hidden_states_15_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_15_groups_0 = const()[name = string("hidden_states_15_groups_0"), val = int32(1)]; tensor hidden_states_15 = conv(dilations = hidden_states_15_dilations_0, groups = hidden_states_15_groups_0, pad = hidden_states_15_pad_0, pad_type = hidden_states_15_pad_type_0, strides = hidden_states_15_strides_0, weight = model_model_layers_1_mlp_down_proj_weight_palettized, x = input_27)[name = string("hidden_states_15")]; tensor var_910_axes_0 = const()[name = string("op_910_axes_0"), val = tensor([2])]; tensor var_910 = squeeze(axes = var_910_axes_0, x = hidden_states_15)[name = string("op_910")]; tensor var_911 = const()[name = string("op_911"), val = tensor([0, 2, 1])]; tensor var_912 = transpose(perm = var_911, x = var_910)[name = string("transpose_96")]; tensor hidden_states_17_cast_fp16 = add(x = hidden_states_13_cast_fp16, y = var_912)[name = string("hidden_states_17_cast_fp16")]; fp16 const_37_promoted_to_fp16 = const()[name = string("const_37_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_915_cast_fp16 = mul(x = hidden_states_17_cast_fp16, y = const_37_promoted_to_fp16)[name = string("op_915_cast_fp16")]; bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; tensor input_29_cast_fp16 = concat(axis = var_73, interleave = input_29_interleave_0, values = (hidden_states_17_cast_fp16, var_915_cast_fp16))[name = string("input_29_cast_fp16")]; tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_76_to_fp16, x = input_29_cast_fp16)[name = string("normed_17_cast_fp16")]; tensor normed_19_begin_0 = const()[name = string("normed_19_begin_0"), val = tensor([0, 0, 0])]; tensor normed_19_end_0 = const()[name = string("normed_19_end_0"), val = tensor([1, 64, 2048])]; tensor normed_19_end_mask_0 = const()[name = string("normed_19_end_mask_0"), val = tensor([true, true, false])]; tensor normed_19_cast_fp16 = slice_by_index(begin = normed_19_begin_0, end = normed_19_end_0, end_mask = normed_19_end_mask_0, x = normed_17_cast_fp16)[name = string("normed_19_cast_fp16")]; tensor const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460702464)))]; tensor hidden_states_19_cast_fp16 = mul(x = normed_19_cast_fp16, y = const_40_promoted_to_fp16)[name = string("hidden_states_19_cast_fp16")]; tensor var_930 = const()[name = string("op_930"), val = tensor([0, 2, 1])]; tensor var_932_axes_0 = const()[name = string("op_932_axes_0"), val = tensor([2])]; tensor var_931_cast_fp16 = transpose(perm = var_930, x = hidden_states_19_cast_fp16)[name = string("transpose_95")]; tensor var_932_cast_fp16 = expand_dims(axes = var_932_axes_0, x = var_931_cast_fp16)[name = string("op_932_cast_fp16")]; string query_states_9_pad_type_0 = const()[name = string("query_states_9_pad_type_0"), val = string("valid")]; tensor query_states_9_strides_0 = const()[name = string("query_states_9_strides_0"), val = tensor([1, 1])]; tensor query_states_9_pad_0 = const()[name = string("query_states_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_9_dilations_0 = const()[name = string("query_states_9_dilations_0"), val = tensor([1, 1])]; int32 query_states_9_groups_0 = const()[name = string("query_states_9_groups_0"), val = int32(1)]; tensor query_states_9 = conv(dilations = query_states_9_dilations_0, groups = query_states_9_groups_0, pad = query_states_9_pad_0, pad_type = query_states_9_pad_type_0, strides = query_states_9_strides_0, weight = model_model_layers_2_self_attn_q_proj_weight_palettized, x = var_932_cast_fp16)[name = string("query_states_9")]; string key_states_13_pad_type_0 = const()[name = string("key_states_13_pad_type_0"), val = string("valid")]; tensor key_states_13_strides_0 = const()[name = string("key_states_13_strides_0"), val = tensor([1, 1])]; tensor key_states_13_pad_0 = const()[name = string("key_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_13_dilations_0 = const()[name = string("key_states_13_dilations_0"), val = tensor([1, 1])]; int32 key_states_13_groups_0 = const()[name = string("key_states_13_groups_0"), val = int32(1)]; tensor key_states_13 = conv(dilations = key_states_13_dilations_0, groups = key_states_13_groups_0, pad = key_states_13_pad_0, pad_type = key_states_13_pad_type_0, strides = key_states_13_strides_0, weight = model_model_layers_2_self_attn_k_proj_weight_palettized, x = var_932_cast_fp16)[name = string("key_states_13")]; string value_states_13_pad_type_0 = const()[name = string("value_states_13_pad_type_0"), val = string("valid")]; tensor value_states_13_strides_0 = const()[name = string("value_states_13_strides_0"), val = tensor([1, 1])]; tensor value_states_13_pad_0 = const()[name = string("value_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_13_dilations_0 = const()[name = string("value_states_13_dilations_0"), val = tensor([1, 1])]; int32 value_states_13_groups_0 = const()[name = string("value_states_13_groups_0"), val = int32(1)]; tensor value_states_13 = conv(dilations = value_states_13_dilations_0, groups = value_states_13_groups_0, pad = value_states_13_pad_0, pad_type = value_states_13_pad_type_0, strides = value_states_13_strides_0, weight = model_model_layers_2_self_attn_v_proj_weight_palettized, x = var_932_cast_fp16)[name = string("value_states_13")]; tensor var_952 = const()[name = string("op_952"), val = tensor([1, 32, 64, 64])]; tensor var_953 = reshape(shape = var_952, x = query_states_9)[name = string("op_953")]; tensor var_954 = const()[name = string("op_954"), val = tensor([0, 1, 3, 2])]; tensor var_956 = const()[name = string("op_956"), val = tensor([1, 8, 64, 64])]; tensor var_957 = reshape(shape = var_956, x = key_states_13)[name = string("op_957")]; tensor var_958 = const()[name = string("op_958"), val = tensor([0, 1, 3, 2])]; tensor var_960 = const()[name = string("op_960"), val = tensor([1, 8, 64, 64])]; tensor var_961 = reshape(shape = var_960, x = value_states_13)[name = string("op_961")]; tensor var_962 = const()[name = string("op_962"), val = tensor([0, 1, 3, 2])]; tensor x1_9_begin_0 = const()[name = string("x1_9_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_9_end_0 = const()[name = string("x1_9_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_9_end_mask_0 = const()[name = string("x1_9_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_57 = transpose(perm = var_954, x = var_953)[name = string("transpose_94")]; tensor x1_9 = slice_by_index(begin = x1_9_begin_0, end = x1_9_end_0, end_mask = x1_9_end_mask_0, x = x_57)[name = string("x1_9")]; tensor x2_9_begin_0 = const()[name = string("x2_9_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_9_end_0 = const()[name = string("x2_9_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_9_end_mask_0 = const()[name = string("x2_9_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_9 = slice_by_index(begin = x2_9_begin_0, end = x2_9_end_0, end_mask = x2_9_end_mask_0, x = x_57)[name = string("x2_9")]; tensor var_980 = mul(x = x1_9, y = cos_7)[name = string("op_980")]; tensor var_981 = mul(x = x2_9, y = sin_7)[name = string("op_981")]; tensor var_982 = sub(x = var_980, y = var_981)[name = string("op_982")]; tensor var_983 = mul(x = x2_9, y = cos_7)[name = string("op_983")]; tensor var_984 = mul(x = x1_9, y = sin_7)[name = string("op_984")]; tensor var_985 = add(x = var_983, y = var_984)[name = string("op_985")]; bool rotated_9_interleave_0 = const()[name = string("rotated_9_interleave_0"), val = bool(false)]; tensor rotated_9 = concat(axis = var_73, interleave = rotated_9_interleave_0, values = (var_982, var_985))[name = string("rotated_9")]; tensor x1_11_begin_0 = const()[name = string("x1_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_11_end_0 = const()[name = string("x1_11_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_11_end_mask_0 = const()[name = string("x1_11_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_61 = transpose(perm = var_958, x = var_957)[name = string("transpose_93")]; tensor x1_11 = slice_by_index(begin = x1_11_begin_0, end = x1_11_end_0, end_mask = x1_11_end_mask_0, x = x_61)[name = string("x1_11")]; tensor x2_11_begin_0 = const()[name = string("x2_11_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_11_end_0 = const()[name = string("x2_11_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_11_end_mask_0 = const()[name = string("x2_11_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_11 = slice_by_index(begin = x2_11_begin_0, end = x2_11_end_0, end_mask = x2_11_end_mask_0, x = x_61)[name = string("x2_11")]; tensor var_1001 = mul(x = x1_11, y = cos_7)[name = string("op_1001")]; tensor var_1002 = mul(x = x2_11, y = sin_7)[name = string("op_1002")]; tensor var_1003 = sub(x = var_1001, y = var_1002)[name = string("op_1003")]; tensor var_1004 = mul(x = x2_11, y = cos_7)[name = string("op_1004")]; tensor var_1005 = mul(x = x1_11, y = sin_7)[name = string("op_1005")]; tensor var_1006 = add(x = var_1004, y = var_1005)[name = string("op_1006")]; bool rotated_11_interleave_0 = const()[name = string("rotated_11_interleave_0"), val = bool(false)]; tensor rotated_11 = concat(axis = var_73, interleave = rotated_11_interleave_0, values = (var_1003, var_1006))[name = string("rotated_11")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; int32 concat_38_axis_0 = const()[name = string("concat_38_axis_0"), val = int32(0)]; bool concat_38_interleave_0 = const()[name = string("concat_38_interleave_0"), val = bool(false)]; tensor concat_38 = concat(axis = concat_38_axis_0, interleave = concat_38_interleave_0, values = (expand_dims_24, expand_dims_25, current_pos, expand_dims_27))[name = string("concat_38")]; tensor concat_39_values1_0 = const()[name = string("concat_39_values1_0"), val = tensor([0])]; tensor concat_39_values3_0 = const()[name = string("concat_39_values3_0"), val = tensor([0])]; int32 concat_39_axis_0 = const()[name = string("concat_39_axis_0"), val = int32(0)]; bool concat_39_interleave_0 = const()[name = string("concat_39_interleave_0"), val = bool(false)]; tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (expand_dims_28, concat_39_values1_0, var_597, concat_39_values3_0))[name = string("concat_39")]; tensor model_model_kv_cache_0_internal_tensor_assign_5_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_38, begin_mask = model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0, end = concat_39, end_mask = model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_5_stride_0, update = rotated_11, x = coreml_update_state_35)[name = string("model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_100_write_state")]; tensor coreml_update_state_36 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_100")]; tensor expand_dims_30 = const()[name = string("expand_dims_30"), val = tensor([18])]; tensor expand_dims_31 = const()[name = string("expand_dims_31"), val = tensor([0])]; tensor expand_dims_33 = const()[name = string("expand_dims_33"), val = tensor([0])]; tensor expand_dims_34 = const()[name = string("expand_dims_34"), val = tensor([19])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_30, expand_dims_31, current_pos, expand_dims_33))[name = string("concat_42")]; tensor concat_43_values1_0 = const()[name = string("concat_43_values1_0"), val = tensor([0])]; tensor concat_43_values3_0 = const()[name = string("concat_43_values3_0"), val = tensor([0])]; int32 concat_43_axis_0 = const()[name = string("concat_43_axis_0"), val = int32(0)]; bool concat_43_interleave_0 = const()[name = string("concat_43_interleave_0"), val = bool(false)]; tensor concat_43 = concat(axis = concat_43_axis_0, interleave = concat_43_interleave_0, values = (expand_dims_34, concat_43_values1_0, var_597, concat_43_values3_0))[name = string("concat_43")]; tensor model_model_kv_cache_0_internal_tensor_assign_6_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_15 = transpose(perm = var_962, x = var_961)[name = string("transpose_92")]; tensor model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_42, begin_mask = model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0, end = concat_43, end_mask = model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_6_stride_0, update = value_states_15, x = coreml_update_state_36)[name = string("model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_101_write_state")]; tensor coreml_update_state_37 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_101")]; tensor var_1029_begin_0 = const()[name = string("op_1029_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1029_end_0 = const()[name = string("op_1029_end_0"), val = tensor([3, 8, 4096, 64])]; tensor var_1029_end_mask_0 = const()[name = string("op_1029_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1029_cast_fp16 = slice_by_index(begin = var_1029_begin_0, end = var_1029_end_0, end_mask = var_1029_end_mask_0, x = coreml_update_state_37)[name = string("op_1029_cast_fp16")]; tensor K_layer_cache_5_axes_0 = const()[name = string("K_layer_cache_5_axes_0"), val = tensor([0])]; tensor K_layer_cache_5_cast_fp16 = squeeze(axes = K_layer_cache_5_axes_0, x = var_1029_cast_fp16)[name = string("K_layer_cache_5_cast_fp16")]; tensor var_1031_begin_0 = const()[name = string("op_1031_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_1031_end_0 = const()[name = string("op_1031_end_0"), val = tensor([19, 8, 4096, 64])]; tensor var_1031_end_mask_0 = const()[name = string("op_1031_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1031_cast_fp16 = slice_by_index(begin = var_1031_begin_0, end = var_1031_end_0, end_mask = var_1031_end_mask_0, x = coreml_update_state_37)[name = string("op_1031_cast_fp16")]; tensor V_layer_cache_5_axes_0 = const()[name = string("V_layer_cache_5_axes_0"), val = tensor([0])]; tensor V_layer_cache_5_cast_fp16 = squeeze(axes = V_layer_cache_5_axes_0, x = var_1031_cast_fp16)[name = string("V_layer_cache_5_cast_fp16")]; tensor x_67_axes_0 = const()[name = string("x_67_axes_0"), val = tensor([1])]; tensor x_67_cast_fp16 = expand_dims(axes = x_67_axes_0, x = K_layer_cache_5_cast_fp16)[name = string("x_67_cast_fp16")]; tensor var_1040 = const()[name = string("op_1040"), val = tensor([1, 4, 1, 1])]; tensor x_69_cast_fp16 = tile(reps = var_1040, x = x_67_cast_fp16)[name = string("x_69_cast_fp16")]; tensor var_1044 = const()[name = string("op_1044"), val = tensor([1, -1, 4096, 64])]; tensor var_1045_cast_fp16 = reshape(shape = var_1044, x = x_69_cast_fp16)[name = string("op_1045_cast_fp16")]; tensor x_73_axes_0 = const()[name = string("x_73_axes_0"), val = tensor([1])]; tensor x_73_cast_fp16 = expand_dims(axes = x_73_axes_0, x = V_layer_cache_5_cast_fp16)[name = string("x_73_cast_fp16")]; tensor var_1047 = const()[name = string("op_1047"), val = tensor([1, 4, 1, 1])]; tensor x_75_cast_fp16 = tile(reps = var_1047, x = x_73_cast_fp16)[name = string("x_75_cast_fp16")]; bool var_1054_transpose_x_0 = const()[name = string("op_1054_transpose_x_0"), val = bool(false)]; bool var_1054_transpose_y_0 = const()[name = string("op_1054_transpose_y_0"), val = bool(true)]; tensor var_1054_cast_fp16 = matmul(transpose_x = var_1054_transpose_x_0, transpose_y = var_1054_transpose_y_0, x = rotated_9, y = var_1045_cast_fp16)[name = string("op_1054_cast_fp16")]; fp16 var_1055_to_fp16 = const()[name = string("op_1055_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_5_cast_fp16 = mul(x = var_1054_cast_fp16, y = var_1055_to_fp16)[name = string("attn_weights_5_cast_fp16")]; tensor x_77_cast_fp16 = add(x = attn_weights_5_cast_fp16, y = causal_mask)[name = string("x_77_cast_fp16")]; tensor reduce_max_2_axes_0 = const()[name = string("reduce_max_2_axes_0"), val = tensor([-1])]; bool reduce_max_2_keep_dims_0 = const()[name = string("reduce_max_2_keep_dims_0"), val = bool(true)]; tensor reduce_max_2_cast_fp16 = reduce_max(axes = reduce_max_2_axes_0, keep_dims = reduce_max_2_keep_dims_0, x = x_77_cast_fp16)[name = string("reduce_max_2_cast_fp16")]; tensor x_79_cast_fp16 = sub(x = x_77_cast_fp16, y = reduce_max_2_cast_fp16)[name = string("x_79_cast_fp16")]; tensor exp_x_5_cast_fp16 = exp(x = x_79_cast_fp16)[name = string("exp_x_5_cast_fp16")]; tensor var_1066_axes_0 = const()[name = string("op_1066_axes_0"), val = tensor([-1])]; bool var_1066_keep_dims_0 = const()[name = string("op_1066_keep_dims_0"), val = bool(true)]; tensor var_1066_cast_fp16 = reduce_sum(axes = var_1066_axes_0, keep_dims = var_1066_keep_dims_0, x = exp_x_5_cast_fp16)[name = string("op_1066_cast_fp16")]; tensor var_1067_cast_fp16 = real_div(x = exp_x_5_cast_fp16, y = var_1066_cast_fp16)[name = string("op_1067_cast_fp16")]; tensor concat_48 = const()[name = string("concat_48"), val = tensor([32, 64, 4096])]; tensor reshape_6_cast_fp16 = reshape(shape = concat_48, x = var_1067_cast_fp16)[name = string("reshape_6_cast_fp16")]; tensor concat_49 = const()[name = string("concat_49"), val = tensor([32, 4096, 64])]; tensor reshape_7_cast_fp16 = reshape(shape = concat_49, x = x_75_cast_fp16)[name = string("reshape_7_cast_fp16")]; bool matmul_2_transpose_x_0 = const()[name = string("matmul_2_transpose_x_0"), val = bool(false)]; bool matmul_2_transpose_y_0 = const()[name = string("matmul_2_transpose_y_0"), val = bool(false)]; tensor matmul_2_cast_fp16 = matmul(transpose_x = matmul_2_transpose_x_0, transpose_y = matmul_2_transpose_y_0, x = reshape_6_cast_fp16, y = reshape_7_cast_fp16)[name = string("matmul_2_cast_fp16")]; tensor concat_53 = const()[name = string("concat_53"), val = tensor([1, 32, 64, 64])]; tensor reshape_8_cast_fp16 = reshape(shape = concat_53, x = matmul_2_cast_fp16)[name = string("reshape_8_cast_fp16")]; tensor var_1070_perm_0 = const()[name = string("op_1070_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1072 = const()[name = string("op_1072"), val = tensor([1, 64, 2048])]; tensor var_1070_cast_fp16 = transpose(perm = var_1070_perm_0, x = reshape_8_cast_fp16)[name = string("transpose_91")]; tensor input_33_cast_fp16 = reshape(shape = var_1072, x = var_1070_cast_fp16)[name = string("input_33_cast_fp16")]; tensor model_model_layers_2_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460706624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(462803840))))[name = string("model_model_layers_2_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_2_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_2_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_33_cast_fp16)[name = string("linear_2_cast_fp16")]; tensor hidden_states_21_cast_fp16 = add(x = hidden_states_17_cast_fp16, y = linear_2_cast_fp16)[name = string("hidden_states_21_cast_fp16")]; fp16 const_51_promoted_to_fp16 = const()[name = string("const_51_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1078_cast_fp16 = mul(x = hidden_states_21_cast_fp16, y = const_51_promoted_to_fp16)[name = string("op_1078_cast_fp16")]; bool input_35_interleave_0 = const()[name = string("input_35_interleave_0"), val = bool(false)]; tensor input_35_cast_fp16 = concat(axis = var_73, interleave = input_35_interleave_0, values = (hidden_states_21_cast_fp16, var_1078_cast_fp16))[name = string("input_35_cast_fp16")]; tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_76_to_fp16, x = input_35_cast_fp16)[name = string("normed_21_cast_fp16")]; tensor normed_23_begin_0 = const()[name = string("normed_23_begin_0"), val = tensor([0, 0, 0])]; tensor normed_23_end_0 = const()[name = string("normed_23_end_0"), val = tensor([1, 64, 2048])]; tensor normed_23_end_mask_0 = const()[name = string("normed_23_end_mask_0"), val = tensor([true, true, false])]; tensor normed_23_cast_fp16 = slice_by_index(begin = normed_23_begin_0, end = normed_23_end_0, end_mask = normed_23_end_mask_0, x = normed_21_cast_fp16)[name = string("normed_23_cast_fp16")]; tensor const_54_promoted_to_fp16 = const()[name = string("const_54_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(462812096)))]; tensor x_81_cast_fp16 = mul(x = normed_23_cast_fp16, y = const_54_promoted_to_fp16)[name = string("x_81_cast_fp16")]; tensor var_1096 = const()[name = string("op_1096"), val = tensor([0, 2, 1])]; tensor input_37_axes_0 = const()[name = string("input_37_axes_0"), val = tensor([2])]; tensor var_1097 = transpose(perm = var_1096, x = x_81_cast_fp16)[name = string("transpose_90")]; tensor input_37 = expand_dims(axes = input_37_axes_0, x = var_1097)[name = string("input_37")]; string input_39_pad_type_0 = const()[name = string("input_39_pad_type_0"), val = string("valid")]; tensor input_39_strides_0 = const()[name = string("input_39_strides_0"), val = tensor([1, 1])]; tensor input_39_pad_0 = const()[name = string("input_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_39_dilations_0 = const()[name = string("input_39_dilations_0"), val = tensor([1, 1])]; int32 input_39_groups_0 = const()[name = string("input_39_groups_0"), val = int32(1)]; tensor input_39 = conv(dilations = input_39_dilations_0, groups = input_39_groups_0, pad = input_39_pad_0, pad_type = input_39_pad_type_0, strides = input_39_strides_0, weight = model_model_layers_2_mlp_gate_proj_weight_palettized, x = input_37)[name = string("input_39")]; string up_states_5_pad_type_0 = const()[name = string("up_states_5_pad_type_0"), val = string("valid")]; tensor up_states_5_strides_0 = const()[name = string("up_states_5_strides_0"), val = tensor([1, 1])]; tensor up_states_5_pad_0 = const()[name = string("up_states_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_5_dilations_0 = const()[name = string("up_states_5_dilations_0"), val = tensor([1, 1])]; int32 up_states_5_groups_0 = const()[name = string("up_states_5_groups_0"), val = int32(1)]; tensor up_states_5 = conv(dilations = up_states_5_dilations_0, groups = up_states_5_groups_0, pad = up_states_5_pad_0, pad_type = up_states_5_pad_type_0, strides = up_states_5_strides_0, weight = model_model_layers_2_mlp_up_proj_weight_palettized, x = input_37)[name = string("up_states_5")]; tensor gate_states_5 = silu(x = input_39)[name = string("gate_states_5")]; tensor input_41 = mul(x = gate_states_5, y = up_states_5)[name = string("input_41")]; string hidden_states_23_pad_type_0 = const()[name = string("hidden_states_23_pad_type_0"), val = string("valid")]; tensor hidden_states_23_strides_0 = const()[name = string("hidden_states_23_strides_0"), val = tensor([1, 1])]; tensor hidden_states_23_pad_0 = const()[name = string("hidden_states_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_23_dilations_0 = const()[name = string("hidden_states_23_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_23_groups_0 = const()[name = string("hidden_states_23_groups_0"), val = int32(1)]; tensor hidden_states_23 = conv(dilations = hidden_states_23_dilations_0, groups = hidden_states_23_groups_0, pad = hidden_states_23_pad_0, pad_type = hidden_states_23_pad_type_0, strides = hidden_states_23_strides_0, weight = model_model_layers_2_mlp_down_proj_weight_palettized, x = input_41)[name = string("hidden_states_23")]; tensor var_1119_axes_0 = const()[name = string("op_1119_axes_0"), val = tensor([2])]; tensor var_1119 = squeeze(axes = var_1119_axes_0, x = hidden_states_23)[name = string("op_1119")]; tensor var_1120 = const()[name = string("op_1120"), val = tensor([0, 2, 1])]; tensor var_1121 = transpose(perm = var_1120, x = var_1119)[name = string("transpose_89")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_21_cast_fp16, y = var_1121)[name = string("hidden_states_25_cast_fp16")]; fp16 const_55_promoted_to_fp16 = const()[name = string("const_55_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1124_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_55_promoted_to_fp16)[name = string("op_1124_cast_fp16")]; bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; tensor input_43_cast_fp16 = concat(axis = var_73, interleave = input_43_interleave_0, values = (hidden_states_25_cast_fp16, var_1124_cast_fp16))[name = string("input_43_cast_fp16")]; tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_76_to_fp16, x = input_43_cast_fp16)[name = string("normed_25_cast_fp16")]; tensor normed_27_begin_0 = const()[name = string("normed_27_begin_0"), val = tensor([0, 0, 0])]; tensor normed_27_end_0 = const()[name = string("normed_27_end_0"), val = tensor([1, 64, 2048])]; tensor normed_27_end_mask_0 = const()[name = string("normed_27_end_mask_0"), val = tensor([true, true, false])]; tensor normed_27_cast_fp16 = slice_by_index(begin = normed_27_begin_0, end = normed_27_end_0, end_mask = normed_27_end_mask_0, x = normed_25_cast_fp16)[name = string("normed_27_cast_fp16")]; tensor const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(462816256)))]; tensor hidden_states_27_cast_fp16 = mul(x = normed_27_cast_fp16, y = const_58_promoted_to_fp16)[name = string("hidden_states_27_cast_fp16")]; tensor var_1139 = const()[name = string("op_1139"), val = tensor([0, 2, 1])]; tensor var_1141_axes_0 = const()[name = string("op_1141_axes_0"), val = tensor([2])]; tensor var_1140_cast_fp16 = transpose(perm = var_1139, x = hidden_states_27_cast_fp16)[name = string("transpose_88")]; tensor var_1141_cast_fp16 = expand_dims(axes = var_1141_axes_0, x = var_1140_cast_fp16)[name = string("op_1141_cast_fp16")]; string query_states_13_pad_type_0 = const()[name = string("query_states_13_pad_type_0"), val = string("valid")]; tensor query_states_13_strides_0 = const()[name = string("query_states_13_strides_0"), val = tensor([1, 1])]; tensor query_states_13_pad_0 = const()[name = string("query_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_13_dilations_0 = const()[name = string("query_states_13_dilations_0"), val = tensor([1, 1])]; int32 query_states_13_groups_0 = const()[name = string("query_states_13_groups_0"), val = int32(1)]; tensor query_states_13 = conv(dilations = query_states_13_dilations_0, groups = query_states_13_groups_0, pad = query_states_13_pad_0, pad_type = query_states_13_pad_type_0, strides = query_states_13_strides_0, weight = model_model_layers_3_self_attn_q_proj_weight_palettized, x = var_1141_cast_fp16)[name = string("query_states_13")]; string key_states_19_pad_type_0 = const()[name = string("key_states_19_pad_type_0"), val = string("valid")]; tensor key_states_19_strides_0 = const()[name = string("key_states_19_strides_0"), val = tensor([1, 1])]; tensor key_states_19_pad_0 = const()[name = string("key_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_19_dilations_0 = const()[name = string("key_states_19_dilations_0"), val = tensor([1, 1])]; int32 key_states_19_groups_0 = const()[name = string("key_states_19_groups_0"), val = int32(1)]; tensor key_states_19 = conv(dilations = key_states_19_dilations_0, groups = key_states_19_groups_0, pad = key_states_19_pad_0, pad_type = key_states_19_pad_type_0, strides = key_states_19_strides_0, weight = model_model_layers_3_self_attn_k_proj_weight_palettized, x = var_1141_cast_fp16)[name = string("key_states_19")]; string value_states_19_pad_type_0 = const()[name = string("value_states_19_pad_type_0"), val = string("valid")]; tensor value_states_19_strides_0 = const()[name = string("value_states_19_strides_0"), val = tensor([1, 1])]; tensor value_states_19_pad_0 = const()[name = string("value_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_19_dilations_0 = const()[name = string("value_states_19_dilations_0"), val = tensor([1, 1])]; int32 value_states_19_groups_0 = const()[name = string("value_states_19_groups_0"), val = int32(1)]; tensor value_states_19 = conv(dilations = value_states_19_dilations_0, groups = value_states_19_groups_0, pad = value_states_19_pad_0, pad_type = value_states_19_pad_type_0, strides = value_states_19_strides_0, weight = model_model_layers_3_self_attn_v_proj_weight_palettized, x = var_1141_cast_fp16)[name = string("value_states_19")]; tensor var_1161 = const()[name = string("op_1161"), val = tensor([1, 32, 64, 64])]; tensor var_1162 = reshape(shape = var_1161, x = query_states_13)[name = string("op_1162")]; tensor var_1163 = const()[name = string("op_1163"), val = tensor([0, 1, 3, 2])]; tensor var_1165 = const()[name = string("op_1165"), val = tensor([1, 8, 64, 64])]; tensor var_1166 = reshape(shape = var_1165, x = key_states_19)[name = string("op_1166")]; tensor var_1167 = const()[name = string("op_1167"), val = tensor([0, 1, 3, 2])]; tensor var_1169 = const()[name = string("op_1169"), val = tensor([1, 8, 64, 64])]; tensor var_1170 = reshape(shape = var_1169, x = value_states_19)[name = string("op_1170")]; tensor var_1171 = const()[name = string("op_1171"), val = tensor([0, 1, 3, 2])]; tensor x1_13_begin_0 = const()[name = string("x1_13_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_13_end_0 = const()[name = string("x1_13_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_13_end_mask_0 = const()[name = string("x1_13_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_85 = transpose(perm = var_1163, x = var_1162)[name = string("transpose_87")]; tensor x1_13 = slice_by_index(begin = x1_13_begin_0, end = x1_13_end_0, end_mask = x1_13_end_mask_0, x = x_85)[name = string("x1_13")]; tensor x2_13_begin_0 = const()[name = string("x2_13_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_13_end_0 = const()[name = string("x2_13_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_13_end_mask_0 = const()[name = string("x2_13_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_13 = slice_by_index(begin = x2_13_begin_0, end = x2_13_end_0, end_mask = x2_13_end_mask_0, x = x_85)[name = string("x2_13")]; tensor var_1189 = mul(x = x1_13, y = cos_7)[name = string("op_1189")]; tensor var_1190 = mul(x = x2_13, y = sin_7)[name = string("op_1190")]; tensor var_1191 = sub(x = var_1189, y = var_1190)[name = string("op_1191")]; tensor var_1192 = mul(x = x2_13, y = cos_7)[name = string("op_1192")]; tensor var_1193 = mul(x = x1_13, y = sin_7)[name = string("op_1193")]; tensor var_1194 = add(x = var_1192, y = var_1193)[name = string("op_1194")]; bool rotated_13_interleave_0 = const()[name = string("rotated_13_interleave_0"), val = bool(false)]; tensor rotated_13 = concat(axis = var_73, interleave = rotated_13_interleave_0, values = (var_1191, var_1194))[name = string("rotated_13")]; tensor x1_15_begin_0 = const()[name = string("x1_15_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_15_end_0 = const()[name = string("x1_15_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_15_end_mask_0 = const()[name = string("x1_15_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_89 = transpose(perm = var_1167, x = var_1166)[name = string("transpose_86")]; tensor x1_15 = slice_by_index(begin = x1_15_begin_0, end = x1_15_end_0, end_mask = x1_15_end_mask_0, x = x_89)[name = string("x1_15")]; tensor x2_15_begin_0 = const()[name = string("x2_15_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_15_end_0 = const()[name = string("x2_15_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_15_end_mask_0 = const()[name = string("x2_15_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_15 = slice_by_index(begin = x2_15_begin_0, end = x2_15_end_0, end_mask = x2_15_end_mask_0, x = x_89)[name = string("x2_15")]; tensor var_1210 = mul(x = x1_15, y = cos_7)[name = string("op_1210")]; tensor var_1211 = mul(x = x2_15, y = sin_7)[name = string("op_1211")]; tensor var_1212 = sub(x = var_1210, y = var_1211)[name = string("op_1212")]; tensor var_1213 = mul(x = x2_15, y = cos_7)[name = string("op_1213")]; tensor var_1214 = mul(x = x1_15, y = sin_7)[name = string("op_1214")]; tensor var_1215 = add(x = var_1213, y = var_1214)[name = string("op_1215")]; bool rotated_15_interleave_0 = const()[name = string("rotated_15_interleave_0"), val = bool(false)]; tensor rotated_15 = concat(axis = var_73, interleave = rotated_15_interleave_0, values = (var_1212, var_1215))[name = string("rotated_15")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; int32 concat_56_axis_0 = const()[name = string("concat_56_axis_0"), val = int32(0)]; bool concat_56_interleave_0 = const()[name = string("concat_56_interleave_0"), val = bool(false)]; tensor concat_56 = concat(axis = concat_56_axis_0, interleave = concat_56_interleave_0, values = (expand_dims_36, expand_dims_37, current_pos, expand_dims_39))[name = string("concat_56")]; tensor concat_57_values1_0 = const()[name = string("concat_57_values1_0"), val = tensor([0])]; tensor concat_57_values3_0 = const()[name = string("concat_57_values3_0"), val = tensor([0])]; int32 concat_57_axis_0 = const()[name = string("concat_57_axis_0"), val = int32(0)]; bool concat_57_interleave_0 = const()[name = string("concat_57_interleave_0"), val = bool(false)]; tensor concat_57 = concat(axis = concat_57_axis_0, interleave = concat_57_interleave_0, values = (expand_dims_40, concat_57_values1_0, var_597, concat_57_values3_0))[name = string("concat_57")]; tensor model_model_kv_cache_0_internal_tensor_assign_7_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_56, begin_mask = model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0, end = concat_57, end_mask = model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_7_stride_0, update = rotated_15, x = coreml_update_state_37)[name = string("model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_102_write_state")]; tensor coreml_update_state_38 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_102")]; tensor expand_dims_42 = const()[name = string("expand_dims_42"), val = tensor([19])]; tensor expand_dims_43 = const()[name = string("expand_dims_43"), val = tensor([0])]; tensor expand_dims_45 = const()[name = string("expand_dims_45"), val = tensor([0])]; tensor expand_dims_46 = const()[name = string("expand_dims_46"), val = tensor([20])]; int32 concat_60_axis_0 = const()[name = string("concat_60_axis_0"), val = int32(0)]; bool concat_60_interleave_0 = const()[name = string("concat_60_interleave_0"), val = bool(false)]; tensor concat_60 = concat(axis = concat_60_axis_0, interleave = concat_60_interleave_0, values = (expand_dims_42, expand_dims_43, current_pos, expand_dims_45))[name = string("concat_60")]; tensor concat_61_values1_0 = const()[name = string("concat_61_values1_0"), val = tensor([0])]; tensor concat_61_values3_0 = const()[name = string("concat_61_values3_0"), val = tensor([0])]; int32 concat_61_axis_0 = const()[name = string("concat_61_axis_0"), val = int32(0)]; bool concat_61_interleave_0 = const()[name = string("concat_61_interleave_0"), val = bool(false)]; tensor concat_61 = concat(axis = concat_61_axis_0, interleave = concat_61_interleave_0, values = (expand_dims_46, concat_61_values1_0, var_597, concat_61_values3_0))[name = string("concat_61")]; tensor model_model_kv_cache_0_internal_tensor_assign_8_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_21 = transpose(perm = var_1171, x = var_1170)[name = string("transpose_85")]; tensor model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_60, begin_mask = model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0, end = concat_61, end_mask = model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_8_stride_0, update = value_states_21, x = coreml_update_state_38)[name = string("model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_103_write_state")]; tensor coreml_update_state_39 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_103")]; tensor var_1238_begin_0 = const()[name = string("op_1238_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_1238_end_0 = const()[name = string("op_1238_end_0"), val = tensor([4, 8, 4096, 64])]; tensor var_1238_end_mask_0 = const()[name = string("op_1238_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1238_cast_fp16 = slice_by_index(begin = var_1238_begin_0, end = var_1238_end_0, end_mask = var_1238_end_mask_0, x = coreml_update_state_39)[name = string("op_1238_cast_fp16")]; tensor K_layer_cache_7_axes_0 = const()[name = string("K_layer_cache_7_axes_0"), val = tensor([0])]; tensor K_layer_cache_7_cast_fp16 = squeeze(axes = K_layer_cache_7_axes_0, x = var_1238_cast_fp16)[name = string("K_layer_cache_7_cast_fp16")]; tensor var_1240_begin_0 = const()[name = string("op_1240_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_1240_end_0 = const()[name = string("op_1240_end_0"), val = tensor([20, 8, 4096, 64])]; tensor var_1240_end_mask_0 = const()[name = string("op_1240_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1240_cast_fp16 = slice_by_index(begin = var_1240_begin_0, end = var_1240_end_0, end_mask = var_1240_end_mask_0, x = coreml_update_state_39)[name = string("op_1240_cast_fp16")]; tensor V_layer_cache_7_axes_0 = const()[name = string("V_layer_cache_7_axes_0"), val = tensor([0])]; tensor V_layer_cache_7_cast_fp16 = squeeze(axes = V_layer_cache_7_axes_0, x = var_1240_cast_fp16)[name = string("V_layer_cache_7_cast_fp16")]; tensor x_95_axes_0 = const()[name = string("x_95_axes_0"), val = tensor([1])]; tensor x_95_cast_fp16 = expand_dims(axes = x_95_axes_0, x = K_layer_cache_7_cast_fp16)[name = string("x_95_cast_fp16")]; tensor var_1249 = const()[name = string("op_1249"), val = tensor([1, 4, 1, 1])]; tensor x_97_cast_fp16 = tile(reps = var_1249, x = x_95_cast_fp16)[name = string("x_97_cast_fp16")]; tensor var_1253 = const()[name = string("op_1253"), val = tensor([1, -1, 4096, 64])]; tensor var_1254_cast_fp16 = reshape(shape = var_1253, x = x_97_cast_fp16)[name = string("op_1254_cast_fp16")]; tensor x_101_axes_0 = const()[name = string("x_101_axes_0"), val = tensor([1])]; tensor x_101_cast_fp16 = expand_dims(axes = x_101_axes_0, x = V_layer_cache_7_cast_fp16)[name = string("x_101_cast_fp16")]; tensor var_1256 = const()[name = string("op_1256"), val = tensor([1, 4, 1, 1])]; tensor x_103_cast_fp16 = tile(reps = var_1256, x = x_101_cast_fp16)[name = string("x_103_cast_fp16")]; bool var_1263_transpose_x_0 = const()[name = string("op_1263_transpose_x_0"), val = bool(false)]; bool var_1263_transpose_y_0 = const()[name = string("op_1263_transpose_y_0"), val = bool(true)]; tensor var_1263_cast_fp16 = matmul(transpose_x = var_1263_transpose_x_0, transpose_y = var_1263_transpose_y_0, x = rotated_13, y = var_1254_cast_fp16)[name = string("op_1263_cast_fp16")]; fp16 var_1264_to_fp16 = const()[name = string("op_1264_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_7_cast_fp16 = mul(x = var_1263_cast_fp16, y = var_1264_to_fp16)[name = string("attn_weights_7_cast_fp16")]; tensor x_105_cast_fp16 = add(x = attn_weights_7_cast_fp16, y = causal_mask)[name = string("x_105_cast_fp16")]; tensor reduce_max_3_axes_0 = const()[name = string("reduce_max_3_axes_0"), val = tensor([-1])]; bool reduce_max_3_keep_dims_0 = const()[name = string("reduce_max_3_keep_dims_0"), val = bool(true)]; tensor reduce_max_3_cast_fp16 = reduce_max(axes = reduce_max_3_axes_0, keep_dims = reduce_max_3_keep_dims_0, x = x_105_cast_fp16)[name = string("reduce_max_3_cast_fp16")]; tensor x_107_cast_fp16 = sub(x = x_105_cast_fp16, y = reduce_max_3_cast_fp16)[name = string("x_107_cast_fp16")]; tensor exp_x_7_cast_fp16 = exp(x = x_107_cast_fp16)[name = string("exp_x_7_cast_fp16")]; tensor var_1275_axes_0 = const()[name = string("op_1275_axes_0"), val = tensor([-1])]; bool var_1275_keep_dims_0 = const()[name = string("op_1275_keep_dims_0"), val = bool(true)]; tensor var_1275_cast_fp16 = reduce_sum(axes = var_1275_axes_0, keep_dims = var_1275_keep_dims_0, x = exp_x_7_cast_fp16)[name = string("op_1275_cast_fp16")]; tensor var_1276_cast_fp16 = real_div(x = exp_x_7_cast_fp16, y = var_1275_cast_fp16)[name = string("op_1276_cast_fp16")]; tensor concat_66 = const()[name = string("concat_66"), val = tensor([32, 64, 4096])]; tensor reshape_9_cast_fp16 = reshape(shape = concat_66, x = var_1276_cast_fp16)[name = string("reshape_9_cast_fp16")]; tensor concat_67 = const()[name = string("concat_67"), val = tensor([32, 4096, 64])]; tensor reshape_10_cast_fp16 = reshape(shape = concat_67, x = x_103_cast_fp16)[name = string("reshape_10_cast_fp16")]; bool matmul_3_transpose_x_0 = const()[name = string("matmul_3_transpose_x_0"), val = bool(false)]; bool matmul_3_transpose_y_0 = const()[name = string("matmul_3_transpose_y_0"), val = bool(false)]; tensor matmul_3_cast_fp16 = matmul(transpose_x = matmul_3_transpose_x_0, transpose_y = matmul_3_transpose_y_0, x = reshape_9_cast_fp16, y = reshape_10_cast_fp16)[name = string("matmul_3_cast_fp16")]; tensor concat_71 = const()[name = string("concat_71"), val = tensor([1, 32, 64, 64])]; tensor reshape_11_cast_fp16 = reshape(shape = concat_71, x = matmul_3_cast_fp16)[name = string("reshape_11_cast_fp16")]; tensor var_1279_perm_0 = const()[name = string("op_1279_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1281 = const()[name = string("op_1281"), val = tensor([1, 64, 2048])]; tensor var_1279_cast_fp16 = transpose(perm = var_1279_perm_0, x = reshape_11_cast_fp16)[name = string("transpose_84")]; tensor input_47_cast_fp16 = reshape(shape = var_1281, x = var_1279_cast_fp16)[name = string("input_47_cast_fp16")]; tensor model_model_layers_3_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(462820416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(464917632))))[name = string("model_model_layers_3_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_3_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_3_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_47_cast_fp16)[name = string("linear_3_cast_fp16")]; tensor hidden_states_29_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = linear_3_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; fp16 const_69_promoted_to_fp16 = const()[name = string("const_69_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1287_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_69_promoted_to_fp16)[name = string("op_1287_cast_fp16")]; bool input_49_interleave_0 = const()[name = string("input_49_interleave_0"), val = bool(false)]; tensor input_49_cast_fp16 = concat(axis = var_73, interleave = input_49_interleave_0, values = (hidden_states_29_cast_fp16, var_1287_cast_fp16))[name = string("input_49_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_76_to_fp16, x = input_49_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor normed_31_begin_0 = const()[name = string("normed_31_begin_0"), val = tensor([0, 0, 0])]; tensor normed_31_end_0 = const()[name = string("normed_31_end_0"), val = tensor([1, 64, 2048])]; tensor normed_31_end_mask_0 = const()[name = string("normed_31_end_mask_0"), val = tensor([true, true, false])]; tensor normed_31_cast_fp16 = slice_by_index(begin = normed_31_begin_0, end = normed_31_end_0, end_mask = normed_31_end_mask_0, x = normed_29_cast_fp16)[name = string("normed_31_cast_fp16")]; tensor const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(464925888)))]; tensor x_109_cast_fp16 = mul(x = normed_31_cast_fp16, y = const_72_promoted_to_fp16)[name = string("x_109_cast_fp16")]; tensor var_1305 = const()[name = string("op_1305"), val = tensor([0, 2, 1])]; tensor input_51_axes_0 = const()[name = string("input_51_axes_0"), val = tensor([2])]; tensor var_1306 = transpose(perm = var_1305, x = x_109_cast_fp16)[name = string("transpose_83")]; tensor input_51 = expand_dims(axes = input_51_axes_0, x = var_1306)[name = string("input_51")]; string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; tensor input_53 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = model_model_layers_3_mlp_gate_proj_weight_palettized, x = input_51)[name = string("input_53")]; string up_states_7_pad_type_0 = const()[name = string("up_states_7_pad_type_0"), val = string("valid")]; tensor up_states_7_strides_0 = const()[name = string("up_states_7_strides_0"), val = tensor([1, 1])]; tensor up_states_7_pad_0 = const()[name = string("up_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_7_dilations_0 = const()[name = string("up_states_7_dilations_0"), val = tensor([1, 1])]; int32 up_states_7_groups_0 = const()[name = string("up_states_7_groups_0"), val = int32(1)]; tensor up_states_7 = conv(dilations = up_states_7_dilations_0, groups = up_states_7_groups_0, pad = up_states_7_pad_0, pad_type = up_states_7_pad_type_0, strides = up_states_7_strides_0, weight = model_model_layers_3_mlp_up_proj_weight_palettized, x = input_51)[name = string("up_states_7")]; tensor gate_states_7 = silu(x = input_53)[name = string("gate_states_7")]; tensor input_55 = mul(x = gate_states_7, y = up_states_7)[name = string("input_55")]; string hidden_states_31_pad_type_0 = const()[name = string("hidden_states_31_pad_type_0"), val = string("valid")]; tensor hidden_states_31_strides_0 = const()[name = string("hidden_states_31_strides_0"), val = tensor([1, 1])]; tensor hidden_states_31_pad_0 = const()[name = string("hidden_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_31_dilations_0 = const()[name = string("hidden_states_31_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_31_groups_0 = const()[name = string("hidden_states_31_groups_0"), val = int32(1)]; tensor hidden_states_31 = conv(dilations = hidden_states_31_dilations_0, groups = hidden_states_31_groups_0, pad = hidden_states_31_pad_0, pad_type = hidden_states_31_pad_type_0, strides = hidden_states_31_strides_0, weight = model_model_layers_3_mlp_down_proj_weight_palettized, x = input_55)[name = string("hidden_states_31")]; tensor var_1328_axes_0 = const()[name = string("op_1328_axes_0"), val = tensor([2])]; tensor var_1328 = squeeze(axes = var_1328_axes_0, x = hidden_states_31)[name = string("op_1328")]; tensor var_1329 = const()[name = string("op_1329"), val = tensor([0, 2, 1])]; tensor var_1330 = transpose(perm = var_1329, x = var_1328)[name = string("transpose_82")]; tensor hidden_states_33_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = var_1330)[name = string("hidden_states_33_cast_fp16")]; fp16 const_73_promoted_to_fp16 = const()[name = string("const_73_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1333_cast_fp16 = mul(x = hidden_states_33_cast_fp16, y = const_73_promoted_to_fp16)[name = string("op_1333_cast_fp16")]; bool input_57_interleave_0 = const()[name = string("input_57_interleave_0"), val = bool(false)]; tensor input_57_cast_fp16 = concat(axis = var_73, interleave = input_57_interleave_0, values = (hidden_states_33_cast_fp16, var_1333_cast_fp16))[name = string("input_57_cast_fp16")]; tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_76_to_fp16, x = input_57_cast_fp16)[name = string("normed_33_cast_fp16")]; tensor normed_35_begin_0 = const()[name = string("normed_35_begin_0"), val = tensor([0, 0, 0])]; tensor normed_35_end_0 = const()[name = string("normed_35_end_0"), val = tensor([1, 64, 2048])]; tensor normed_35_end_mask_0 = const()[name = string("normed_35_end_mask_0"), val = tensor([true, true, false])]; tensor normed_35_cast_fp16 = slice_by_index(begin = normed_35_begin_0, end = normed_35_end_0, end_mask = normed_35_end_mask_0, x = normed_33_cast_fp16)[name = string("normed_35_cast_fp16")]; tensor const_76_promoted_to_fp16 = const()[name = string("const_76_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(464930048)))]; tensor hidden_states_35_cast_fp16 = mul(x = normed_35_cast_fp16, y = const_76_promoted_to_fp16)[name = string("hidden_states_35_cast_fp16")]; tensor var_1348 = const()[name = string("op_1348"), val = tensor([0, 2, 1])]; tensor var_1350_axes_0 = const()[name = string("op_1350_axes_0"), val = tensor([2])]; tensor var_1349_cast_fp16 = transpose(perm = var_1348, x = hidden_states_35_cast_fp16)[name = string("transpose_81")]; tensor var_1350_cast_fp16 = expand_dims(axes = var_1350_axes_0, x = var_1349_cast_fp16)[name = string("op_1350_cast_fp16")]; string query_states_17_pad_type_0 = const()[name = string("query_states_17_pad_type_0"), val = string("valid")]; tensor query_states_17_strides_0 = const()[name = string("query_states_17_strides_0"), val = tensor([1, 1])]; tensor query_states_17_pad_0 = const()[name = string("query_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_17_dilations_0 = const()[name = string("query_states_17_dilations_0"), val = tensor([1, 1])]; int32 query_states_17_groups_0 = const()[name = string("query_states_17_groups_0"), val = int32(1)]; tensor query_states_17 = conv(dilations = query_states_17_dilations_0, groups = query_states_17_groups_0, pad = query_states_17_pad_0, pad_type = query_states_17_pad_type_0, strides = query_states_17_strides_0, weight = model_model_layers_4_self_attn_q_proj_weight_palettized, x = var_1350_cast_fp16)[name = string("query_states_17")]; string key_states_25_pad_type_0 = const()[name = string("key_states_25_pad_type_0"), val = string("valid")]; tensor key_states_25_strides_0 = const()[name = string("key_states_25_strides_0"), val = tensor([1, 1])]; tensor key_states_25_pad_0 = const()[name = string("key_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_25_dilations_0 = const()[name = string("key_states_25_dilations_0"), val = tensor([1, 1])]; int32 key_states_25_groups_0 = const()[name = string("key_states_25_groups_0"), val = int32(1)]; tensor key_states_25 = conv(dilations = key_states_25_dilations_0, groups = key_states_25_groups_0, pad = key_states_25_pad_0, pad_type = key_states_25_pad_type_0, strides = key_states_25_strides_0, weight = model_model_layers_4_self_attn_k_proj_weight_palettized, x = var_1350_cast_fp16)[name = string("key_states_25")]; string value_states_25_pad_type_0 = const()[name = string("value_states_25_pad_type_0"), val = string("valid")]; tensor value_states_25_strides_0 = const()[name = string("value_states_25_strides_0"), val = tensor([1, 1])]; tensor value_states_25_pad_0 = const()[name = string("value_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_25_dilations_0 = const()[name = string("value_states_25_dilations_0"), val = tensor([1, 1])]; int32 value_states_25_groups_0 = const()[name = string("value_states_25_groups_0"), val = int32(1)]; tensor value_states_25 = conv(dilations = value_states_25_dilations_0, groups = value_states_25_groups_0, pad = value_states_25_pad_0, pad_type = value_states_25_pad_type_0, strides = value_states_25_strides_0, weight = model_model_layers_4_self_attn_v_proj_weight_palettized, x = var_1350_cast_fp16)[name = string("value_states_25")]; tensor var_1370 = const()[name = string("op_1370"), val = tensor([1, 32, 64, 64])]; tensor var_1371 = reshape(shape = var_1370, x = query_states_17)[name = string("op_1371")]; tensor var_1372 = const()[name = string("op_1372"), val = tensor([0, 1, 3, 2])]; tensor var_1374 = const()[name = string("op_1374"), val = tensor([1, 8, 64, 64])]; tensor var_1375 = reshape(shape = var_1374, x = key_states_25)[name = string("op_1375")]; tensor var_1376 = const()[name = string("op_1376"), val = tensor([0, 1, 3, 2])]; tensor var_1378 = const()[name = string("op_1378"), val = tensor([1, 8, 64, 64])]; tensor var_1379 = reshape(shape = var_1378, x = value_states_25)[name = string("op_1379")]; tensor var_1380 = const()[name = string("op_1380"), val = tensor([0, 1, 3, 2])]; tensor x1_17_begin_0 = const()[name = string("x1_17_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_17_end_0 = const()[name = string("x1_17_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_17_end_mask_0 = const()[name = string("x1_17_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_113 = transpose(perm = var_1372, x = var_1371)[name = string("transpose_80")]; tensor x1_17 = slice_by_index(begin = x1_17_begin_0, end = x1_17_end_0, end_mask = x1_17_end_mask_0, x = x_113)[name = string("x1_17")]; tensor x2_17_begin_0 = const()[name = string("x2_17_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_17_end_0 = const()[name = string("x2_17_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_17_end_mask_0 = const()[name = string("x2_17_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_17 = slice_by_index(begin = x2_17_begin_0, end = x2_17_end_0, end_mask = x2_17_end_mask_0, x = x_113)[name = string("x2_17")]; tensor var_1398 = mul(x = x1_17, y = cos_7)[name = string("op_1398")]; tensor var_1399 = mul(x = x2_17, y = sin_7)[name = string("op_1399")]; tensor var_1400 = sub(x = var_1398, y = var_1399)[name = string("op_1400")]; tensor var_1401 = mul(x = x2_17, y = cos_7)[name = string("op_1401")]; tensor var_1402 = mul(x = x1_17, y = sin_7)[name = string("op_1402")]; tensor var_1403 = add(x = var_1401, y = var_1402)[name = string("op_1403")]; bool rotated_17_interleave_0 = const()[name = string("rotated_17_interleave_0"), val = bool(false)]; tensor rotated_17 = concat(axis = var_73, interleave = rotated_17_interleave_0, values = (var_1400, var_1403))[name = string("rotated_17")]; tensor x1_19_begin_0 = const()[name = string("x1_19_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_19_end_0 = const()[name = string("x1_19_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_19_end_mask_0 = const()[name = string("x1_19_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_117 = transpose(perm = var_1376, x = var_1375)[name = string("transpose_79")]; tensor x1_19 = slice_by_index(begin = x1_19_begin_0, end = x1_19_end_0, end_mask = x1_19_end_mask_0, x = x_117)[name = string("x1_19")]; tensor x2_19_begin_0 = const()[name = string("x2_19_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_19_end_0 = const()[name = string("x2_19_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_19_end_mask_0 = const()[name = string("x2_19_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_19 = slice_by_index(begin = x2_19_begin_0, end = x2_19_end_0, end_mask = x2_19_end_mask_0, x = x_117)[name = string("x2_19")]; tensor var_1419 = mul(x = x1_19, y = cos_7)[name = string("op_1419")]; tensor var_1420 = mul(x = x2_19, y = sin_7)[name = string("op_1420")]; tensor var_1421 = sub(x = var_1419, y = var_1420)[name = string("op_1421")]; tensor var_1422 = mul(x = x2_19, y = cos_7)[name = string("op_1422")]; tensor var_1423 = mul(x = x1_19, y = sin_7)[name = string("op_1423")]; tensor var_1424 = add(x = var_1422, y = var_1423)[name = string("op_1424")]; bool rotated_19_interleave_0 = const()[name = string("rotated_19_interleave_0"), val = bool(false)]; tensor rotated_19 = concat(axis = var_73, interleave = rotated_19_interleave_0, values = (var_1421, var_1424))[name = string("rotated_19")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; int32 concat_74_axis_0 = const()[name = string("concat_74_axis_0"), val = int32(0)]; bool concat_74_interleave_0 = const()[name = string("concat_74_interleave_0"), val = bool(false)]; tensor concat_74 = concat(axis = concat_74_axis_0, interleave = concat_74_interleave_0, values = (expand_dims_48, expand_dims_49, current_pos, expand_dims_51))[name = string("concat_74")]; tensor concat_75_values1_0 = const()[name = string("concat_75_values1_0"), val = tensor([0])]; tensor concat_75_values3_0 = const()[name = string("concat_75_values3_0"), val = tensor([0])]; int32 concat_75_axis_0 = const()[name = string("concat_75_axis_0"), val = int32(0)]; bool concat_75_interleave_0 = const()[name = string("concat_75_interleave_0"), val = bool(false)]; tensor concat_75 = concat(axis = concat_75_axis_0, interleave = concat_75_interleave_0, values = (expand_dims_52, concat_75_values1_0, var_597, concat_75_values3_0))[name = string("concat_75")]; tensor model_model_kv_cache_0_internal_tensor_assign_9_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_74, begin_mask = model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0, end = concat_75, end_mask = model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_9_stride_0, update = rotated_19, x = coreml_update_state_39)[name = string("model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_104_write_state")]; tensor coreml_update_state_40 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_104")]; tensor expand_dims_54 = const()[name = string("expand_dims_54"), val = tensor([20])]; tensor expand_dims_55 = const()[name = string("expand_dims_55"), val = tensor([0])]; tensor expand_dims_57 = const()[name = string("expand_dims_57"), val = tensor([0])]; tensor expand_dims_58 = const()[name = string("expand_dims_58"), val = tensor([21])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_54, expand_dims_55, current_pos, expand_dims_57))[name = string("concat_78")]; tensor concat_79_values1_0 = const()[name = string("concat_79_values1_0"), val = tensor([0])]; tensor concat_79_values3_0 = const()[name = string("concat_79_values3_0"), val = tensor([0])]; int32 concat_79_axis_0 = const()[name = string("concat_79_axis_0"), val = int32(0)]; bool concat_79_interleave_0 = const()[name = string("concat_79_interleave_0"), val = bool(false)]; tensor concat_79 = concat(axis = concat_79_axis_0, interleave = concat_79_interleave_0, values = (expand_dims_58, concat_79_values1_0, var_597, concat_79_values3_0))[name = string("concat_79")]; tensor model_model_kv_cache_0_internal_tensor_assign_10_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_27 = transpose(perm = var_1380, x = var_1379)[name = string("transpose_78")]; tensor model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_78, begin_mask = model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0, end = concat_79, end_mask = model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_10_stride_0, update = value_states_27, x = coreml_update_state_40)[name = string("model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_105_write_state")]; tensor coreml_update_state_41 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_105")]; tensor var_1447_begin_0 = const()[name = string("op_1447_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_1447_end_0 = const()[name = string("op_1447_end_0"), val = tensor([5, 8, 4096, 64])]; tensor var_1447_end_mask_0 = const()[name = string("op_1447_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1447_cast_fp16 = slice_by_index(begin = var_1447_begin_0, end = var_1447_end_0, end_mask = var_1447_end_mask_0, x = coreml_update_state_41)[name = string("op_1447_cast_fp16")]; tensor K_layer_cache_9_axes_0 = const()[name = string("K_layer_cache_9_axes_0"), val = tensor([0])]; tensor K_layer_cache_9_cast_fp16 = squeeze(axes = K_layer_cache_9_axes_0, x = var_1447_cast_fp16)[name = string("K_layer_cache_9_cast_fp16")]; tensor var_1449_begin_0 = const()[name = string("op_1449_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_1449_end_0 = const()[name = string("op_1449_end_0"), val = tensor([21, 8, 4096, 64])]; tensor var_1449_end_mask_0 = const()[name = string("op_1449_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1449_cast_fp16 = slice_by_index(begin = var_1449_begin_0, end = var_1449_end_0, end_mask = var_1449_end_mask_0, x = coreml_update_state_41)[name = string("op_1449_cast_fp16")]; tensor V_layer_cache_9_axes_0 = const()[name = string("V_layer_cache_9_axes_0"), val = tensor([0])]; tensor V_layer_cache_9_cast_fp16 = squeeze(axes = V_layer_cache_9_axes_0, x = var_1449_cast_fp16)[name = string("V_layer_cache_9_cast_fp16")]; tensor x_123_axes_0 = const()[name = string("x_123_axes_0"), val = tensor([1])]; tensor x_123_cast_fp16 = expand_dims(axes = x_123_axes_0, x = K_layer_cache_9_cast_fp16)[name = string("x_123_cast_fp16")]; tensor var_1458 = const()[name = string("op_1458"), val = tensor([1, 4, 1, 1])]; tensor x_125_cast_fp16 = tile(reps = var_1458, x = x_123_cast_fp16)[name = string("x_125_cast_fp16")]; tensor var_1462 = const()[name = string("op_1462"), val = tensor([1, -1, 4096, 64])]; tensor var_1463_cast_fp16 = reshape(shape = var_1462, x = x_125_cast_fp16)[name = string("op_1463_cast_fp16")]; tensor x_129_axes_0 = const()[name = string("x_129_axes_0"), val = tensor([1])]; tensor x_129_cast_fp16 = expand_dims(axes = x_129_axes_0, x = V_layer_cache_9_cast_fp16)[name = string("x_129_cast_fp16")]; tensor var_1465 = const()[name = string("op_1465"), val = tensor([1, 4, 1, 1])]; tensor x_131_cast_fp16 = tile(reps = var_1465, x = x_129_cast_fp16)[name = string("x_131_cast_fp16")]; bool var_1472_transpose_x_0 = const()[name = string("op_1472_transpose_x_0"), val = bool(false)]; bool var_1472_transpose_y_0 = const()[name = string("op_1472_transpose_y_0"), val = bool(true)]; tensor var_1472_cast_fp16 = matmul(transpose_x = var_1472_transpose_x_0, transpose_y = var_1472_transpose_y_0, x = rotated_17, y = var_1463_cast_fp16)[name = string("op_1472_cast_fp16")]; fp16 var_1473_to_fp16 = const()[name = string("op_1473_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_9_cast_fp16 = mul(x = var_1472_cast_fp16, y = var_1473_to_fp16)[name = string("attn_weights_9_cast_fp16")]; tensor x_133_cast_fp16 = add(x = attn_weights_9_cast_fp16, y = causal_mask)[name = string("x_133_cast_fp16")]; tensor reduce_max_4_axes_0 = const()[name = string("reduce_max_4_axes_0"), val = tensor([-1])]; bool reduce_max_4_keep_dims_0 = const()[name = string("reduce_max_4_keep_dims_0"), val = bool(true)]; tensor reduce_max_4_cast_fp16 = reduce_max(axes = reduce_max_4_axes_0, keep_dims = reduce_max_4_keep_dims_0, x = x_133_cast_fp16)[name = string("reduce_max_4_cast_fp16")]; tensor x_135_cast_fp16 = sub(x = x_133_cast_fp16, y = reduce_max_4_cast_fp16)[name = string("x_135_cast_fp16")]; tensor exp_x_9_cast_fp16 = exp(x = x_135_cast_fp16)[name = string("exp_x_9_cast_fp16")]; tensor var_1484_axes_0 = const()[name = string("op_1484_axes_0"), val = tensor([-1])]; bool var_1484_keep_dims_0 = const()[name = string("op_1484_keep_dims_0"), val = bool(true)]; tensor var_1484_cast_fp16 = reduce_sum(axes = var_1484_axes_0, keep_dims = var_1484_keep_dims_0, x = exp_x_9_cast_fp16)[name = string("op_1484_cast_fp16")]; tensor var_1485_cast_fp16 = real_div(x = exp_x_9_cast_fp16, y = var_1484_cast_fp16)[name = string("op_1485_cast_fp16")]; tensor concat_84 = const()[name = string("concat_84"), val = tensor([32, 64, 4096])]; tensor reshape_12_cast_fp16 = reshape(shape = concat_84, x = var_1485_cast_fp16)[name = string("reshape_12_cast_fp16")]; tensor concat_85 = const()[name = string("concat_85"), val = tensor([32, 4096, 64])]; tensor reshape_13_cast_fp16 = reshape(shape = concat_85, x = x_131_cast_fp16)[name = string("reshape_13_cast_fp16")]; bool matmul_4_transpose_x_0 = const()[name = string("matmul_4_transpose_x_0"), val = bool(false)]; bool matmul_4_transpose_y_0 = const()[name = string("matmul_4_transpose_y_0"), val = bool(false)]; tensor matmul_4_cast_fp16 = matmul(transpose_x = matmul_4_transpose_x_0, transpose_y = matmul_4_transpose_y_0, x = reshape_12_cast_fp16, y = reshape_13_cast_fp16)[name = string("matmul_4_cast_fp16")]; tensor concat_89 = const()[name = string("concat_89"), val = tensor([1, 32, 64, 64])]; tensor reshape_14_cast_fp16 = reshape(shape = concat_89, x = matmul_4_cast_fp16)[name = string("reshape_14_cast_fp16")]; tensor var_1488_perm_0 = const()[name = string("op_1488_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1490 = const()[name = string("op_1490"), val = tensor([1, 64, 2048])]; tensor var_1488_cast_fp16 = transpose(perm = var_1488_perm_0, x = reshape_14_cast_fp16)[name = string("transpose_77")]; tensor input_61_cast_fp16 = reshape(shape = var_1490, x = var_1488_cast_fp16)[name = string("input_61_cast_fp16")]; tensor model_model_layers_4_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(464934208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(467031424))))[name = string("model_model_layers_4_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_4_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_4_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_61_cast_fp16)[name = string("linear_4_cast_fp16")]; tensor hidden_states_37_cast_fp16 = add(x = hidden_states_33_cast_fp16, y = linear_4_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; fp16 const_87_promoted_to_fp16 = const()[name = string("const_87_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1496_cast_fp16 = mul(x = hidden_states_37_cast_fp16, y = const_87_promoted_to_fp16)[name = string("op_1496_cast_fp16")]; bool input_63_interleave_0 = const()[name = string("input_63_interleave_0"), val = bool(false)]; tensor input_63_cast_fp16 = concat(axis = var_73, interleave = input_63_interleave_0, values = (hidden_states_37_cast_fp16, var_1496_cast_fp16))[name = string("input_63_cast_fp16")]; tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_76_to_fp16, x = input_63_cast_fp16)[name = string("normed_37_cast_fp16")]; tensor normed_39_begin_0 = const()[name = string("normed_39_begin_0"), val = tensor([0, 0, 0])]; tensor normed_39_end_0 = const()[name = string("normed_39_end_0"), val = tensor([1, 64, 2048])]; tensor normed_39_end_mask_0 = const()[name = string("normed_39_end_mask_0"), val = tensor([true, true, false])]; tensor normed_39_cast_fp16 = slice_by_index(begin = normed_39_begin_0, end = normed_39_end_0, end_mask = normed_39_end_mask_0, x = normed_37_cast_fp16)[name = string("normed_39_cast_fp16")]; tensor const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(467039680)))]; tensor x_137_cast_fp16 = mul(x = normed_39_cast_fp16, y = const_90_promoted_to_fp16)[name = string("x_137_cast_fp16")]; tensor var_1514 = const()[name = string("op_1514"), val = tensor([0, 2, 1])]; tensor input_65_axes_0 = const()[name = string("input_65_axes_0"), val = tensor([2])]; tensor var_1515 = transpose(perm = var_1514, x = x_137_cast_fp16)[name = string("transpose_76")]; tensor input_65 = expand_dims(axes = input_65_axes_0, x = var_1515)[name = string("input_65")]; string input_67_pad_type_0 = const()[name = string("input_67_pad_type_0"), val = string("valid")]; tensor input_67_strides_0 = const()[name = string("input_67_strides_0"), val = tensor([1, 1])]; tensor input_67_pad_0 = const()[name = string("input_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_67_dilations_0 = const()[name = string("input_67_dilations_0"), val = tensor([1, 1])]; int32 input_67_groups_0 = const()[name = string("input_67_groups_0"), val = int32(1)]; tensor input_67 = conv(dilations = input_67_dilations_0, groups = input_67_groups_0, pad = input_67_pad_0, pad_type = input_67_pad_type_0, strides = input_67_strides_0, weight = model_model_layers_4_mlp_gate_proj_weight_palettized, x = input_65)[name = string("input_67")]; string up_states_9_pad_type_0 = const()[name = string("up_states_9_pad_type_0"), val = string("valid")]; tensor up_states_9_strides_0 = const()[name = string("up_states_9_strides_0"), val = tensor([1, 1])]; tensor up_states_9_pad_0 = const()[name = string("up_states_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_9_dilations_0 = const()[name = string("up_states_9_dilations_0"), val = tensor([1, 1])]; int32 up_states_9_groups_0 = const()[name = string("up_states_9_groups_0"), val = int32(1)]; tensor up_states_9 = conv(dilations = up_states_9_dilations_0, groups = up_states_9_groups_0, pad = up_states_9_pad_0, pad_type = up_states_9_pad_type_0, strides = up_states_9_strides_0, weight = model_model_layers_4_mlp_up_proj_weight_palettized, x = input_65)[name = string("up_states_9")]; tensor gate_states_9 = silu(x = input_67)[name = string("gate_states_9")]; tensor input_69 = mul(x = gate_states_9, y = up_states_9)[name = string("input_69")]; string hidden_states_39_pad_type_0 = const()[name = string("hidden_states_39_pad_type_0"), val = string("valid")]; tensor hidden_states_39_strides_0 = const()[name = string("hidden_states_39_strides_0"), val = tensor([1, 1])]; tensor hidden_states_39_pad_0 = const()[name = string("hidden_states_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_39_dilations_0 = const()[name = string("hidden_states_39_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_39_groups_0 = const()[name = string("hidden_states_39_groups_0"), val = int32(1)]; tensor hidden_states_39 = conv(dilations = hidden_states_39_dilations_0, groups = hidden_states_39_groups_0, pad = hidden_states_39_pad_0, pad_type = hidden_states_39_pad_type_0, strides = hidden_states_39_strides_0, weight = model_model_layers_4_mlp_down_proj_weight_palettized, x = input_69)[name = string("hidden_states_39")]; tensor var_1537_axes_0 = const()[name = string("op_1537_axes_0"), val = tensor([2])]; tensor var_1537 = squeeze(axes = var_1537_axes_0, x = hidden_states_39)[name = string("op_1537")]; tensor var_1538 = const()[name = string("op_1538"), val = tensor([0, 2, 1])]; tensor var_1539 = transpose(perm = var_1538, x = var_1537)[name = string("transpose_75")]; tensor hidden_states_41_cast_fp16 = add(x = hidden_states_37_cast_fp16, y = var_1539)[name = string("hidden_states_41_cast_fp16")]; fp16 const_91_promoted_to_fp16 = const()[name = string("const_91_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1542_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_91_promoted_to_fp16)[name = string("op_1542_cast_fp16")]; bool input_71_interleave_0 = const()[name = string("input_71_interleave_0"), val = bool(false)]; tensor input_71_cast_fp16 = concat(axis = var_73, interleave = input_71_interleave_0, values = (hidden_states_41_cast_fp16, var_1542_cast_fp16))[name = string("input_71_cast_fp16")]; tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_76_to_fp16, x = input_71_cast_fp16)[name = string("normed_41_cast_fp16")]; tensor normed_43_begin_0 = const()[name = string("normed_43_begin_0"), val = tensor([0, 0, 0])]; tensor normed_43_end_0 = const()[name = string("normed_43_end_0"), val = tensor([1, 64, 2048])]; tensor normed_43_end_mask_0 = const()[name = string("normed_43_end_mask_0"), val = tensor([true, true, false])]; tensor normed_43_cast_fp16 = slice_by_index(begin = normed_43_begin_0, end = normed_43_end_0, end_mask = normed_43_end_mask_0, x = normed_41_cast_fp16)[name = string("normed_43_cast_fp16")]; tensor const_94_promoted_to_fp16 = const()[name = string("const_94_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(467043840)))]; tensor hidden_states_43_cast_fp16 = mul(x = normed_43_cast_fp16, y = const_94_promoted_to_fp16)[name = string("hidden_states_43_cast_fp16")]; tensor var_1557 = const()[name = string("op_1557"), val = tensor([0, 2, 1])]; tensor var_1559_axes_0 = const()[name = string("op_1559_axes_0"), val = tensor([2])]; tensor var_1558_cast_fp16 = transpose(perm = var_1557, x = hidden_states_43_cast_fp16)[name = string("transpose_74")]; tensor var_1559_cast_fp16 = expand_dims(axes = var_1559_axes_0, x = var_1558_cast_fp16)[name = string("op_1559_cast_fp16")]; string query_states_21_pad_type_0 = const()[name = string("query_states_21_pad_type_0"), val = string("valid")]; tensor query_states_21_strides_0 = const()[name = string("query_states_21_strides_0"), val = tensor([1, 1])]; tensor query_states_21_pad_0 = const()[name = string("query_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_21_dilations_0 = const()[name = string("query_states_21_dilations_0"), val = tensor([1, 1])]; int32 query_states_21_groups_0 = const()[name = string("query_states_21_groups_0"), val = int32(1)]; tensor query_states_21 = conv(dilations = query_states_21_dilations_0, groups = query_states_21_groups_0, pad = query_states_21_pad_0, pad_type = query_states_21_pad_type_0, strides = query_states_21_strides_0, weight = model_model_layers_5_self_attn_q_proj_weight_palettized, x = var_1559_cast_fp16)[name = string("query_states_21")]; string key_states_31_pad_type_0 = const()[name = string("key_states_31_pad_type_0"), val = string("valid")]; tensor key_states_31_strides_0 = const()[name = string("key_states_31_strides_0"), val = tensor([1, 1])]; tensor key_states_31_pad_0 = const()[name = string("key_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_31_dilations_0 = const()[name = string("key_states_31_dilations_0"), val = tensor([1, 1])]; int32 key_states_31_groups_0 = const()[name = string("key_states_31_groups_0"), val = int32(1)]; tensor key_states_31 = conv(dilations = key_states_31_dilations_0, groups = key_states_31_groups_0, pad = key_states_31_pad_0, pad_type = key_states_31_pad_type_0, strides = key_states_31_strides_0, weight = model_model_layers_5_self_attn_k_proj_weight_palettized, x = var_1559_cast_fp16)[name = string("key_states_31")]; string value_states_31_pad_type_0 = const()[name = string("value_states_31_pad_type_0"), val = string("valid")]; tensor value_states_31_strides_0 = const()[name = string("value_states_31_strides_0"), val = tensor([1, 1])]; tensor value_states_31_pad_0 = const()[name = string("value_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_31_dilations_0 = const()[name = string("value_states_31_dilations_0"), val = tensor([1, 1])]; int32 value_states_31_groups_0 = const()[name = string("value_states_31_groups_0"), val = int32(1)]; tensor value_states_31 = conv(dilations = value_states_31_dilations_0, groups = value_states_31_groups_0, pad = value_states_31_pad_0, pad_type = value_states_31_pad_type_0, strides = value_states_31_strides_0, weight = model_model_layers_5_self_attn_v_proj_weight_palettized, x = var_1559_cast_fp16)[name = string("value_states_31")]; tensor var_1579 = const()[name = string("op_1579"), val = tensor([1, 32, 64, 64])]; tensor var_1580 = reshape(shape = var_1579, x = query_states_21)[name = string("op_1580")]; tensor var_1581 = const()[name = string("op_1581"), val = tensor([0, 1, 3, 2])]; tensor var_1583 = const()[name = string("op_1583"), val = tensor([1, 8, 64, 64])]; tensor var_1584 = reshape(shape = var_1583, x = key_states_31)[name = string("op_1584")]; tensor var_1585 = const()[name = string("op_1585"), val = tensor([0, 1, 3, 2])]; tensor var_1587 = const()[name = string("op_1587"), val = tensor([1, 8, 64, 64])]; tensor var_1588 = reshape(shape = var_1587, x = value_states_31)[name = string("op_1588")]; tensor var_1589 = const()[name = string("op_1589"), val = tensor([0, 1, 3, 2])]; tensor x1_21_begin_0 = const()[name = string("x1_21_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_21_end_0 = const()[name = string("x1_21_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_21_end_mask_0 = const()[name = string("x1_21_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_141 = transpose(perm = var_1581, x = var_1580)[name = string("transpose_73")]; tensor x1_21 = slice_by_index(begin = x1_21_begin_0, end = x1_21_end_0, end_mask = x1_21_end_mask_0, x = x_141)[name = string("x1_21")]; tensor x2_21_begin_0 = const()[name = string("x2_21_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_21_end_0 = const()[name = string("x2_21_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_21_end_mask_0 = const()[name = string("x2_21_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_21 = slice_by_index(begin = x2_21_begin_0, end = x2_21_end_0, end_mask = x2_21_end_mask_0, x = x_141)[name = string("x2_21")]; tensor var_1607 = mul(x = x1_21, y = cos_7)[name = string("op_1607")]; tensor var_1608 = mul(x = x2_21, y = sin_7)[name = string("op_1608")]; tensor var_1609 = sub(x = var_1607, y = var_1608)[name = string("op_1609")]; tensor var_1610 = mul(x = x2_21, y = cos_7)[name = string("op_1610")]; tensor var_1611 = mul(x = x1_21, y = sin_7)[name = string("op_1611")]; tensor var_1612 = add(x = var_1610, y = var_1611)[name = string("op_1612")]; bool rotated_21_interleave_0 = const()[name = string("rotated_21_interleave_0"), val = bool(false)]; tensor rotated_21 = concat(axis = var_73, interleave = rotated_21_interleave_0, values = (var_1609, var_1612))[name = string("rotated_21")]; tensor x1_23_begin_0 = const()[name = string("x1_23_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_23_end_0 = const()[name = string("x1_23_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_23_end_mask_0 = const()[name = string("x1_23_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_145 = transpose(perm = var_1585, x = var_1584)[name = string("transpose_72")]; tensor x1_23 = slice_by_index(begin = x1_23_begin_0, end = x1_23_end_0, end_mask = x1_23_end_mask_0, x = x_145)[name = string("x1_23")]; tensor x2_23_begin_0 = const()[name = string("x2_23_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_23_end_0 = const()[name = string("x2_23_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_23_end_mask_0 = const()[name = string("x2_23_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_23 = slice_by_index(begin = x2_23_begin_0, end = x2_23_end_0, end_mask = x2_23_end_mask_0, x = x_145)[name = string("x2_23")]; tensor var_1628 = mul(x = x1_23, y = cos_7)[name = string("op_1628")]; tensor var_1629 = mul(x = x2_23, y = sin_7)[name = string("op_1629")]; tensor var_1630 = sub(x = var_1628, y = var_1629)[name = string("op_1630")]; tensor var_1631 = mul(x = x2_23, y = cos_7)[name = string("op_1631")]; tensor var_1632 = mul(x = x1_23, y = sin_7)[name = string("op_1632")]; tensor var_1633 = add(x = var_1631, y = var_1632)[name = string("op_1633")]; bool rotated_23_interleave_0 = const()[name = string("rotated_23_interleave_0"), val = bool(false)]; tensor rotated_23 = concat(axis = var_73, interleave = rotated_23_interleave_0, values = (var_1630, var_1633))[name = string("rotated_23")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; int32 concat_92_axis_0 = const()[name = string("concat_92_axis_0"), val = int32(0)]; bool concat_92_interleave_0 = const()[name = string("concat_92_interleave_0"), val = bool(false)]; tensor concat_92 = concat(axis = concat_92_axis_0, interleave = concat_92_interleave_0, values = (expand_dims_60, expand_dims_61, current_pos, expand_dims_63))[name = string("concat_92")]; tensor concat_93_values1_0 = const()[name = string("concat_93_values1_0"), val = tensor([0])]; tensor concat_93_values3_0 = const()[name = string("concat_93_values3_0"), val = tensor([0])]; int32 concat_93_axis_0 = const()[name = string("concat_93_axis_0"), val = int32(0)]; bool concat_93_interleave_0 = const()[name = string("concat_93_interleave_0"), val = bool(false)]; tensor concat_93 = concat(axis = concat_93_axis_0, interleave = concat_93_interleave_0, values = (expand_dims_64, concat_93_values1_0, var_597, concat_93_values3_0))[name = string("concat_93")]; tensor model_model_kv_cache_0_internal_tensor_assign_11_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_92, begin_mask = model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0, end = concat_93, end_mask = model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_11_stride_0, update = rotated_23, x = coreml_update_state_41)[name = string("model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_106_write_state")]; tensor coreml_update_state_42 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_106")]; tensor expand_dims_66 = const()[name = string("expand_dims_66"), val = tensor([21])]; tensor expand_dims_67 = const()[name = string("expand_dims_67"), val = tensor([0])]; tensor expand_dims_69 = const()[name = string("expand_dims_69"), val = tensor([0])]; tensor expand_dims_70 = const()[name = string("expand_dims_70"), val = tensor([22])]; int32 concat_96_axis_0 = const()[name = string("concat_96_axis_0"), val = int32(0)]; bool concat_96_interleave_0 = const()[name = string("concat_96_interleave_0"), val = bool(false)]; tensor concat_96 = concat(axis = concat_96_axis_0, interleave = concat_96_interleave_0, values = (expand_dims_66, expand_dims_67, current_pos, expand_dims_69))[name = string("concat_96")]; tensor concat_97_values1_0 = const()[name = string("concat_97_values1_0"), val = tensor([0])]; tensor concat_97_values3_0 = const()[name = string("concat_97_values3_0"), val = tensor([0])]; int32 concat_97_axis_0 = const()[name = string("concat_97_axis_0"), val = int32(0)]; bool concat_97_interleave_0 = const()[name = string("concat_97_interleave_0"), val = bool(false)]; tensor concat_97 = concat(axis = concat_97_axis_0, interleave = concat_97_interleave_0, values = (expand_dims_70, concat_97_values1_0, var_597, concat_97_values3_0))[name = string("concat_97")]; tensor model_model_kv_cache_0_internal_tensor_assign_12_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_33 = transpose(perm = var_1589, x = var_1588)[name = string("transpose_71")]; tensor model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_96, begin_mask = model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0, end = concat_97, end_mask = model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_12_stride_0, update = value_states_33, x = coreml_update_state_42)[name = string("model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_107_write_state")]; tensor coreml_update_state_43 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_107")]; tensor var_1656_begin_0 = const()[name = string("op_1656_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_1656_end_0 = const()[name = string("op_1656_end_0"), val = tensor([6, 8, 4096, 64])]; tensor var_1656_end_mask_0 = const()[name = string("op_1656_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1656_cast_fp16 = slice_by_index(begin = var_1656_begin_0, end = var_1656_end_0, end_mask = var_1656_end_mask_0, x = coreml_update_state_43)[name = string("op_1656_cast_fp16")]; tensor K_layer_cache_11_axes_0 = const()[name = string("K_layer_cache_11_axes_0"), val = tensor([0])]; tensor K_layer_cache_11_cast_fp16 = squeeze(axes = K_layer_cache_11_axes_0, x = var_1656_cast_fp16)[name = string("K_layer_cache_11_cast_fp16")]; tensor var_1658_begin_0 = const()[name = string("op_1658_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_1658_end_0 = const()[name = string("op_1658_end_0"), val = tensor([22, 8, 4096, 64])]; tensor var_1658_end_mask_0 = const()[name = string("op_1658_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1658_cast_fp16 = slice_by_index(begin = var_1658_begin_0, end = var_1658_end_0, end_mask = var_1658_end_mask_0, x = coreml_update_state_43)[name = string("op_1658_cast_fp16")]; tensor V_layer_cache_11_axes_0 = const()[name = string("V_layer_cache_11_axes_0"), val = tensor([0])]; tensor V_layer_cache_11_cast_fp16 = squeeze(axes = V_layer_cache_11_axes_0, x = var_1658_cast_fp16)[name = string("V_layer_cache_11_cast_fp16")]; tensor x_151_axes_0 = const()[name = string("x_151_axes_0"), val = tensor([1])]; tensor x_151_cast_fp16 = expand_dims(axes = x_151_axes_0, x = K_layer_cache_11_cast_fp16)[name = string("x_151_cast_fp16")]; tensor var_1667 = const()[name = string("op_1667"), val = tensor([1, 4, 1, 1])]; tensor x_153_cast_fp16 = tile(reps = var_1667, x = x_151_cast_fp16)[name = string("x_153_cast_fp16")]; tensor var_1671 = const()[name = string("op_1671"), val = tensor([1, -1, 4096, 64])]; tensor var_1672_cast_fp16 = reshape(shape = var_1671, x = x_153_cast_fp16)[name = string("op_1672_cast_fp16")]; tensor x_157_axes_0 = const()[name = string("x_157_axes_0"), val = tensor([1])]; tensor x_157_cast_fp16 = expand_dims(axes = x_157_axes_0, x = V_layer_cache_11_cast_fp16)[name = string("x_157_cast_fp16")]; tensor var_1674 = const()[name = string("op_1674"), val = tensor([1, 4, 1, 1])]; tensor x_159_cast_fp16 = tile(reps = var_1674, x = x_157_cast_fp16)[name = string("x_159_cast_fp16")]; bool var_1681_transpose_x_0 = const()[name = string("op_1681_transpose_x_0"), val = bool(false)]; bool var_1681_transpose_y_0 = const()[name = string("op_1681_transpose_y_0"), val = bool(true)]; tensor var_1681_cast_fp16 = matmul(transpose_x = var_1681_transpose_x_0, transpose_y = var_1681_transpose_y_0, x = rotated_21, y = var_1672_cast_fp16)[name = string("op_1681_cast_fp16")]; fp16 var_1682_to_fp16 = const()[name = string("op_1682_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_11_cast_fp16 = mul(x = var_1681_cast_fp16, y = var_1682_to_fp16)[name = string("attn_weights_11_cast_fp16")]; tensor x_161_cast_fp16 = add(x = attn_weights_11_cast_fp16, y = causal_mask)[name = string("x_161_cast_fp16")]; tensor reduce_max_5_axes_0 = const()[name = string("reduce_max_5_axes_0"), val = tensor([-1])]; bool reduce_max_5_keep_dims_0 = const()[name = string("reduce_max_5_keep_dims_0"), val = bool(true)]; tensor reduce_max_5_cast_fp16 = reduce_max(axes = reduce_max_5_axes_0, keep_dims = reduce_max_5_keep_dims_0, x = x_161_cast_fp16)[name = string("reduce_max_5_cast_fp16")]; tensor x_163_cast_fp16 = sub(x = x_161_cast_fp16, y = reduce_max_5_cast_fp16)[name = string("x_163_cast_fp16")]; tensor exp_x_11_cast_fp16 = exp(x = x_163_cast_fp16)[name = string("exp_x_11_cast_fp16")]; tensor var_1693_axes_0 = const()[name = string("op_1693_axes_0"), val = tensor([-1])]; bool var_1693_keep_dims_0 = const()[name = string("op_1693_keep_dims_0"), val = bool(true)]; tensor var_1693_cast_fp16 = reduce_sum(axes = var_1693_axes_0, keep_dims = var_1693_keep_dims_0, x = exp_x_11_cast_fp16)[name = string("op_1693_cast_fp16")]; tensor var_1694_cast_fp16 = real_div(x = exp_x_11_cast_fp16, y = var_1693_cast_fp16)[name = string("op_1694_cast_fp16")]; tensor concat_102 = const()[name = string("concat_102"), val = tensor([32, 64, 4096])]; tensor reshape_15_cast_fp16 = reshape(shape = concat_102, x = var_1694_cast_fp16)[name = string("reshape_15_cast_fp16")]; tensor concat_103 = const()[name = string("concat_103"), val = tensor([32, 4096, 64])]; tensor reshape_16_cast_fp16 = reshape(shape = concat_103, x = x_159_cast_fp16)[name = string("reshape_16_cast_fp16")]; bool matmul_5_transpose_x_0 = const()[name = string("matmul_5_transpose_x_0"), val = bool(false)]; bool matmul_5_transpose_y_0 = const()[name = string("matmul_5_transpose_y_0"), val = bool(false)]; tensor matmul_5_cast_fp16 = matmul(transpose_x = matmul_5_transpose_x_0, transpose_y = matmul_5_transpose_y_0, x = reshape_15_cast_fp16, y = reshape_16_cast_fp16)[name = string("matmul_5_cast_fp16")]; tensor concat_107 = const()[name = string("concat_107"), val = tensor([1, 32, 64, 64])]; tensor reshape_17_cast_fp16 = reshape(shape = concat_107, x = matmul_5_cast_fp16)[name = string("reshape_17_cast_fp16")]; tensor var_1697_perm_0 = const()[name = string("op_1697_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1699 = const()[name = string("op_1699"), val = tensor([1, 64, 2048])]; tensor var_1697_cast_fp16 = transpose(perm = var_1697_perm_0, x = reshape_17_cast_fp16)[name = string("transpose_70")]; tensor input_75_cast_fp16 = reshape(shape = var_1699, x = var_1697_cast_fp16)[name = string("input_75_cast_fp16")]; tensor model_model_layers_5_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(467048000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(469145216))))[name = string("model_model_layers_5_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_5_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_5_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_75_cast_fp16)[name = string("linear_5_cast_fp16")]; tensor hidden_states_45_cast_fp16 = add(x = hidden_states_41_cast_fp16, y = linear_5_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; fp16 const_105_promoted_to_fp16 = const()[name = string("const_105_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1705_cast_fp16 = mul(x = hidden_states_45_cast_fp16, y = const_105_promoted_to_fp16)[name = string("op_1705_cast_fp16")]; bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; tensor input_77_cast_fp16 = concat(axis = var_73, interleave = input_77_interleave_0, values = (hidden_states_45_cast_fp16, var_1705_cast_fp16))[name = string("input_77_cast_fp16")]; tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_76_to_fp16, x = input_77_cast_fp16)[name = string("normed_45_cast_fp16")]; tensor normed_47_begin_0 = const()[name = string("normed_47_begin_0"), val = tensor([0, 0, 0])]; tensor normed_47_end_0 = const()[name = string("normed_47_end_0"), val = tensor([1, 64, 2048])]; tensor normed_47_end_mask_0 = const()[name = string("normed_47_end_mask_0"), val = tensor([true, true, false])]; tensor normed_47_cast_fp16 = slice_by_index(begin = normed_47_begin_0, end = normed_47_end_0, end_mask = normed_47_end_mask_0, x = normed_45_cast_fp16)[name = string("normed_47_cast_fp16")]; tensor const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(469153472)))]; tensor x_165_cast_fp16 = mul(x = normed_47_cast_fp16, y = const_108_promoted_to_fp16)[name = string("x_165_cast_fp16")]; tensor var_1723 = const()[name = string("op_1723"), val = tensor([0, 2, 1])]; tensor input_79_axes_0 = const()[name = string("input_79_axes_0"), val = tensor([2])]; tensor var_1724 = transpose(perm = var_1723, x = x_165_cast_fp16)[name = string("transpose_69")]; tensor input_79 = expand_dims(axes = input_79_axes_0, x = var_1724)[name = string("input_79")]; string input_81_pad_type_0 = const()[name = string("input_81_pad_type_0"), val = string("valid")]; tensor input_81_strides_0 = const()[name = string("input_81_strides_0"), val = tensor([1, 1])]; tensor input_81_pad_0 = const()[name = string("input_81_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_81_dilations_0 = const()[name = string("input_81_dilations_0"), val = tensor([1, 1])]; int32 input_81_groups_0 = const()[name = string("input_81_groups_0"), val = int32(1)]; tensor input_81 = conv(dilations = input_81_dilations_0, groups = input_81_groups_0, pad = input_81_pad_0, pad_type = input_81_pad_type_0, strides = input_81_strides_0, weight = model_model_layers_5_mlp_gate_proj_weight_palettized, x = input_79)[name = string("input_81")]; string up_states_11_pad_type_0 = const()[name = string("up_states_11_pad_type_0"), val = string("valid")]; tensor up_states_11_strides_0 = const()[name = string("up_states_11_strides_0"), val = tensor([1, 1])]; tensor up_states_11_pad_0 = const()[name = string("up_states_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_11_dilations_0 = const()[name = string("up_states_11_dilations_0"), val = tensor([1, 1])]; int32 up_states_11_groups_0 = const()[name = string("up_states_11_groups_0"), val = int32(1)]; tensor up_states_11 = conv(dilations = up_states_11_dilations_0, groups = up_states_11_groups_0, pad = up_states_11_pad_0, pad_type = up_states_11_pad_type_0, strides = up_states_11_strides_0, weight = model_model_layers_5_mlp_up_proj_weight_palettized, x = input_79)[name = string("up_states_11")]; tensor gate_states_11 = silu(x = input_81)[name = string("gate_states_11")]; tensor input_83 = mul(x = gate_states_11, y = up_states_11)[name = string("input_83")]; string hidden_states_47_pad_type_0 = const()[name = string("hidden_states_47_pad_type_0"), val = string("valid")]; tensor hidden_states_47_strides_0 = const()[name = string("hidden_states_47_strides_0"), val = tensor([1, 1])]; tensor hidden_states_47_pad_0 = const()[name = string("hidden_states_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_47_dilations_0 = const()[name = string("hidden_states_47_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_47_groups_0 = const()[name = string("hidden_states_47_groups_0"), val = int32(1)]; tensor hidden_states_47 = conv(dilations = hidden_states_47_dilations_0, groups = hidden_states_47_groups_0, pad = hidden_states_47_pad_0, pad_type = hidden_states_47_pad_type_0, strides = hidden_states_47_strides_0, weight = model_model_layers_5_mlp_down_proj_weight_palettized, x = input_83)[name = string("hidden_states_47")]; tensor var_1746_axes_0 = const()[name = string("op_1746_axes_0"), val = tensor([2])]; tensor var_1746 = squeeze(axes = var_1746_axes_0, x = hidden_states_47)[name = string("op_1746")]; tensor var_1747 = const()[name = string("op_1747"), val = tensor([0, 2, 1])]; tensor var_1748 = transpose(perm = var_1747, x = var_1746)[name = string("transpose_68")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = var_1748)[name = string("hidden_states_49_cast_fp16")]; fp16 const_109_promoted_to_fp16 = const()[name = string("const_109_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1751_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_109_promoted_to_fp16)[name = string("op_1751_cast_fp16")]; bool input_85_interleave_0 = const()[name = string("input_85_interleave_0"), val = bool(false)]; tensor input_85_cast_fp16 = concat(axis = var_73, interleave = input_85_interleave_0, values = (hidden_states_49_cast_fp16, var_1751_cast_fp16))[name = string("input_85_cast_fp16")]; tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_76_to_fp16, x = input_85_cast_fp16)[name = string("normed_49_cast_fp16")]; tensor normed_51_begin_0 = const()[name = string("normed_51_begin_0"), val = tensor([0, 0, 0])]; tensor normed_51_end_0 = const()[name = string("normed_51_end_0"), val = tensor([1, 64, 2048])]; tensor normed_51_end_mask_0 = const()[name = string("normed_51_end_mask_0"), val = tensor([true, true, false])]; tensor normed_51_cast_fp16 = slice_by_index(begin = normed_51_begin_0, end = normed_51_end_0, end_mask = normed_51_end_mask_0, x = normed_49_cast_fp16)[name = string("normed_51_cast_fp16")]; tensor const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(469157632)))]; tensor hidden_states_51_cast_fp16 = mul(x = normed_51_cast_fp16, y = const_112_promoted_to_fp16)[name = string("hidden_states_51_cast_fp16")]; tensor var_1766 = const()[name = string("op_1766"), val = tensor([0, 2, 1])]; tensor var_1768_axes_0 = const()[name = string("op_1768_axes_0"), val = tensor([2])]; tensor var_1767_cast_fp16 = transpose(perm = var_1766, x = hidden_states_51_cast_fp16)[name = string("transpose_67")]; tensor var_1768_cast_fp16 = expand_dims(axes = var_1768_axes_0, x = var_1767_cast_fp16)[name = string("op_1768_cast_fp16")]; string query_states_25_pad_type_0 = const()[name = string("query_states_25_pad_type_0"), val = string("valid")]; tensor query_states_25_strides_0 = const()[name = string("query_states_25_strides_0"), val = tensor([1, 1])]; tensor query_states_25_pad_0 = const()[name = string("query_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_25_dilations_0 = const()[name = string("query_states_25_dilations_0"), val = tensor([1, 1])]; int32 query_states_25_groups_0 = const()[name = string("query_states_25_groups_0"), val = int32(1)]; tensor query_states_25 = conv(dilations = query_states_25_dilations_0, groups = query_states_25_groups_0, pad = query_states_25_pad_0, pad_type = query_states_25_pad_type_0, strides = query_states_25_strides_0, weight = model_model_layers_6_self_attn_q_proj_weight_palettized, x = var_1768_cast_fp16)[name = string("query_states_25")]; string key_states_37_pad_type_0 = const()[name = string("key_states_37_pad_type_0"), val = string("valid")]; tensor key_states_37_strides_0 = const()[name = string("key_states_37_strides_0"), val = tensor([1, 1])]; tensor key_states_37_pad_0 = const()[name = string("key_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_37_dilations_0 = const()[name = string("key_states_37_dilations_0"), val = tensor([1, 1])]; int32 key_states_37_groups_0 = const()[name = string("key_states_37_groups_0"), val = int32(1)]; tensor key_states_37 = conv(dilations = key_states_37_dilations_0, groups = key_states_37_groups_0, pad = key_states_37_pad_0, pad_type = key_states_37_pad_type_0, strides = key_states_37_strides_0, weight = model_model_layers_6_self_attn_k_proj_weight_palettized, x = var_1768_cast_fp16)[name = string("key_states_37")]; string value_states_37_pad_type_0 = const()[name = string("value_states_37_pad_type_0"), val = string("valid")]; tensor value_states_37_strides_0 = const()[name = string("value_states_37_strides_0"), val = tensor([1, 1])]; tensor value_states_37_pad_0 = const()[name = string("value_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_37_dilations_0 = const()[name = string("value_states_37_dilations_0"), val = tensor([1, 1])]; int32 value_states_37_groups_0 = const()[name = string("value_states_37_groups_0"), val = int32(1)]; tensor value_states_37 = conv(dilations = value_states_37_dilations_0, groups = value_states_37_groups_0, pad = value_states_37_pad_0, pad_type = value_states_37_pad_type_0, strides = value_states_37_strides_0, weight = model_model_layers_6_self_attn_v_proj_weight_palettized, x = var_1768_cast_fp16)[name = string("value_states_37")]; tensor var_1788 = const()[name = string("op_1788"), val = tensor([1, 32, 64, 64])]; tensor var_1789 = reshape(shape = var_1788, x = query_states_25)[name = string("op_1789")]; tensor var_1790 = const()[name = string("op_1790"), val = tensor([0, 1, 3, 2])]; tensor var_1792 = const()[name = string("op_1792"), val = tensor([1, 8, 64, 64])]; tensor var_1793 = reshape(shape = var_1792, x = key_states_37)[name = string("op_1793")]; tensor var_1794 = const()[name = string("op_1794"), val = tensor([0, 1, 3, 2])]; tensor var_1796 = const()[name = string("op_1796"), val = tensor([1, 8, 64, 64])]; tensor var_1797 = reshape(shape = var_1796, x = value_states_37)[name = string("op_1797")]; tensor var_1798 = const()[name = string("op_1798"), val = tensor([0, 1, 3, 2])]; tensor x1_25_begin_0 = const()[name = string("x1_25_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_25_end_0 = const()[name = string("x1_25_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_25_end_mask_0 = const()[name = string("x1_25_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_169 = transpose(perm = var_1790, x = var_1789)[name = string("transpose_66")]; tensor x1_25 = slice_by_index(begin = x1_25_begin_0, end = x1_25_end_0, end_mask = x1_25_end_mask_0, x = x_169)[name = string("x1_25")]; tensor x2_25_begin_0 = const()[name = string("x2_25_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_25_end_0 = const()[name = string("x2_25_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_25_end_mask_0 = const()[name = string("x2_25_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_25 = slice_by_index(begin = x2_25_begin_0, end = x2_25_end_0, end_mask = x2_25_end_mask_0, x = x_169)[name = string("x2_25")]; tensor var_1816 = mul(x = x1_25, y = cos_7)[name = string("op_1816")]; tensor var_1817 = mul(x = x2_25, y = sin_7)[name = string("op_1817")]; tensor var_1818 = sub(x = var_1816, y = var_1817)[name = string("op_1818")]; tensor var_1819 = mul(x = x2_25, y = cos_7)[name = string("op_1819")]; tensor var_1820 = mul(x = x1_25, y = sin_7)[name = string("op_1820")]; tensor var_1821 = add(x = var_1819, y = var_1820)[name = string("op_1821")]; bool rotated_25_interleave_0 = const()[name = string("rotated_25_interleave_0"), val = bool(false)]; tensor rotated_25 = concat(axis = var_73, interleave = rotated_25_interleave_0, values = (var_1818, var_1821))[name = string("rotated_25")]; tensor x1_27_begin_0 = const()[name = string("x1_27_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_27_end_0 = const()[name = string("x1_27_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_27_end_mask_0 = const()[name = string("x1_27_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_173 = transpose(perm = var_1794, x = var_1793)[name = string("transpose_65")]; tensor x1_27 = slice_by_index(begin = x1_27_begin_0, end = x1_27_end_0, end_mask = x1_27_end_mask_0, x = x_173)[name = string("x1_27")]; tensor x2_27_begin_0 = const()[name = string("x2_27_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_27_end_0 = const()[name = string("x2_27_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_27_end_mask_0 = const()[name = string("x2_27_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_27 = slice_by_index(begin = x2_27_begin_0, end = x2_27_end_0, end_mask = x2_27_end_mask_0, x = x_173)[name = string("x2_27")]; tensor var_1837 = mul(x = x1_27, y = cos_7)[name = string("op_1837")]; tensor var_1838 = mul(x = x2_27, y = sin_7)[name = string("op_1838")]; tensor var_1839 = sub(x = var_1837, y = var_1838)[name = string("op_1839")]; tensor var_1840 = mul(x = x2_27, y = cos_7)[name = string("op_1840")]; tensor var_1841 = mul(x = x1_27, y = sin_7)[name = string("op_1841")]; tensor var_1842 = add(x = var_1840, y = var_1841)[name = string("op_1842")]; bool rotated_27_interleave_0 = const()[name = string("rotated_27_interleave_0"), val = bool(false)]; tensor rotated_27 = concat(axis = var_73, interleave = rotated_27_interleave_0, values = (var_1839, var_1842))[name = string("rotated_27")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; int32 concat_110_axis_0 = const()[name = string("concat_110_axis_0"), val = int32(0)]; bool concat_110_interleave_0 = const()[name = string("concat_110_interleave_0"), val = bool(false)]; tensor concat_110 = concat(axis = concat_110_axis_0, interleave = concat_110_interleave_0, values = (expand_dims_72, expand_dims_73, current_pos, expand_dims_75))[name = string("concat_110")]; tensor concat_111_values1_0 = const()[name = string("concat_111_values1_0"), val = tensor([0])]; tensor concat_111_values3_0 = const()[name = string("concat_111_values3_0"), val = tensor([0])]; int32 concat_111_axis_0 = const()[name = string("concat_111_axis_0"), val = int32(0)]; bool concat_111_interleave_0 = const()[name = string("concat_111_interleave_0"), val = bool(false)]; tensor concat_111 = concat(axis = concat_111_axis_0, interleave = concat_111_interleave_0, values = (expand_dims_76, concat_111_values1_0, var_597, concat_111_values3_0))[name = string("concat_111")]; tensor model_model_kv_cache_0_internal_tensor_assign_13_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_110, begin_mask = model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0, end = concat_111, end_mask = model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_13_stride_0, update = rotated_27, x = coreml_update_state_43)[name = string("model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_108_write_state")]; tensor coreml_update_state_44 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_108")]; tensor expand_dims_78 = const()[name = string("expand_dims_78"), val = tensor([22])]; tensor expand_dims_79 = const()[name = string("expand_dims_79"), val = tensor([0])]; tensor expand_dims_81 = const()[name = string("expand_dims_81"), val = tensor([0])]; tensor expand_dims_82 = const()[name = string("expand_dims_82"), val = tensor([23])]; int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_78, expand_dims_79, current_pos, expand_dims_81))[name = string("concat_114")]; tensor concat_115_values1_0 = const()[name = string("concat_115_values1_0"), val = tensor([0])]; tensor concat_115_values3_0 = const()[name = string("concat_115_values3_0"), val = tensor([0])]; int32 concat_115_axis_0 = const()[name = string("concat_115_axis_0"), val = int32(0)]; bool concat_115_interleave_0 = const()[name = string("concat_115_interleave_0"), val = bool(false)]; tensor concat_115 = concat(axis = concat_115_axis_0, interleave = concat_115_interleave_0, values = (expand_dims_82, concat_115_values1_0, var_597, concat_115_values3_0))[name = string("concat_115")]; tensor model_model_kv_cache_0_internal_tensor_assign_14_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_39 = transpose(perm = var_1798, x = var_1797)[name = string("transpose_64")]; tensor model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_114, begin_mask = model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0, end = concat_115, end_mask = model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_14_stride_0, update = value_states_39, x = coreml_update_state_44)[name = string("model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_109_write_state")]; tensor coreml_update_state_45 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_109")]; tensor var_1865_begin_0 = const()[name = string("op_1865_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_1865_end_0 = const()[name = string("op_1865_end_0"), val = tensor([7, 8, 4096, 64])]; tensor var_1865_end_mask_0 = const()[name = string("op_1865_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1865_cast_fp16 = slice_by_index(begin = var_1865_begin_0, end = var_1865_end_0, end_mask = var_1865_end_mask_0, x = coreml_update_state_45)[name = string("op_1865_cast_fp16")]; tensor K_layer_cache_13_axes_0 = const()[name = string("K_layer_cache_13_axes_0"), val = tensor([0])]; tensor K_layer_cache_13_cast_fp16 = squeeze(axes = K_layer_cache_13_axes_0, x = var_1865_cast_fp16)[name = string("K_layer_cache_13_cast_fp16")]; tensor var_1867_begin_0 = const()[name = string("op_1867_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_1867_end_0 = const()[name = string("op_1867_end_0"), val = tensor([23, 8, 4096, 64])]; tensor var_1867_end_mask_0 = const()[name = string("op_1867_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1867_cast_fp16 = slice_by_index(begin = var_1867_begin_0, end = var_1867_end_0, end_mask = var_1867_end_mask_0, x = coreml_update_state_45)[name = string("op_1867_cast_fp16")]; tensor V_layer_cache_13_axes_0 = const()[name = string("V_layer_cache_13_axes_0"), val = tensor([0])]; tensor V_layer_cache_13_cast_fp16 = squeeze(axes = V_layer_cache_13_axes_0, x = var_1867_cast_fp16)[name = string("V_layer_cache_13_cast_fp16")]; tensor x_179_axes_0 = const()[name = string("x_179_axes_0"), val = tensor([1])]; tensor x_179_cast_fp16 = expand_dims(axes = x_179_axes_0, x = K_layer_cache_13_cast_fp16)[name = string("x_179_cast_fp16")]; tensor var_1876 = const()[name = string("op_1876"), val = tensor([1, 4, 1, 1])]; tensor x_181_cast_fp16 = tile(reps = var_1876, x = x_179_cast_fp16)[name = string("x_181_cast_fp16")]; tensor var_1880 = const()[name = string("op_1880"), val = tensor([1, -1, 4096, 64])]; tensor var_1881_cast_fp16 = reshape(shape = var_1880, x = x_181_cast_fp16)[name = string("op_1881_cast_fp16")]; tensor x_185_axes_0 = const()[name = string("x_185_axes_0"), val = tensor([1])]; tensor x_185_cast_fp16 = expand_dims(axes = x_185_axes_0, x = V_layer_cache_13_cast_fp16)[name = string("x_185_cast_fp16")]; tensor var_1883 = const()[name = string("op_1883"), val = tensor([1, 4, 1, 1])]; tensor x_187_cast_fp16 = tile(reps = var_1883, x = x_185_cast_fp16)[name = string("x_187_cast_fp16")]; bool var_1890_transpose_x_0 = const()[name = string("op_1890_transpose_x_0"), val = bool(false)]; bool var_1890_transpose_y_0 = const()[name = string("op_1890_transpose_y_0"), val = bool(true)]; tensor var_1890_cast_fp16 = matmul(transpose_x = var_1890_transpose_x_0, transpose_y = var_1890_transpose_y_0, x = rotated_25, y = var_1881_cast_fp16)[name = string("op_1890_cast_fp16")]; fp16 var_1891_to_fp16 = const()[name = string("op_1891_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_13_cast_fp16 = mul(x = var_1890_cast_fp16, y = var_1891_to_fp16)[name = string("attn_weights_13_cast_fp16")]; tensor x_189_cast_fp16 = add(x = attn_weights_13_cast_fp16, y = causal_mask)[name = string("x_189_cast_fp16")]; tensor reduce_max_6_axes_0 = const()[name = string("reduce_max_6_axes_0"), val = tensor([-1])]; bool reduce_max_6_keep_dims_0 = const()[name = string("reduce_max_6_keep_dims_0"), val = bool(true)]; tensor reduce_max_6_cast_fp16 = reduce_max(axes = reduce_max_6_axes_0, keep_dims = reduce_max_6_keep_dims_0, x = x_189_cast_fp16)[name = string("reduce_max_6_cast_fp16")]; tensor x_191_cast_fp16 = sub(x = x_189_cast_fp16, y = reduce_max_6_cast_fp16)[name = string("x_191_cast_fp16")]; tensor exp_x_13_cast_fp16 = exp(x = x_191_cast_fp16)[name = string("exp_x_13_cast_fp16")]; tensor var_1902_axes_0 = const()[name = string("op_1902_axes_0"), val = tensor([-1])]; bool var_1902_keep_dims_0 = const()[name = string("op_1902_keep_dims_0"), val = bool(true)]; tensor var_1902_cast_fp16 = reduce_sum(axes = var_1902_axes_0, keep_dims = var_1902_keep_dims_0, x = exp_x_13_cast_fp16)[name = string("op_1902_cast_fp16")]; tensor var_1903_cast_fp16 = real_div(x = exp_x_13_cast_fp16, y = var_1902_cast_fp16)[name = string("op_1903_cast_fp16")]; tensor concat_120 = const()[name = string("concat_120"), val = tensor([32, 64, 4096])]; tensor reshape_18_cast_fp16 = reshape(shape = concat_120, x = var_1903_cast_fp16)[name = string("reshape_18_cast_fp16")]; tensor concat_121 = const()[name = string("concat_121"), val = tensor([32, 4096, 64])]; tensor reshape_19_cast_fp16 = reshape(shape = concat_121, x = x_187_cast_fp16)[name = string("reshape_19_cast_fp16")]; bool matmul_6_transpose_x_0 = const()[name = string("matmul_6_transpose_x_0"), val = bool(false)]; bool matmul_6_transpose_y_0 = const()[name = string("matmul_6_transpose_y_0"), val = bool(false)]; tensor matmul_6_cast_fp16 = matmul(transpose_x = matmul_6_transpose_x_0, transpose_y = matmul_6_transpose_y_0, x = reshape_18_cast_fp16, y = reshape_19_cast_fp16)[name = string("matmul_6_cast_fp16")]; tensor concat_125 = const()[name = string("concat_125"), val = tensor([1, 32, 64, 64])]; tensor reshape_20_cast_fp16 = reshape(shape = concat_125, x = matmul_6_cast_fp16)[name = string("reshape_20_cast_fp16")]; tensor var_1906_perm_0 = const()[name = string("op_1906_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_1908 = const()[name = string("op_1908"), val = tensor([1, 64, 2048])]; tensor var_1906_cast_fp16 = transpose(perm = var_1906_perm_0, x = reshape_20_cast_fp16)[name = string("transpose_63")]; tensor input_89_cast_fp16 = reshape(shape = var_1908, x = var_1906_cast_fp16)[name = string("input_89_cast_fp16")]; tensor model_model_layers_6_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(469161792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(471259008))))[name = string("model_model_layers_6_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_6_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_6_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_89_cast_fp16)[name = string("linear_6_cast_fp16")]; tensor hidden_states_53_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = linear_6_cast_fp16)[name = string("hidden_states_53_cast_fp16")]; fp16 const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1914_cast_fp16 = mul(x = hidden_states_53_cast_fp16, y = const_123_promoted_to_fp16)[name = string("op_1914_cast_fp16")]; bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; tensor input_91_cast_fp16 = concat(axis = var_73, interleave = input_91_interleave_0, values = (hidden_states_53_cast_fp16, var_1914_cast_fp16))[name = string("input_91_cast_fp16")]; tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_76_to_fp16, x = input_91_cast_fp16)[name = string("normed_53_cast_fp16")]; tensor normed_55_begin_0 = const()[name = string("normed_55_begin_0"), val = tensor([0, 0, 0])]; tensor normed_55_end_0 = const()[name = string("normed_55_end_0"), val = tensor([1, 64, 2048])]; tensor normed_55_end_mask_0 = const()[name = string("normed_55_end_mask_0"), val = tensor([true, true, false])]; tensor normed_55_cast_fp16 = slice_by_index(begin = normed_55_begin_0, end = normed_55_end_0, end_mask = normed_55_end_mask_0, x = normed_53_cast_fp16)[name = string("normed_55_cast_fp16")]; tensor const_126_promoted_to_fp16 = const()[name = string("const_126_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(471267264)))]; tensor x_193_cast_fp16 = mul(x = normed_55_cast_fp16, y = const_126_promoted_to_fp16)[name = string("x_193_cast_fp16")]; tensor var_1932 = const()[name = string("op_1932"), val = tensor([0, 2, 1])]; tensor input_93_axes_0 = const()[name = string("input_93_axes_0"), val = tensor([2])]; tensor var_1933 = transpose(perm = var_1932, x = x_193_cast_fp16)[name = string("transpose_62")]; tensor input_93 = expand_dims(axes = input_93_axes_0, x = var_1933)[name = string("input_93")]; string input_95_pad_type_0 = const()[name = string("input_95_pad_type_0"), val = string("valid")]; tensor input_95_strides_0 = const()[name = string("input_95_strides_0"), val = tensor([1, 1])]; tensor input_95_pad_0 = const()[name = string("input_95_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_95_dilations_0 = const()[name = string("input_95_dilations_0"), val = tensor([1, 1])]; int32 input_95_groups_0 = const()[name = string("input_95_groups_0"), val = int32(1)]; tensor input_95 = conv(dilations = input_95_dilations_0, groups = input_95_groups_0, pad = input_95_pad_0, pad_type = input_95_pad_type_0, strides = input_95_strides_0, weight = model_model_layers_6_mlp_gate_proj_weight_palettized, x = input_93)[name = string("input_95")]; string up_states_13_pad_type_0 = const()[name = string("up_states_13_pad_type_0"), val = string("valid")]; tensor up_states_13_strides_0 = const()[name = string("up_states_13_strides_0"), val = tensor([1, 1])]; tensor up_states_13_pad_0 = const()[name = string("up_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_13_dilations_0 = const()[name = string("up_states_13_dilations_0"), val = tensor([1, 1])]; int32 up_states_13_groups_0 = const()[name = string("up_states_13_groups_0"), val = int32(1)]; tensor up_states_13 = conv(dilations = up_states_13_dilations_0, groups = up_states_13_groups_0, pad = up_states_13_pad_0, pad_type = up_states_13_pad_type_0, strides = up_states_13_strides_0, weight = model_model_layers_6_mlp_up_proj_weight_palettized, x = input_93)[name = string("up_states_13")]; tensor gate_states_13 = silu(x = input_95)[name = string("gate_states_13")]; tensor input_97 = mul(x = gate_states_13, y = up_states_13)[name = string("input_97")]; string hidden_states_55_pad_type_0 = const()[name = string("hidden_states_55_pad_type_0"), val = string("valid")]; tensor hidden_states_55_strides_0 = const()[name = string("hidden_states_55_strides_0"), val = tensor([1, 1])]; tensor hidden_states_55_pad_0 = const()[name = string("hidden_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_55_dilations_0 = const()[name = string("hidden_states_55_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_55_groups_0 = const()[name = string("hidden_states_55_groups_0"), val = int32(1)]; tensor hidden_states_55 = conv(dilations = hidden_states_55_dilations_0, groups = hidden_states_55_groups_0, pad = hidden_states_55_pad_0, pad_type = hidden_states_55_pad_type_0, strides = hidden_states_55_strides_0, weight = model_model_layers_6_mlp_down_proj_weight_palettized, x = input_97)[name = string("hidden_states_55")]; tensor var_1955_axes_0 = const()[name = string("op_1955_axes_0"), val = tensor([2])]; tensor var_1955 = squeeze(axes = var_1955_axes_0, x = hidden_states_55)[name = string("op_1955")]; tensor var_1956 = const()[name = string("op_1956"), val = tensor([0, 2, 1])]; tensor var_1957 = transpose(perm = var_1956, x = var_1955)[name = string("transpose_61")]; tensor hidden_states_57_cast_fp16 = add(x = hidden_states_53_cast_fp16, y = var_1957)[name = string("hidden_states_57_cast_fp16")]; fp16 const_127_promoted_to_fp16 = const()[name = string("const_127_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1960_cast_fp16 = mul(x = hidden_states_57_cast_fp16, y = const_127_promoted_to_fp16)[name = string("op_1960_cast_fp16")]; bool input_99_interleave_0 = const()[name = string("input_99_interleave_0"), val = bool(false)]; tensor input_99_cast_fp16 = concat(axis = var_73, interleave = input_99_interleave_0, values = (hidden_states_57_cast_fp16, var_1960_cast_fp16))[name = string("input_99_cast_fp16")]; tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_76_to_fp16, x = input_99_cast_fp16)[name = string("normed_57_cast_fp16")]; tensor normed_59_begin_0 = const()[name = string("normed_59_begin_0"), val = tensor([0, 0, 0])]; tensor normed_59_end_0 = const()[name = string("normed_59_end_0"), val = tensor([1, 64, 2048])]; tensor normed_59_end_mask_0 = const()[name = string("normed_59_end_mask_0"), val = tensor([true, true, false])]; tensor normed_59_cast_fp16 = slice_by_index(begin = normed_59_begin_0, end = normed_59_end_0, end_mask = normed_59_end_mask_0, x = normed_57_cast_fp16)[name = string("normed_59_cast_fp16")]; tensor const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(471271424)))]; tensor hidden_states_59_cast_fp16 = mul(x = normed_59_cast_fp16, y = const_130_promoted_to_fp16)[name = string("hidden_states_59_cast_fp16")]; tensor var_1975 = const()[name = string("op_1975"), val = tensor([0, 2, 1])]; tensor var_1977_axes_0 = const()[name = string("op_1977_axes_0"), val = tensor([2])]; tensor var_1976_cast_fp16 = transpose(perm = var_1975, x = hidden_states_59_cast_fp16)[name = string("transpose_60")]; tensor var_1977_cast_fp16 = expand_dims(axes = var_1977_axes_0, x = var_1976_cast_fp16)[name = string("op_1977_cast_fp16")]; string query_states_29_pad_type_0 = const()[name = string("query_states_29_pad_type_0"), val = string("valid")]; tensor query_states_29_strides_0 = const()[name = string("query_states_29_strides_0"), val = tensor([1, 1])]; tensor query_states_29_pad_0 = const()[name = string("query_states_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_29_dilations_0 = const()[name = string("query_states_29_dilations_0"), val = tensor([1, 1])]; int32 query_states_29_groups_0 = const()[name = string("query_states_29_groups_0"), val = int32(1)]; tensor query_states_29 = conv(dilations = query_states_29_dilations_0, groups = query_states_29_groups_0, pad = query_states_29_pad_0, pad_type = query_states_29_pad_type_0, strides = query_states_29_strides_0, weight = model_model_layers_7_self_attn_q_proj_weight_palettized, x = var_1977_cast_fp16)[name = string("query_states_29")]; string key_states_43_pad_type_0 = const()[name = string("key_states_43_pad_type_0"), val = string("valid")]; tensor key_states_43_strides_0 = const()[name = string("key_states_43_strides_0"), val = tensor([1, 1])]; tensor key_states_43_pad_0 = const()[name = string("key_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_43_dilations_0 = const()[name = string("key_states_43_dilations_0"), val = tensor([1, 1])]; int32 key_states_43_groups_0 = const()[name = string("key_states_43_groups_0"), val = int32(1)]; tensor key_states_43 = conv(dilations = key_states_43_dilations_0, groups = key_states_43_groups_0, pad = key_states_43_pad_0, pad_type = key_states_43_pad_type_0, strides = key_states_43_strides_0, weight = model_model_layers_7_self_attn_k_proj_weight_palettized, x = var_1977_cast_fp16)[name = string("key_states_43")]; string value_states_43_pad_type_0 = const()[name = string("value_states_43_pad_type_0"), val = string("valid")]; tensor value_states_43_strides_0 = const()[name = string("value_states_43_strides_0"), val = tensor([1, 1])]; tensor value_states_43_pad_0 = const()[name = string("value_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_43_dilations_0 = const()[name = string("value_states_43_dilations_0"), val = tensor([1, 1])]; int32 value_states_43_groups_0 = const()[name = string("value_states_43_groups_0"), val = int32(1)]; tensor value_states_43 = conv(dilations = value_states_43_dilations_0, groups = value_states_43_groups_0, pad = value_states_43_pad_0, pad_type = value_states_43_pad_type_0, strides = value_states_43_strides_0, weight = model_model_layers_7_self_attn_v_proj_weight_palettized, x = var_1977_cast_fp16)[name = string("value_states_43")]; tensor var_1997 = const()[name = string("op_1997"), val = tensor([1, 32, 64, 64])]; tensor var_1998 = reshape(shape = var_1997, x = query_states_29)[name = string("op_1998")]; tensor var_1999 = const()[name = string("op_1999"), val = tensor([0, 1, 3, 2])]; tensor var_2001 = const()[name = string("op_2001"), val = tensor([1, 8, 64, 64])]; tensor var_2002 = reshape(shape = var_2001, x = key_states_43)[name = string("op_2002")]; tensor var_2003 = const()[name = string("op_2003"), val = tensor([0, 1, 3, 2])]; tensor var_2005 = const()[name = string("op_2005"), val = tensor([1, 8, 64, 64])]; tensor var_2006 = reshape(shape = var_2005, x = value_states_43)[name = string("op_2006")]; tensor var_2007 = const()[name = string("op_2007"), val = tensor([0, 1, 3, 2])]; tensor x1_29_begin_0 = const()[name = string("x1_29_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_29_end_0 = const()[name = string("x1_29_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_29_end_mask_0 = const()[name = string("x1_29_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_197 = transpose(perm = var_1999, x = var_1998)[name = string("transpose_59")]; tensor x1_29 = slice_by_index(begin = x1_29_begin_0, end = x1_29_end_0, end_mask = x1_29_end_mask_0, x = x_197)[name = string("x1_29")]; tensor x2_29_begin_0 = const()[name = string("x2_29_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_29_end_0 = const()[name = string("x2_29_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_29_end_mask_0 = const()[name = string("x2_29_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_29 = slice_by_index(begin = x2_29_begin_0, end = x2_29_end_0, end_mask = x2_29_end_mask_0, x = x_197)[name = string("x2_29")]; tensor var_2025 = mul(x = x1_29, y = cos_7)[name = string("op_2025")]; tensor var_2026 = mul(x = x2_29, y = sin_7)[name = string("op_2026")]; tensor var_2027 = sub(x = var_2025, y = var_2026)[name = string("op_2027")]; tensor var_2028 = mul(x = x2_29, y = cos_7)[name = string("op_2028")]; tensor var_2029 = mul(x = x1_29, y = sin_7)[name = string("op_2029")]; tensor var_2030 = add(x = var_2028, y = var_2029)[name = string("op_2030")]; bool rotated_29_interleave_0 = const()[name = string("rotated_29_interleave_0"), val = bool(false)]; tensor rotated_29 = concat(axis = var_73, interleave = rotated_29_interleave_0, values = (var_2027, var_2030))[name = string("rotated_29")]; tensor x1_31_begin_0 = const()[name = string("x1_31_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_31_end_0 = const()[name = string("x1_31_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_31_end_mask_0 = const()[name = string("x1_31_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_201 = transpose(perm = var_2003, x = var_2002)[name = string("transpose_58")]; tensor x1_31 = slice_by_index(begin = x1_31_begin_0, end = x1_31_end_0, end_mask = x1_31_end_mask_0, x = x_201)[name = string("x1_31")]; tensor x2_31_begin_0 = const()[name = string("x2_31_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_31_end_0 = const()[name = string("x2_31_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_31_end_mask_0 = const()[name = string("x2_31_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_31 = slice_by_index(begin = x2_31_begin_0, end = x2_31_end_0, end_mask = x2_31_end_mask_0, x = x_201)[name = string("x2_31")]; tensor var_2046 = mul(x = x1_31, y = cos_7)[name = string("op_2046")]; tensor var_2047 = mul(x = x2_31, y = sin_7)[name = string("op_2047")]; tensor var_2048 = sub(x = var_2046, y = var_2047)[name = string("op_2048")]; tensor var_2049 = mul(x = x2_31, y = cos_7)[name = string("op_2049")]; tensor var_2050 = mul(x = x1_31, y = sin_7)[name = string("op_2050")]; tensor var_2051 = add(x = var_2049, y = var_2050)[name = string("op_2051")]; bool rotated_31_interleave_0 = const()[name = string("rotated_31_interleave_0"), val = bool(false)]; tensor rotated_31 = concat(axis = var_73, interleave = rotated_31_interleave_0, values = (var_2048, var_2051))[name = string("rotated_31")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; int32 concat_128_axis_0 = const()[name = string("concat_128_axis_0"), val = int32(0)]; bool concat_128_interleave_0 = const()[name = string("concat_128_interleave_0"), val = bool(false)]; tensor concat_128 = concat(axis = concat_128_axis_0, interleave = concat_128_interleave_0, values = (expand_dims_84, expand_dims_85, current_pos, expand_dims_87))[name = string("concat_128")]; tensor concat_129_values1_0 = const()[name = string("concat_129_values1_0"), val = tensor([0])]; tensor concat_129_values3_0 = const()[name = string("concat_129_values3_0"), val = tensor([0])]; int32 concat_129_axis_0 = const()[name = string("concat_129_axis_0"), val = int32(0)]; bool concat_129_interleave_0 = const()[name = string("concat_129_interleave_0"), val = bool(false)]; tensor concat_129 = concat(axis = concat_129_axis_0, interleave = concat_129_interleave_0, values = (expand_dims_88, concat_129_values1_0, var_597, concat_129_values3_0))[name = string("concat_129")]; tensor model_model_kv_cache_0_internal_tensor_assign_15_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_128, begin_mask = model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0, end = concat_129, end_mask = model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_15_stride_0, update = rotated_31, x = coreml_update_state_45)[name = string("model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_110_write_state")]; tensor coreml_update_state_46 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_110")]; tensor expand_dims_90 = const()[name = string("expand_dims_90"), val = tensor([23])]; tensor expand_dims_91 = const()[name = string("expand_dims_91"), val = tensor([0])]; tensor expand_dims_93 = const()[name = string("expand_dims_93"), val = tensor([0])]; tensor expand_dims_94 = const()[name = string("expand_dims_94"), val = tensor([24])]; int32 concat_132_axis_0 = const()[name = string("concat_132_axis_0"), val = int32(0)]; bool concat_132_interleave_0 = const()[name = string("concat_132_interleave_0"), val = bool(false)]; tensor concat_132 = concat(axis = concat_132_axis_0, interleave = concat_132_interleave_0, values = (expand_dims_90, expand_dims_91, current_pos, expand_dims_93))[name = string("concat_132")]; tensor concat_133_values1_0 = const()[name = string("concat_133_values1_0"), val = tensor([0])]; tensor concat_133_values3_0 = const()[name = string("concat_133_values3_0"), val = tensor([0])]; int32 concat_133_axis_0 = const()[name = string("concat_133_axis_0"), val = int32(0)]; bool concat_133_interleave_0 = const()[name = string("concat_133_interleave_0"), val = bool(false)]; tensor concat_133 = concat(axis = concat_133_axis_0, interleave = concat_133_interleave_0, values = (expand_dims_94, concat_133_values1_0, var_597, concat_133_values3_0))[name = string("concat_133")]; tensor model_model_kv_cache_0_internal_tensor_assign_16_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_45 = transpose(perm = var_2007, x = var_2006)[name = string("transpose_57")]; tensor model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_132, begin_mask = model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0, end = concat_133, end_mask = model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_16_stride_0, update = value_states_45, x = coreml_update_state_46)[name = string("model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_111_write_state")]; tensor coreml_update_state_47 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_111")]; tensor var_2074_begin_0 = const()[name = string("op_2074_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_2074_end_0 = const()[name = string("op_2074_end_0"), val = tensor([8, 8, 4096, 64])]; tensor var_2074_end_mask_0 = const()[name = string("op_2074_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2074_cast_fp16 = slice_by_index(begin = var_2074_begin_0, end = var_2074_end_0, end_mask = var_2074_end_mask_0, x = coreml_update_state_47)[name = string("op_2074_cast_fp16")]; tensor K_layer_cache_15_axes_0 = const()[name = string("K_layer_cache_15_axes_0"), val = tensor([0])]; tensor K_layer_cache_15_cast_fp16 = squeeze(axes = K_layer_cache_15_axes_0, x = var_2074_cast_fp16)[name = string("K_layer_cache_15_cast_fp16")]; tensor var_2076_begin_0 = const()[name = string("op_2076_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_2076_end_0 = const()[name = string("op_2076_end_0"), val = tensor([24, 8, 4096, 64])]; tensor var_2076_end_mask_0 = const()[name = string("op_2076_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2076_cast_fp16 = slice_by_index(begin = var_2076_begin_0, end = var_2076_end_0, end_mask = var_2076_end_mask_0, x = coreml_update_state_47)[name = string("op_2076_cast_fp16")]; tensor V_layer_cache_15_axes_0 = const()[name = string("V_layer_cache_15_axes_0"), val = tensor([0])]; tensor V_layer_cache_15_cast_fp16 = squeeze(axes = V_layer_cache_15_axes_0, x = var_2076_cast_fp16)[name = string("V_layer_cache_15_cast_fp16")]; tensor x_207_axes_0 = const()[name = string("x_207_axes_0"), val = tensor([1])]; tensor x_207_cast_fp16 = expand_dims(axes = x_207_axes_0, x = K_layer_cache_15_cast_fp16)[name = string("x_207_cast_fp16")]; tensor var_2085 = const()[name = string("op_2085"), val = tensor([1, 4, 1, 1])]; tensor x_209_cast_fp16 = tile(reps = var_2085, x = x_207_cast_fp16)[name = string("x_209_cast_fp16")]; tensor var_2089 = const()[name = string("op_2089"), val = tensor([1, -1, 4096, 64])]; tensor var_2090_cast_fp16 = reshape(shape = var_2089, x = x_209_cast_fp16)[name = string("op_2090_cast_fp16")]; tensor x_213_axes_0 = const()[name = string("x_213_axes_0"), val = tensor([1])]; tensor x_213_cast_fp16 = expand_dims(axes = x_213_axes_0, x = V_layer_cache_15_cast_fp16)[name = string("x_213_cast_fp16")]; tensor var_2092 = const()[name = string("op_2092"), val = tensor([1, 4, 1, 1])]; tensor x_215_cast_fp16 = tile(reps = var_2092, x = x_213_cast_fp16)[name = string("x_215_cast_fp16")]; bool var_2099_transpose_x_0 = const()[name = string("op_2099_transpose_x_0"), val = bool(false)]; bool var_2099_transpose_y_0 = const()[name = string("op_2099_transpose_y_0"), val = bool(true)]; tensor var_2099_cast_fp16 = matmul(transpose_x = var_2099_transpose_x_0, transpose_y = var_2099_transpose_y_0, x = rotated_29, y = var_2090_cast_fp16)[name = string("op_2099_cast_fp16")]; fp16 var_2100_to_fp16 = const()[name = string("op_2100_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_15_cast_fp16 = mul(x = var_2099_cast_fp16, y = var_2100_to_fp16)[name = string("attn_weights_15_cast_fp16")]; tensor x_217_cast_fp16 = add(x = attn_weights_15_cast_fp16, y = causal_mask)[name = string("x_217_cast_fp16")]; tensor reduce_max_7_axes_0 = const()[name = string("reduce_max_7_axes_0"), val = tensor([-1])]; bool reduce_max_7_keep_dims_0 = const()[name = string("reduce_max_7_keep_dims_0"), val = bool(true)]; tensor reduce_max_7_cast_fp16 = reduce_max(axes = reduce_max_7_axes_0, keep_dims = reduce_max_7_keep_dims_0, x = x_217_cast_fp16)[name = string("reduce_max_7_cast_fp16")]; tensor x_219_cast_fp16 = sub(x = x_217_cast_fp16, y = reduce_max_7_cast_fp16)[name = string("x_219_cast_fp16")]; tensor exp_x_15_cast_fp16 = exp(x = x_219_cast_fp16)[name = string("exp_x_15_cast_fp16")]; tensor var_2111_axes_0 = const()[name = string("op_2111_axes_0"), val = tensor([-1])]; bool var_2111_keep_dims_0 = const()[name = string("op_2111_keep_dims_0"), val = bool(true)]; tensor var_2111_cast_fp16 = reduce_sum(axes = var_2111_axes_0, keep_dims = var_2111_keep_dims_0, x = exp_x_15_cast_fp16)[name = string("op_2111_cast_fp16")]; tensor var_2112_cast_fp16 = real_div(x = exp_x_15_cast_fp16, y = var_2111_cast_fp16)[name = string("op_2112_cast_fp16")]; tensor concat_138 = const()[name = string("concat_138"), val = tensor([32, 64, 4096])]; tensor reshape_21_cast_fp16 = reshape(shape = concat_138, x = var_2112_cast_fp16)[name = string("reshape_21_cast_fp16")]; tensor concat_139 = const()[name = string("concat_139"), val = tensor([32, 4096, 64])]; tensor reshape_22_cast_fp16 = reshape(shape = concat_139, x = x_215_cast_fp16)[name = string("reshape_22_cast_fp16")]; bool matmul_7_transpose_x_0 = const()[name = string("matmul_7_transpose_x_0"), val = bool(false)]; bool matmul_7_transpose_y_0 = const()[name = string("matmul_7_transpose_y_0"), val = bool(false)]; tensor matmul_7_cast_fp16 = matmul(transpose_x = matmul_7_transpose_x_0, transpose_y = matmul_7_transpose_y_0, x = reshape_21_cast_fp16, y = reshape_22_cast_fp16)[name = string("matmul_7_cast_fp16")]; tensor concat_143 = const()[name = string("concat_143"), val = tensor([1, 32, 64, 64])]; tensor reshape_23_cast_fp16 = reshape(shape = concat_143, x = matmul_7_cast_fp16)[name = string("reshape_23_cast_fp16")]; tensor var_2115_perm_0 = const()[name = string("op_2115_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2117 = const()[name = string("op_2117"), val = tensor([1, 64, 2048])]; tensor var_2115_cast_fp16 = transpose(perm = var_2115_perm_0, x = reshape_23_cast_fp16)[name = string("transpose_56")]; tensor input_103_cast_fp16 = reshape(shape = var_2117, x = var_2115_cast_fp16)[name = string("input_103_cast_fp16")]; tensor model_model_layers_7_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(471275584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473372800))))[name = string("model_model_layers_7_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_7_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_7_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_103_cast_fp16)[name = string("linear_7_cast_fp16")]; tensor hidden_states_61_cast_fp16 = add(x = hidden_states_57_cast_fp16, y = linear_7_cast_fp16)[name = string("hidden_states_61_cast_fp16")]; fp16 const_141_promoted_to_fp16 = const()[name = string("const_141_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2123_cast_fp16 = mul(x = hidden_states_61_cast_fp16, y = const_141_promoted_to_fp16)[name = string("op_2123_cast_fp16")]; bool input_105_interleave_0 = const()[name = string("input_105_interleave_0"), val = bool(false)]; tensor input_105_cast_fp16 = concat(axis = var_73, interleave = input_105_interleave_0, values = (hidden_states_61_cast_fp16, var_2123_cast_fp16))[name = string("input_105_cast_fp16")]; tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_76_to_fp16, x = input_105_cast_fp16)[name = string("normed_61_cast_fp16")]; tensor normed_63_begin_0 = const()[name = string("normed_63_begin_0"), val = tensor([0, 0, 0])]; tensor normed_63_end_0 = const()[name = string("normed_63_end_0"), val = tensor([1, 64, 2048])]; tensor normed_63_end_mask_0 = const()[name = string("normed_63_end_mask_0"), val = tensor([true, true, false])]; tensor normed_63_cast_fp16 = slice_by_index(begin = normed_63_begin_0, end = normed_63_end_0, end_mask = normed_63_end_mask_0, x = normed_61_cast_fp16)[name = string("normed_63_cast_fp16")]; tensor const_144_promoted_to_fp16 = const()[name = string("const_144_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473381056)))]; tensor x_221_cast_fp16 = mul(x = normed_63_cast_fp16, y = const_144_promoted_to_fp16)[name = string("x_221_cast_fp16")]; tensor var_2141 = const()[name = string("op_2141"), val = tensor([0, 2, 1])]; tensor input_107_axes_0 = const()[name = string("input_107_axes_0"), val = tensor([2])]; tensor var_2142 = transpose(perm = var_2141, x = x_221_cast_fp16)[name = string("transpose_55")]; tensor input_107 = expand_dims(axes = input_107_axes_0, x = var_2142)[name = string("input_107")]; string input_109_pad_type_0 = const()[name = string("input_109_pad_type_0"), val = string("valid")]; tensor input_109_strides_0 = const()[name = string("input_109_strides_0"), val = tensor([1, 1])]; tensor input_109_pad_0 = const()[name = string("input_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_109_dilations_0 = const()[name = string("input_109_dilations_0"), val = tensor([1, 1])]; int32 input_109_groups_0 = const()[name = string("input_109_groups_0"), val = int32(1)]; tensor input_109 = conv(dilations = input_109_dilations_0, groups = input_109_groups_0, pad = input_109_pad_0, pad_type = input_109_pad_type_0, strides = input_109_strides_0, weight = model_model_layers_7_mlp_gate_proj_weight_palettized, x = input_107)[name = string("input_109")]; string up_states_15_pad_type_0 = const()[name = string("up_states_15_pad_type_0"), val = string("valid")]; tensor up_states_15_strides_0 = const()[name = string("up_states_15_strides_0"), val = tensor([1, 1])]; tensor up_states_15_pad_0 = const()[name = string("up_states_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_15_dilations_0 = const()[name = string("up_states_15_dilations_0"), val = tensor([1, 1])]; int32 up_states_15_groups_0 = const()[name = string("up_states_15_groups_0"), val = int32(1)]; tensor up_states_15 = conv(dilations = up_states_15_dilations_0, groups = up_states_15_groups_0, pad = up_states_15_pad_0, pad_type = up_states_15_pad_type_0, strides = up_states_15_strides_0, weight = model_model_layers_7_mlp_up_proj_weight_palettized, x = input_107)[name = string("up_states_15")]; tensor gate_states_15 = silu(x = input_109)[name = string("gate_states_15")]; tensor input_111 = mul(x = gate_states_15, y = up_states_15)[name = string("input_111")]; string hidden_states_63_pad_type_0 = const()[name = string("hidden_states_63_pad_type_0"), val = string("valid")]; tensor hidden_states_63_strides_0 = const()[name = string("hidden_states_63_strides_0"), val = tensor([1, 1])]; tensor hidden_states_63_pad_0 = const()[name = string("hidden_states_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_63_dilations_0 = const()[name = string("hidden_states_63_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_63_groups_0 = const()[name = string("hidden_states_63_groups_0"), val = int32(1)]; tensor hidden_states_63 = conv(dilations = hidden_states_63_dilations_0, groups = hidden_states_63_groups_0, pad = hidden_states_63_pad_0, pad_type = hidden_states_63_pad_type_0, strides = hidden_states_63_strides_0, weight = model_model_layers_7_mlp_down_proj_weight_palettized, x = input_111)[name = string("hidden_states_63")]; tensor var_2164_axes_0 = const()[name = string("op_2164_axes_0"), val = tensor([2])]; tensor var_2164 = squeeze(axes = var_2164_axes_0, x = hidden_states_63)[name = string("op_2164")]; tensor var_2165 = const()[name = string("op_2165"), val = tensor([0, 2, 1])]; tensor var_2166 = transpose(perm = var_2165, x = var_2164)[name = string("transpose_54")]; tensor hidden_states_65_cast_fp16 = add(x = hidden_states_61_cast_fp16, y = var_2166)[name = string("hidden_states_65_cast_fp16")]; fp16 const_145_promoted_to_fp16 = const()[name = string("const_145_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2169_cast_fp16 = mul(x = hidden_states_65_cast_fp16, y = const_145_promoted_to_fp16)[name = string("op_2169_cast_fp16")]; bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; tensor input_113_cast_fp16 = concat(axis = var_73, interleave = input_113_interleave_0, values = (hidden_states_65_cast_fp16, var_2169_cast_fp16))[name = string("input_113_cast_fp16")]; tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_76_to_fp16, x = input_113_cast_fp16)[name = string("normed_65_cast_fp16")]; tensor normed_67_begin_0 = const()[name = string("normed_67_begin_0"), val = tensor([0, 0, 0])]; tensor normed_67_end_0 = const()[name = string("normed_67_end_0"), val = tensor([1, 64, 2048])]; tensor normed_67_end_mask_0 = const()[name = string("normed_67_end_mask_0"), val = tensor([true, true, false])]; tensor normed_67_cast_fp16 = slice_by_index(begin = normed_67_begin_0, end = normed_67_end_0, end_mask = normed_67_end_mask_0, x = normed_65_cast_fp16)[name = string("normed_67_cast_fp16")]; tensor const_148_promoted_to_fp16 = const()[name = string("const_148_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473385216)))]; tensor hidden_states_67_cast_fp16 = mul(x = normed_67_cast_fp16, y = const_148_promoted_to_fp16)[name = string("hidden_states_67_cast_fp16")]; tensor var_2184 = const()[name = string("op_2184"), val = tensor([0, 2, 1])]; tensor var_2186_axes_0 = const()[name = string("op_2186_axes_0"), val = tensor([2])]; tensor var_2185_cast_fp16 = transpose(perm = var_2184, x = hidden_states_67_cast_fp16)[name = string("transpose_53")]; tensor var_2186_cast_fp16 = expand_dims(axes = var_2186_axes_0, x = var_2185_cast_fp16)[name = string("op_2186_cast_fp16")]; string query_states_33_pad_type_0 = const()[name = string("query_states_33_pad_type_0"), val = string("valid")]; tensor query_states_33_strides_0 = const()[name = string("query_states_33_strides_0"), val = tensor([1, 1])]; tensor query_states_33_pad_0 = const()[name = string("query_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_33_dilations_0 = const()[name = string("query_states_33_dilations_0"), val = tensor([1, 1])]; int32 query_states_33_groups_0 = const()[name = string("query_states_33_groups_0"), val = int32(1)]; tensor query_states_33 = conv(dilations = query_states_33_dilations_0, groups = query_states_33_groups_0, pad = query_states_33_pad_0, pad_type = query_states_33_pad_type_0, strides = query_states_33_strides_0, weight = model_model_layers_8_self_attn_q_proj_weight_palettized, x = var_2186_cast_fp16)[name = string("query_states_33")]; string key_states_49_pad_type_0 = const()[name = string("key_states_49_pad_type_0"), val = string("valid")]; tensor key_states_49_strides_0 = const()[name = string("key_states_49_strides_0"), val = tensor([1, 1])]; tensor key_states_49_pad_0 = const()[name = string("key_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_49_dilations_0 = const()[name = string("key_states_49_dilations_0"), val = tensor([1, 1])]; int32 key_states_49_groups_0 = const()[name = string("key_states_49_groups_0"), val = int32(1)]; tensor key_states_49 = conv(dilations = key_states_49_dilations_0, groups = key_states_49_groups_0, pad = key_states_49_pad_0, pad_type = key_states_49_pad_type_0, strides = key_states_49_strides_0, weight = model_model_layers_8_self_attn_k_proj_weight_palettized, x = var_2186_cast_fp16)[name = string("key_states_49")]; string value_states_49_pad_type_0 = const()[name = string("value_states_49_pad_type_0"), val = string("valid")]; tensor value_states_49_strides_0 = const()[name = string("value_states_49_strides_0"), val = tensor([1, 1])]; tensor value_states_49_pad_0 = const()[name = string("value_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_49_dilations_0 = const()[name = string("value_states_49_dilations_0"), val = tensor([1, 1])]; int32 value_states_49_groups_0 = const()[name = string("value_states_49_groups_0"), val = int32(1)]; tensor value_states_49 = conv(dilations = value_states_49_dilations_0, groups = value_states_49_groups_0, pad = value_states_49_pad_0, pad_type = value_states_49_pad_type_0, strides = value_states_49_strides_0, weight = model_model_layers_8_self_attn_v_proj_weight_palettized, x = var_2186_cast_fp16)[name = string("value_states_49")]; tensor var_2206 = const()[name = string("op_2206"), val = tensor([1, 32, 64, 64])]; tensor var_2207 = reshape(shape = var_2206, x = query_states_33)[name = string("op_2207")]; tensor var_2208 = const()[name = string("op_2208"), val = tensor([0, 1, 3, 2])]; tensor var_2210 = const()[name = string("op_2210"), val = tensor([1, 8, 64, 64])]; tensor var_2211 = reshape(shape = var_2210, x = key_states_49)[name = string("op_2211")]; tensor var_2212 = const()[name = string("op_2212"), val = tensor([0, 1, 3, 2])]; tensor var_2214 = const()[name = string("op_2214"), val = tensor([1, 8, 64, 64])]; tensor var_2215 = reshape(shape = var_2214, x = value_states_49)[name = string("op_2215")]; tensor var_2216 = const()[name = string("op_2216"), val = tensor([0, 1, 3, 2])]; tensor x1_33_begin_0 = const()[name = string("x1_33_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_33_end_0 = const()[name = string("x1_33_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_33_end_mask_0 = const()[name = string("x1_33_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_225 = transpose(perm = var_2208, x = var_2207)[name = string("transpose_52")]; tensor x1_33 = slice_by_index(begin = x1_33_begin_0, end = x1_33_end_0, end_mask = x1_33_end_mask_0, x = x_225)[name = string("x1_33")]; tensor x2_33_begin_0 = const()[name = string("x2_33_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_33_end_0 = const()[name = string("x2_33_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_33_end_mask_0 = const()[name = string("x2_33_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_33 = slice_by_index(begin = x2_33_begin_0, end = x2_33_end_0, end_mask = x2_33_end_mask_0, x = x_225)[name = string("x2_33")]; tensor var_2234 = mul(x = x1_33, y = cos_7)[name = string("op_2234")]; tensor var_2235 = mul(x = x2_33, y = sin_7)[name = string("op_2235")]; tensor var_2236 = sub(x = var_2234, y = var_2235)[name = string("op_2236")]; tensor var_2237 = mul(x = x2_33, y = cos_7)[name = string("op_2237")]; tensor var_2238 = mul(x = x1_33, y = sin_7)[name = string("op_2238")]; tensor var_2239 = add(x = var_2237, y = var_2238)[name = string("op_2239")]; bool rotated_33_interleave_0 = const()[name = string("rotated_33_interleave_0"), val = bool(false)]; tensor rotated_33 = concat(axis = var_73, interleave = rotated_33_interleave_0, values = (var_2236, var_2239))[name = string("rotated_33")]; tensor x1_35_begin_0 = const()[name = string("x1_35_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_35_end_0 = const()[name = string("x1_35_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_35_end_mask_0 = const()[name = string("x1_35_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_229 = transpose(perm = var_2212, x = var_2211)[name = string("transpose_51")]; tensor x1_35 = slice_by_index(begin = x1_35_begin_0, end = x1_35_end_0, end_mask = x1_35_end_mask_0, x = x_229)[name = string("x1_35")]; tensor x2_35_begin_0 = const()[name = string("x2_35_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_35_end_0 = const()[name = string("x2_35_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_35_end_mask_0 = const()[name = string("x2_35_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_35 = slice_by_index(begin = x2_35_begin_0, end = x2_35_end_0, end_mask = x2_35_end_mask_0, x = x_229)[name = string("x2_35")]; tensor var_2255 = mul(x = x1_35, y = cos_7)[name = string("op_2255")]; tensor var_2256 = mul(x = x2_35, y = sin_7)[name = string("op_2256")]; tensor var_2257 = sub(x = var_2255, y = var_2256)[name = string("op_2257")]; tensor var_2258 = mul(x = x2_35, y = cos_7)[name = string("op_2258")]; tensor var_2259 = mul(x = x1_35, y = sin_7)[name = string("op_2259")]; tensor var_2260 = add(x = var_2258, y = var_2259)[name = string("op_2260")]; bool rotated_35_interleave_0 = const()[name = string("rotated_35_interleave_0"), val = bool(false)]; tensor rotated_35 = concat(axis = var_73, interleave = rotated_35_interleave_0, values = (var_2257, var_2260))[name = string("rotated_35")]; tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; int32 concat_146_axis_0 = const()[name = string("concat_146_axis_0"), val = int32(0)]; bool concat_146_interleave_0 = const()[name = string("concat_146_interleave_0"), val = bool(false)]; tensor concat_146 = concat(axis = concat_146_axis_0, interleave = concat_146_interleave_0, values = (expand_dims_96, expand_dims_97, current_pos, expand_dims_99))[name = string("concat_146")]; tensor concat_147_values1_0 = const()[name = string("concat_147_values1_0"), val = tensor([0])]; tensor concat_147_values3_0 = const()[name = string("concat_147_values3_0"), val = tensor([0])]; int32 concat_147_axis_0 = const()[name = string("concat_147_axis_0"), val = int32(0)]; bool concat_147_interleave_0 = const()[name = string("concat_147_interleave_0"), val = bool(false)]; tensor concat_147 = concat(axis = concat_147_axis_0, interleave = concat_147_interleave_0, values = (expand_dims_100, concat_147_values1_0, var_597, concat_147_values3_0))[name = string("concat_147")]; tensor model_model_kv_cache_0_internal_tensor_assign_17_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_146, begin_mask = model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0, end = concat_147, end_mask = model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_17_stride_0, update = rotated_35, x = coreml_update_state_47)[name = string("model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_112_write_state")]; tensor coreml_update_state_48 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_112")]; tensor expand_dims_102 = const()[name = string("expand_dims_102"), val = tensor([24])]; tensor expand_dims_103 = const()[name = string("expand_dims_103"), val = tensor([0])]; tensor expand_dims_105 = const()[name = string("expand_dims_105"), val = tensor([0])]; tensor expand_dims_106 = const()[name = string("expand_dims_106"), val = tensor([25])]; int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_102, expand_dims_103, current_pos, expand_dims_105))[name = string("concat_150")]; tensor concat_151_values1_0 = const()[name = string("concat_151_values1_0"), val = tensor([0])]; tensor concat_151_values3_0 = const()[name = string("concat_151_values3_0"), val = tensor([0])]; int32 concat_151_axis_0 = const()[name = string("concat_151_axis_0"), val = int32(0)]; bool concat_151_interleave_0 = const()[name = string("concat_151_interleave_0"), val = bool(false)]; tensor concat_151 = concat(axis = concat_151_axis_0, interleave = concat_151_interleave_0, values = (expand_dims_106, concat_151_values1_0, var_597, concat_151_values3_0))[name = string("concat_151")]; tensor model_model_kv_cache_0_internal_tensor_assign_18_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_51 = transpose(perm = var_2216, x = var_2215)[name = string("transpose_50")]; tensor model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_150, begin_mask = model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0, end = concat_151, end_mask = model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_18_stride_0, update = value_states_51, x = coreml_update_state_48)[name = string("model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_113_write_state")]; tensor coreml_update_state_49 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_113")]; tensor var_2283_begin_0 = const()[name = string("op_2283_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_2283_end_0 = const()[name = string("op_2283_end_0"), val = tensor([9, 8, 4096, 64])]; tensor var_2283_end_mask_0 = const()[name = string("op_2283_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2283_cast_fp16 = slice_by_index(begin = var_2283_begin_0, end = var_2283_end_0, end_mask = var_2283_end_mask_0, x = coreml_update_state_49)[name = string("op_2283_cast_fp16")]; tensor K_layer_cache_17_axes_0 = const()[name = string("K_layer_cache_17_axes_0"), val = tensor([0])]; tensor K_layer_cache_17_cast_fp16 = squeeze(axes = K_layer_cache_17_axes_0, x = var_2283_cast_fp16)[name = string("K_layer_cache_17_cast_fp16")]; tensor var_2285_begin_0 = const()[name = string("op_2285_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_2285_end_0 = const()[name = string("op_2285_end_0"), val = tensor([25, 8, 4096, 64])]; tensor var_2285_end_mask_0 = const()[name = string("op_2285_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2285_cast_fp16 = slice_by_index(begin = var_2285_begin_0, end = var_2285_end_0, end_mask = var_2285_end_mask_0, x = coreml_update_state_49)[name = string("op_2285_cast_fp16")]; tensor V_layer_cache_17_axes_0 = const()[name = string("V_layer_cache_17_axes_0"), val = tensor([0])]; tensor V_layer_cache_17_cast_fp16 = squeeze(axes = V_layer_cache_17_axes_0, x = var_2285_cast_fp16)[name = string("V_layer_cache_17_cast_fp16")]; tensor x_235_axes_0 = const()[name = string("x_235_axes_0"), val = tensor([1])]; tensor x_235_cast_fp16 = expand_dims(axes = x_235_axes_0, x = K_layer_cache_17_cast_fp16)[name = string("x_235_cast_fp16")]; tensor var_2294 = const()[name = string("op_2294"), val = tensor([1, 4, 1, 1])]; tensor x_237_cast_fp16 = tile(reps = var_2294, x = x_235_cast_fp16)[name = string("x_237_cast_fp16")]; tensor var_2298 = const()[name = string("op_2298"), val = tensor([1, -1, 4096, 64])]; tensor var_2299_cast_fp16 = reshape(shape = var_2298, x = x_237_cast_fp16)[name = string("op_2299_cast_fp16")]; tensor x_241_axes_0 = const()[name = string("x_241_axes_0"), val = tensor([1])]; tensor x_241_cast_fp16 = expand_dims(axes = x_241_axes_0, x = V_layer_cache_17_cast_fp16)[name = string("x_241_cast_fp16")]; tensor var_2301 = const()[name = string("op_2301"), val = tensor([1, 4, 1, 1])]; tensor x_243_cast_fp16 = tile(reps = var_2301, x = x_241_cast_fp16)[name = string("x_243_cast_fp16")]; bool var_2308_transpose_x_0 = const()[name = string("op_2308_transpose_x_0"), val = bool(false)]; bool var_2308_transpose_y_0 = const()[name = string("op_2308_transpose_y_0"), val = bool(true)]; tensor var_2308_cast_fp16 = matmul(transpose_x = var_2308_transpose_x_0, transpose_y = var_2308_transpose_y_0, x = rotated_33, y = var_2299_cast_fp16)[name = string("op_2308_cast_fp16")]; fp16 var_2309_to_fp16 = const()[name = string("op_2309_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_17_cast_fp16 = mul(x = var_2308_cast_fp16, y = var_2309_to_fp16)[name = string("attn_weights_17_cast_fp16")]; tensor x_245_cast_fp16 = add(x = attn_weights_17_cast_fp16, y = causal_mask)[name = string("x_245_cast_fp16")]; tensor reduce_max_8_axes_0 = const()[name = string("reduce_max_8_axes_0"), val = tensor([-1])]; bool reduce_max_8_keep_dims_0 = const()[name = string("reduce_max_8_keep_dims_0"), val = bool(true)]; tensor reduce_max_8_cast_fp16 = reduce_max(axes = reduce_max_8_axes_0, keep_dims = reduce_max_8_keep_dims_0, x = x_245_cast_fp16)[name = string("reduce_max_8_cast_fp16")]; tensor x_247_cast_fp16 = sub(x = x_245_cast_fp16, y = reduce_max_8_cast_fp16)[name = string("x_247_cast_fp16")]; tensor exp_x_17_cast_fp16 = exp(x = x_247_cast_fp16)[name = string("exp_x_17_cast_fp16")]; tensor var_2320_axes_0 = const()[name = string("op_2320_axes_0"), val = tensor([-1])]; bool var_2320_keep_dims_0 = const()[name = string("op_2320_keep_dims_0"), val = bool(true)]; tensor var_2320_cast_fp16 = reduce_sum(axes = var_2320_axes_0, keep_dims = var_2320_keep_dims_0, x = exp_x_17_cast_fp16)[name = string("op_2320_cast_fp16")]; tensor var_2321_cast_fp16 = real_div(x = exp_x_17_cast_fp16, y = var_2320_cast_fp16)[name = string("op_2321_cast_fp16")]; tensor concat_156 = const()[name = string("concat_156"), val = tensor([32, 64, 4096])]; tensor reshape_24_cast_fp16 = reshape(shape = concat_156, x = var_2321_cast_fp16)[name = string("reshape_24_cast_fp16")]; tensor concat_157 = const()[name = string("concat_157"), val = tensor([32, 4096, 64])]; tensor reshape_25_cast_fp16 = reshape(shape = concat_157, x = x_243_cast_fp16)[name = string("reshape_25_cast_fp16")]; bool matmul_8_transpose_x_0 = const()[name = string("matmul_8_transpose_x_0"), val = bool(false)]; bool matmul_8_transpose_y_0 = const()[name = string("matmul_8_transpose_y_0"), val = bool(false)]; tensor matmul_8_cast_fp16 = matmul(transpose_x = matmul_8_transpose_x_0, transpose_y = matmul_8_transpose_y_0, x = reshape_24_cast_fp16, y = reshape_25_cast_fp16)[name = string("matmul_8_cast_fp16")]; tensor concat_161 = const()[name = string("concat_161"), val = tensor([1, 32, 64, 64])]; tensor reshape_26_cast_fp16 = reshape(shape = concat_161, x = matmul_8_cast_fp16)[name = string("reshape_26_cast_fp16")]; tensor var_2324_perm_0 = const()[name = string("op_2324_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2326 = const()[name = string("op_2326"), val = tensor([1, 64, 2048])]; tensor var_2324_cast_fp16 = transpose(perm = var_2324_perm_0, x = reshape_26_cast_fp16)[name = string("transpose_49")]; tensor input_117_cast_fp16 = reshape(shape = var_2326, x = var_2324_cast_fp16)[name = string("input_117_cast_fp16")]; tensor model_model_layers_8_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473389376))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(475486592))))[name = string("model_model_layers_8_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_8_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_8_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_117_cast_fp16)[name = string("linear_8_cast_fp16")]; tensor hidden_states_69_cast_fp16 = add(x = hidden_states_65_cast_fp16, y = linear_8_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; fp16 const_159_promoted_to_fp16 = const()[name = string("const_159_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2332_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_159_promoted_to_fp16)[name = string("op_2332_cast_fp16")]; bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; tensor input_119_cast_fp16 = concat(axis = var_73, interleave = input_119_interleave_0, values = (hidden_states_69_cast_fp16, var_2332_cast_fp16))[name = string("input_119_cast_fp16")]; tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_76_to_fp16, x = input_119_cast_fp16)[name = string("normed_69_cast_fp16")]; tensor normed_71_begin_0 = const()[name = string("normed_71_begin_0"), val = tensor([0, 0, 0])]; tensor normed_71_end_0 = const()[name = string("normed_71_end_0"), val = tensor([1, 64, 2048])]; tensor normed_71_end_mask_0 = const()[name = string("normed_71_end_mask_0"), val = tensor([true, true, false])]; tensor normed_71_cast_fp16 = slice_by_index(begin = normed_71_begin_0, end = normed_71_end_0, end_mask = normed_71_end_mask_0, x = normed_69_cast_fp16)[name = string("normed_71_cast_fp16")]; tensor const_162_promoted_to_fp16 = const()[name = string("const_162_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(475494848)))]; tensor x_249_cast_fp16 = mul(x = normed_71_cast_fp16, y = const_162_promoted_to_fp16)[name = string("x_249_cast_fp16")]; tensor var_2350 = const()[name = string("op_2350"), val = tensor([0, 2, 1])]; tensor input_121_axes_0 = const()[name = string("input_121_axes_0"), val = tensor([2])]; tensor var_2351 = transpose(perm = var_2350, x = x_249_cast_fp16)[name = string("transpose_48")]; tensor input_121 = expand_dims(axes = input_121_axes_0, x = var_2351)[name = string("input_121")]; string input_123_pad_type_0 = const()[name = string("input_123_pad_type_0"), val = string("valid")]; tensor input_123_strides_0 = const()[name = string("input_123_strides_0"), val = tensor([1, 1])]; tensor input_123_pad_0 = const()[name = string("input_123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_123_dilations_0 = const()[name = string("input_123_dilations_0"), val = tensor([1, 1])]; int32 input_123_groups_0 = const()[name = string("input_123_groups_0"), val = int32(1)]; tensor input_123 = conv(dilations = input_123_dilations_0, groups = input_123_groups_0, pad = input_123_pad_0, pad_type = input_123_pad_type_0, strides = input_123_strides_0, weight = model_model_layers_8_mlp_gate_proj_weight_palettized, x = input_121)[name = string("input_123")]; string up_states_17_pad_type_0 = const()[name = string("up_states_17_pad_type_0"), val = string("valid")]; tensor up_states_17_strides_0 = const()[name = string("up_states_17_strides_0"), val = tensor([1, 1])]; tensor up_states_17_pad_0 = const()[name = string("up_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_17_dilations_0 = const()[name = string("up_states_17_dilations_0"), val = tensor([1, 1])]; int32 up_states_17_groups_0 = const()[name = string("up_states_17_groups_0"), val = int32(1)]; tensor up_states_17 = conv(dilations = up_states_17_dilations_0, groups = up_states_17_groups_0, pad = up_states_17_pad_0, pad_type = up_states_17_pad_type_0, strides = up_states_17_strides_0, weight = model_model_layers_8_mlp_up_proj_weight_palettized, x = input_121)[name = string("up_states_17")]; tensor gate_states_17 = silu(x = input_123)[name = string("gate_states_17")]; tensor input_125 = mul(x = gate_states_17, y = up_states_17)[name = string("input_125")]; string hidden_states_71_pad_type_0 = const()[name = string("hidden_states_71_pad_type_0"), val = string("valid")]; tensor hidden_states_71_strides_0 = const()[name = string("hidden_states_71_strides_0"), val = tensor([1, 1])]; tensor hidden_states_71_pad_0 = const()[name = string("hidden_states_71_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_71_dilations_0 = const()[name = string("hidden_states_71_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_71_groups_0 = const()[name = string("hidden_states_71_groups_0"), val = int32(1)]; tensor hidden_states_71 = conv(dilations = hidden_states_71_dilations_0, groups = hidden_states_71_groups_0, pad = hidden_states_71_pad_0, pad_type = hidden_states_71_pad_type_0, strides = hidden_states_71_strides_0, weight = model_model_layers_8_mlp_down_proj_weight_palettized, x = input_125)[name = string("hidden_states_71")]; tensor var_2373_axes_0 = const()[name = string("op_2373_axes_0"), val = tensor([2])]; tensor var_2373 = squeeze(axes = var_2373_axes_0, x = hidden_states_71)[name = string("op_2373")]; tensor var_2374 = const()[name = string("op_2374"), val = tensor([0, 2, 1])]; tensor var_2375 = transpose(perm = var_2374, x = var_2373)[name = string("transpose_47")]; tensor hidden_states_73_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = var_2375)[name = string("hidden_states_73_cast_fp16")]; fp16 const_163_promoted_to_fp16 = const()[name = string("const_163_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2378_cast_fp16 = mul(x = hidden_states_73_cast_fp16, y = const_163_promoted_to_fp16)[name = string("op_2378_cast_fp16")]; bool input_127_interleave_0 = const()[name = string("input_127_interleave_0"), val = bool(false)]; tensor input_127_cast_fp16 = concat(axis = var_73, interleave = input_127_interleave_0, values = (hidden_states_73_cast_fp16, var_2378_cast_fp16))[name = string("input_127_cast_fp16")]; tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_76_to_fp16, x = input_127_cast_fp16)[name = string("normed_73_cast_fp16")]; tensor normed_75_begin_0 = const()[name = string("normed_75_begin_0"), val = tensor([0, 0, 0])]; tensor normed_75_end_0 = const()[name = string("normed_75_end_0"), val = tensor([1, 64, 2048])]; tensor normed_75_end_mask_0 = const()[name = string("normed_75_end_mask_0"), val = tensor([true, true, false])]; tensor normed_75_cast_fp16 = slice_by_index(begin = normed_75_begin_0, end = normed_75_end_0, end_mask = normed_75_end_mask_0, x = normed_73_cast_fp16)[name = string("normed_75_cast_fp16")]; tensor const_166_promoted_to_fp16 = const()[name = string("const_166_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(475499008)))]; tensor hidden_states_75_cast_fp16 = mul(x = normed_75_cast_fp16, y = const_166_promoted_to_fp16)[name = string("hidden_states_75_cast_fp16")]; tensor var_2393 = const()[name = string("op_2393"), val = tensor([0, 2, 1])]; tensor var_2395_axes_0 = const()[name = string("op_2395_axes_0"), val = tensor([2])]; tensor var_2394_cast_fp16 = transpose(perm = var_2393, x = hidden_states_75_cast_fp16)[name = string("transpose_46")]; tensor var_2395_cast_fp16 = expand_dims(axes = var_2395_axes_0, x = var_2394_cast_fp16)[name = string("op_2395_cast_fp16")]; string query_states_37_pad_type_0 = const()[name = string("query_states_37_pad_type_0"), val = string("valid")]; tensor query_states_37_strides_0 = const()[name = string("query_states_37_strides_0"), val = tensor([1, 1])]; tensor query_states_37_pad_0 = const()[name = string("query_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_37_dilations_0 = const()[name = string("query_states_37_dilations_0"), val = tensor([1, 1])]; int32 query_states_37_groups_0 = const()[name = string("query_states_37_groups_0"), val = int32(1)]; tensor query_states_37 = conv(dilations = query_states_37_dilations_0, groups = query_states_37_groups_0, pad = query_states_37_pad_0, pad_type = query_states_37_pad_type_0, strides = query_states_37_strides_0, weight = model_model_layers_9_self_attn_q_proj_weight_palettized, x = var_2395_cast_fp16)[name = string("query_states_37")]; string key_states_55_pad_type_0 = const()[name = string("key_states_55_pad_type_0"), val = string("valid")]; tensor key_states_55_strides_0 = const()[name = string("key_states_55_strides_0"), val = tensor([1, 1])]; tensor key_states_55_pad_0 = const()[name = string("key_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_55_dilations_0 = const()[name = string("key_states_55_dilations_0"), val = tensor([1, 1])]; int32 key_states_55_groups_0 = const()[name = string("key_states_55_groups_0"), val = int32(1)]; tensor key_states_55 = conv(dilations = key_states_55_dilations_0, groups = key_states_55_groups_0, pad = key_states_55_pad_0, pad_type = key_states_55_pad_type_0, strides = key_states_55_strides_0, weight = model_model_layers_9_self_attn_k_proj_weight_palettized, x = var_2395_cast_fp16)[name = string("key_states_55")]; string value_states_55_pad_type_0 = const()[name = string("value_states_55_pad_type_0"), val = string("valid")]; tensor value_states_55_strides_0 = const()[name = string("value_states_55_strides_0"), val = tensor([1, 1])]; tensor value_states_55_pad_0 = const()[name = string("value_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_55_dilations_0 = const()[name = string("value_states_55_dilations_0"), val = tensor([1, 1])]; int32 value_states_55_groups_0 = const()[name = string("value_states_55_groups_0"), val = int32(1)]; tensor value_states_55 = conv(dilations = value_states_55_dilations_0, groups = value_states_55_groups_0, pad = value_states_55_pad_0, pad_type = value_states_55_pad_type_0, strides = value_states_55_strides_0, weight = model_model_layers_9_self_attn_v_proj_weight_palettized, x = var_2395_cast_fp16)[name = string("value_states_55")]; tensor var_2415 = const()[name = string("op_2415"), val = tensor([1, 32, 64, 64])]; tensor var_2416 = reshape(shape = var_2415, x = query_states_37)[name = string("op_2416")]; tensor var_2417 = const()[name = string("op_2417"), val = tensor([0, 1, 3, 2])]; tensor var_2419 = const()[name = string("op_2419"), val = tensor([1, 8, 64, 64])]; tensor var_2420 = reshape(shape = var_2419, x = key_states_55)[name = string("op_2420")]; tensor var_2421 = const()[name = string("op_2421"), val = tensor([0, 1, 3, 2])]; tensor var_2423 = const()[name = string("op_2423"), val = tensor([1, 8, 64, 64])]; tensor var_2424 = reshape(shape = var_2423, x = value_states_55)[name = string("op_2424")]; tensor var_2425 = const()[name = string("op_2425"), val = tensor([0, 1, 3, 2])]; tensor x1_37_begin_0 = const()[name = string("x1_37_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_37_end_0 = const()[name = string("x1_37_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_37_end_mask_0 = const()[name = string("x1_37_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_253 = transpose(perm = var_2417, x = var_2416)[name = string("transpose_45")]; tensor x1_37 = slice_by_index(begin = x1_37_begin_0, end = x1_37_end_0, end_mask = x1_37_end_mask_0, x = x_253)[name = string("x1_37")]; tensor x2_37_begin_0 = const()[name = string("x2_37_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_37_end_0 = const()[name = string("x2_37_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_37_end_mask_0 = const()[name = string("x2_37_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_37 = slice_by_index(begin = x2_37_begin_0, end = x2_37_end_0, end_mask = x2_37_end_mask_0, x = x_253)[name = string("x2_37")]; tensor var_2443 = mul(x = x1_37, y = cos_7)[name = string("op_2443")]; tensor var_2444 = mul(x = x2_37, y = sin_7)[name = string("op_2444")]; tensor var_2445 = sub(x = var_2443, y = var_2444)[name = string("op_2445")]; tensor var_2446 = mul(x = x2_37, y = cos_7)[name = string("op_2446")]; tensor var_2447 = mul(x = x1_37, y = sin_7)[name = string("op_2447")]; tensor var_2448 = add(x = var_2446, y = var_2447)[name = string("op_2448")]; bool rotated_37_interleave_0 = const()[name = string("rotated_37_interleave_0"), val = bool(false)]; tensor rotated_37 = concat(axis = var_73, interleave = rotated_37_interleave_0, values = (var_2445, var_2448))[name = string("rotated_37")]; tensor x1_39_begin_0 = const()[name = string("x1_39_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_39_end_0 = const()[name = string("x1_39_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_39_end_mask_0 = const()[name = string("x1_39_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_257 = transpose(perm = var_2421, x = var_2420)[name = string("transpose_44")]; tensor x1_39 = slice_by_index(begin = x1_39_begin_0, end = x1_39_end_0, end_mask = x1_39_end_mask_0, x = x_257)[name = string("x1_39")]; tensor x2_39_begin_0 = const()[name = string("x2_39_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_39_end_0 = const()[name = string("x2_39_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_39_end_mask_0 = const()[name = string("x2_39_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_39 = slice_by_index(begin = x2_39_begin_0, end = x2_39_end_0, end_mask = x2_39_end_mask_0, x = x_257)[name = string("x2_39")]; tensor var_2464 = mul(x = x1_39, y = cos_7)[name = string("op_2464")]; tensor var_2465 = mul(x = x2_39, y = sin_7)[name = string("op_2465")]; tensor var_2466 = sub(x = var_2464, y = var_2465)[name = string("op_2466")]; tensor var_2467 = mul(x = x2_39, y = cos_7)[name = string("op_2467")]; tensor var_2468 = mul(x = x1_39, y = sin_7)[name = string("op_2468")]; tensor var_2469 = add(x = var_2467, y = var_2468)[name = string("op_2469")]; bool rotated_39_interleave_0 = const()[name = string("rotated_39_interleave_0"), val = bool(false)]; tensor rotated_39 = concat(axis = var_73, interleave = rotated_39_interleave_0, values = (var_2466, var_2469))[name = string("rotated_39")]; tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; int32 concat_164_axis_0 = const()[name = string("concat_164_axis_0"), val = int32(0)]; bool concat_164_interleave_0 = const()[name = string("concat_164_interleave_0"), val = bool(false)]; tensor concat_164 = concat(axis = concat_164_axis_0, interleave = concat_164_interleave_0, values = (expand_dims_108, expand_dims_109, current_pos, expand_dims_111))[name = string("concat_164")]; tensor concat_165_values1_0 = const()[name = string("concat_165_values1_0"), val = tensor([0])]; tensor concat_165_values3_0 = const()[name = string("concat_165_values3_0"), val = tensor([0])]; int32 concat_165_axis_0 = const()[name = string("concat_165_axis_0"), val = int32(0)]; bool concat_165_interleave_0 = const()[name = string("concat_165_interleave_0"), val = bool(false)]; tensor concat_165 = concat(axis = concat_165_axis_0, interleave = concat_165_interleave_0, values = (expand_dims_112, concat_165_values1_0, var_597, concat_165_values3_0))[name = string("concat_165")]; tensor model_model_kv_cache_0_internal_tensor_assign_19_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_164, begin_mask = model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0, end = concat_165, end_mask = model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_19_stride_0, update = rotated_39, x = coreml_update_state_49)[name = string("model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_114_write_state")]; tensor coreml_update_state_50 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_114")]; tensor expand_dims_114 = const()[name = string("expand_dims_114"), val = tensor([25])]; tensor expand_dims_115 = const()[name = string("expand_dims_115"), val = tensor([0])]; tensor expand_dims_117 = const()[name = string("expand_dims_117"), val = tensor([0])]; tensor expand_dims_118 = const()[name = string("expand_dims_118"), val = tensor([26])]; int32 concat_168_axis_0 = const()[name = string("concat_168_axis_0"), val = int32(0)]; bool concat_168_interleave_0 = const()[name = string("concat_168_interleave_0"), val = bool(false)]; tensor concat_168 = concat(axis = concat_168_axis_0, interleave = concat_168_interleave_0, values = (expand_dims_114, expand_dims_115, current_pos, expand_dims_117))[name = string("concat_168")]; tensor concat_169_values1_0 = const()[name = string("concat_169_values1_0"), val = tensor([0])]; tensor concat_169_values3_0 = const()[name = string("concat_169_values3_0"), val = tensor([0])]; int32 concat_169_axis_0 = const()[name = string("concat_169_axis_0"), val = int32(0)]; bool concat_169_interleave_0 = const()[name = string("concat_169_interleave_0"), val = bool(false)]; tensor concat_169 = concat(axis = concat_169_axis_0, interleave = concat_169_interleave_0, values = (expand_dims_118, concat_169_values1_0, var_597, concat_169_values3_0))[name = string("concat_169")]; tensor model_model_kv_cache_0_internal_tensor_assign_20_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_57 = transpose(perm = var_2425, x = var_2424)[name = string("transpose_43")]; tensor model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_168, begin_mask = model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0, end = concat_169, end_mask = model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_20_stride_0, update = value_states_57, x = coreml_update_state_50)[name = string("model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_115_write_state")]; tensor coreml_update_state_51 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_115")]; tensor var_2492_begin_0 = const()[name = string("op_2492_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_2492_end_0 = const()[name = string("op_2492_end_0"), val = tensor([10, 8, 4096, 64])]; tensor var_2492_end_mask_0 = const()[name = string("op_2492_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2492_cast_fp16 = slice_by_index(begin = var_2492_begin_0, end = var_2492_end_0, end_mask = var_2492_end_mask_0, x = coreml_update_state_51)[name = string("op_2492_cast_fp16")]; tensor K_layer_cache_19_axes_0 = const()[name = string("K_layer_cache_19_axes_0"), val = tensor([0])]; tensor K_layer_cache_19_cast_fp16 = squeeze(axes = K_layer_cache_19_axes_0, x = var_2492_cast_fp16)[name = string("K_layer_cache_19_cast_fp16")]; tensor var_2494_begin_0 = const()[name = string("op_2494_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_2494_end_0 = const()[name = string("op_2494_end_0"), val = tensor([26, 8, 4096, 64])]; tensor var_2494_end_mask_0 = const()[name = string("op_2494_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2494_cast_fp16 = slice_by_index(begin = var_2494_begin_0, end = var_2494_end_0, end_mask = var_2494_end_mask_0, x = coreml_update_state_51)[name = string("op_2494_cast_fp16")]; tensor V_layer_cache_19_axes_0 = const()[name = string("V_layer_cache_19_axes_0"), val = tensor([0])]; tensor V_layer_cache_19_cast_fp16 = squeeze(axes = V_layer_cache_19_axes_0, x = var_2494_cast_fp16)[name = string("V_layer_cache_19_cast_fp16")]; tensor x_263_axes_0 = const()[name = string("x_263_axes_0"), val = tensor([1])]; tensor x_263_cast_fp16 = expand_dims(axes = x_263_axes_0, x = K_layer_cache_19_cast_fp16)[name = string("x_263_cast_fp16")]; tensor var_2503 = const()[name = string("op_2503"), val = tensor([1, 4, 1, 1])]; tensor x_265_cast_fp16 = tile(reps = var_2503, x = x_263_cast_fp16)[name = string("x_265_cast_fp16")]; tensor var_2507 = const()[name = string("op_2507"), val = tensor([1, -1, 4096, 64])]; tensor var_2508_cast_fp16 = reshape(shape = var_2507, x = x_265_cast_fp16)[name = string("op_2508_cast_fp16")]; tensor x_269_axes_0 = const()[name = string("x_269_axes_0"), val = tensor([1])]; tensor x_269_cast_fp16 = expand_dims(axes = x_269_axes_0, x = V_layer_cache_19_cast_fp16)[name = string("x_269_cast_fp16")]; tensor var_2510 = const()[name = string("op_2510"), val = tensor([1, 4, 1, 1])]; tensor x_271_cast_fp16 = tile(reps = var_2510, x = x_269_cast_fp16)[name = string("x_271_cast_fp16")]; bool var_2517_transpose_x_0 = const()[name = string("op_2517_transpose_x_0"), val = bool(false)]; bool var_2517_transpose_y_0 = const()[name = string("op_2517_transpose_y_0"), val = bool(true)]; tensor var_2517_cast_fp16 = matmul(transpose_x = var_2517_transpose_x_0, transpose_y = var_2517_transpose_y_0, x = rotated_37, y = var_2508_cast_fp16)[name = string("op_2517_cast_fp16")]; fp16 var_2518_to_fp16 = const()[name = string("op_2518_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_19_cast_fp16 = mul(x = var_2517_cast_fp16, y = var_2518_to_fp16)[name = string("attn_weights_19_cast_fp16")]; tensor x_273_cast_fp16 = add(x = attn_weights_19_cast_fp16, y = causal_mask)[name = string("x_273_cast_fp16")]; tensor reduce_max_9_axes_0 = const()[name = string("reduce_max_9_axes_0"), val = tensor([-1])]; bool reduce_max_9_keep_dims_0 = const()[name = string("reduce_max_9_keep_dims_0"), val = bool(true)]; tensor reduce_max_9_cast_fp16 = reduce_max(axes = reduce_max_9_axes_0, keep_dims = reduce_max_9_keep_dims_0, x = x_273_cast_fp16)[name = string("reduce_max_9_cast_fp16")]; tensor x_275_cast_fp16 = sub(x = x_273_cast_fp16, y = reduce_max_9_cast_fp16)[name = string("x_275_cast_fp16")]; tensor exp_x_19_cast_fp16 = exp(x = x_275_cast_fp16)[name = string("exp_x_19_cast_fp16")]; tensor var_2529_axes_0 = const()[name = string("op_2529_axes_0"), val = tensor([-1])]; bool var_2529_keep_dims_0 = const()[name = string("op_2529_keep_dims_0"), val = bool(true)]; tensor var_2529_cast_fp16 = reduce_sum(axes = var_2529_axes_0, keep_dims = var_2529_keep_dims_0, x = exp_x_19_cast_fp16)[name = string("op_2529_cast_fp16")]; tensor var_2530_cast_fp16 = real_div(x = exp_x_19_cast_fp16, y = var_2529_cast_fp16)[name = string("op_2530_cast_fp16")]; tensor concat_174 = const()[name = string("concat_174"), val = tensor([32, 64, 4096])]; tensor reshape_27_cast_fp16 = reshape(shape = concat_174, x = var_2530_cast_fp16)[name = string("reshape_27_cast_fp16")]; tensor concat_175 = const()[name = string("concat_175"), val = tensor([32, 4096, 64])]; tensor reshape_28_cast_fp16 = reshape(shape = concat_175, x = x_271_cast_fp16)[name = string("reshape_28_cast_fp16")]; bool matmul_9_transpose_x_0 = const()[name = string("matmul_9_transpose_x_0"), val = bool(false)]; bool matmul_9_transpose_y_0 = const()[name = string("matmul_9_transpose_y_0"), val = bool(false)]; tensor matmul_9_cast_fp16 = matmul(transpose_x = matmul_9_transpose_x_0, transpose_y = matmul_9_transpose_y_0, x = reshape_27_cast_fp16, y = reshape_28_cast_fp16)[name = string("matmul_9_cast_fp16")]; tensor concat_179 = const()[name = string("concat_179"), val = tensor([1, 32, 64, 64])]; tensor reshape_29_cast_fp16 = reshape(shape = concat_179, x = matmul_9_cast_fp16)[name = string("reshape_29_cast_fp16")]; tensor var_2533_perm_0 = const()[name = string("op_2533_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2535 = const()[name = string("op_2535"), val = tensor([1, 64, 2048])]; tensor var_2533_cast_fp16 = transpose(perm = var_2533_perm_0, x = reshape_29_cast_fp16)[name = string("transpose_42")]; tensor input_131_cast_fp16 = reshape(shape = var_2535, x = var_2533_cast_fp16)[name = string("input_131_cast_fp16")]; tensor model_model_layers_9_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(475503168))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477600384))))[name = string("model_model_layers_9_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_9_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_9_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_131_cast_fp16)[name = string("linear_9_cast_fp16")]; tensor hidden_states_77_cast_fp16 = add(x = hidden_states_73_cast_fp16, y = linear_9_cast_fp16)[name = string("hidden_states_77_cast_fp16")]; fp16 const_177_promoted_to_fp16 = const()[name = string("const_177_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2541_cast_fp16 = mul(x = hidden_states_77_cast_fp16, y = const_177_promoted_to_fp16)[name = string("op_2541_cast_fp16")]; bool input_133_interleave_0 = const()[name = string("input_133_interleave_0"), val = bool(false)]; tensor input_133_cast_fp16 = concat(axis = var_73, interleave = input_133_interleave_0, values = (hidden_states_77_cast_fp16, var_2541_cast_fp16))[name = string("input_133_cast_fp16")]; tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_76_to_fp16, x = input_133_cast_fp16)[name = string("normed_77_cast_fp16")]; tensor normed_79_begin_0 = const()[name = string("normed_79_begin_0"), val = tensor([0, 0, 0])]; tensor normed_79_end_0 = const()[name = string("normed_79_end_0"), val = tensor([1, 64, 2048])]; tensor normed_79_end_mask_0 = const()[name = string("normed_79_end_mask_0"), val = tensor([true, true, false])]; tensor normed_79_cast_fp16 = slice_by_index(begin = normed_79_begin_0, end = normed_79_end_0, end_mask = normed_79_end_mask_0, x = normed_77_cast_fp16)[name = string("normed_79_cast_fp16")]; tensor const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477608640)))]; tensor x_277_cast_fp16 = mul(x = normed_79_cast_fp16, y = const_180_promoted_to_fp16)[name = string("x_277_cast_fp16")]; tensor var_2559 = const()[name = string("op_2559"), val = tensor([0, 2, 1])]; tensor input_135_axes_0 = const()[name = string("input_135_axes_0"), val = tensor([2])]; tensor var_2560 = transpose(perm = var_2559, x = x_277_cast_fp16)[name = string("transpose_41")]; tensor input_135 = expand_dims(axes = input_135_axes_0, x = var_2560)[name = string("input_135")]; string input_137_pad_type_0 = const()[name = string("input_137_pad_type_0"), val = string("valid")]; tensor input_137_strides_0 = const()[name = string("input_137_strides_0"), val = tensor([1, 1])]; tensor input_137_pad_0 = const()[name = string("input_137_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_137_dilations_0 = const()[name = string("input_137_dilations_0"), val = tensor([1, 1])]; int32 input_137_groups_0 = const()[name = string("input_137_groups_0"), val = int32(1)]; tensor input_137 = conv(dilations = input_137_dilations_0, groups = input_137_groups_0, pad = input_137_pad_0, pad_type = input_137_pad_type_0, strides = input_137_strides_0, weight = model_model_layers_9_mlp_gate_proj_weight_palettized, x = input_135)[name = string("input_137")]; string up_states_19_pad_type_0 = const()[name = string("up_states_19_pad_type_0"), val = string("valid")]; tensor up_states_19_strides_0 = const()[name = string("up_states_19_strides_0"), val = tensor([1, 1])]; tensor up_states_19_pad_0 = const()[name = string("up_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_19_dilations_0 = const()[name = string("up_states_19_dilations_0"), val = tensor([1, 1])]; int32 up_states_19_groups_0 = const()[name = string("up_states_19_groups_0"), val = int32(1)]; tensor up_states_19 = conv(dilations = up_states_19_dilations_0, groups = up_states_19_groups_0, pad = up_states_19_pad_0, pad_type = up_states_19_pad_type_0, strides = up_states_19_strides_0, weight = model_model_layers_9_mlp_up_proj_weight_palettized, x = input_135)[name = string("up_states_19")]; tensor gate_states_19 = silu(x = input_137)[name = string("gate_states_19")]; tensor input_139 = mul(x = gate_states_19, y = up_states_19)[name = string("input_139")]; string hidden_states_79_pad_type_0 = const()[name = string("hidden_states_79_pad_type_0"), val = string("valid")]; tensor hidden_states_79_strides_0 = const()[name = string("hidden_states_79_strides_0"), val = tensor([1, 1])]; tensor hidden_states_79_pad_0 = const()[name = string("hidden_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_79_dilations_0 = const()[name = string("hidden_states_79_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_79_groups_0 = const()[name = string("hidden_states_79_groups_0"), val = int32(1)]; tensor hidden_states_79 = conv(dilations = hidden_states_79_dilations_0, groups = hidden_states_79_groups_0, pad = hidden_states_79_pad_0, pad_type = hidden_states_79_pad_type_0, strides = hidden_states_79_strides_0, weight = model_model_layers_9_mlp_down_proj_weight_palettized, x = input_139)[name = string("hidden_states_79")]; tensor var_2582_axes_0 = const()[name = string("op_2582_axes_0"), val = tensor([2])]; tensor var_2582 = squeeze(axes = var_2582_axes_0, x = hidden_states_79)[name = string("op_2582")]; tensor var_2583 = const()[name = string("op_2583"), val = tensor([0, 2, 1])]; tensor var_2584 = transpose(perm = var_2583, x = var_2582)[name = string("transpose_40")]; tensor hidden_states_81_cast_fp16 = add(x = hidden_states_77_cast_fp16, y = var_2584)[name = string("hidden_states_81_cast_fp16")]; fp16 const_181_promoted_to_fp16 = const()[name = string("const_181_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2587_cast_fp16 = mul(x = hidden_states_81_cast_fp16, y = const_181_promoted_to_fp16)[name = string("op_2587_cast_fp16")]; bool input_141_interleave_0 = const()[name = string("input_141_interleave_0"), val = bool(false)]; tensor input_141_cast_fp16 = concat(axis = var_73, interleave = input_141_interleave_0, values = (hidden_states_81_cast_fp16, var_2587_cast_fp16))[name = string("input_141_cast_fp16")]; tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_76_to_fp16, x = input_141_cast_fp16)[name = string("normed_81_cast_fp16")]; tensor normed_83_begin_0 = const()[name = string("normed_83_begin_0"), val = tensor([0, 0, 0])]; tensor normed_83_end_0 = const()[name = string("normed_83_end_0"), val = tensor([1, 64, 2048])]; tensor normed_83_end_mask_0 = const()[name = string("normed_83_end_mask_0"), val = tensor([true, true, false])]; tensor normed_83_cast_fp16 = slice_by_index(begin = normed_83_begin_0, end = normed_83_end_0, end_mask = normed_83_end_mask_0, x = normed_81_cast_fp16)[name = string("normed_83_cast_fp16")]; tensor const_184_promoted_to_fp16 = const()[name = string("const_184_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477612800)))]; tensor hidden_states_83_cast_fp16 = mul(x = normed_83_cast_fp16, y = const_184_promoted_to_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor var_2602 = const()[name = string("op_2602"), val = tensor([0, 2, 1])]; tensor var_2604_axes_0 = const()[name = string("op_2604_axes_0"), val = tensor([2])]; tensor var_2603_cast_fp16 = transpose(perm = var_2602, x = hidden_states_83_cast_fp16)[name = string("transpose_39")]; tensor var_2604_cast_fp16 = expand_dims(axes = var_2604_axes_0, x = var_2603_cast_fp16)[name = string("op_2604_cast_fp16")]; string query_states_41_pad_type_0 = const()[name = string("query_states_41_pad_type_0"), val = string("valid")]; tensor query_states_41_strides_0 = const()[name = string("query_states_41_strides_0"), val = tensor([1, 1])]; tensor query_states_41_pad_0 = const()[name = string("query_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_41_dilations_0 = const()[name = string("query_states_41_dilations_0"), val = tensor([1, 1])]; int32 query_states_41_groups_0 = const()[name = string("query_states_41_groups_0"), val = int32(1)]; tensor query_states_41 = conv(dilations = query_states_41_dilations_0, groups = query_states_41_groups_0, pad = query_states_41_pad_0, pad_type = query_states_41_pad_type_0, strides = query_states_41_strides_0, weight = model_model_layers_10_self_attn_q_proj_weight_palettized, x = var_2604_cast_fp16)[name = string("query_states_41")]; string key_states_61_pad_type_0 = const()[name = string("key_states_61_pad_type_0"), val = string("valid")]; tensor key_states_61_strides_0 = const()[name = string("key_states_61_strides_0"), val = tensor([1, 1])]; tensor key_states_61_pad_0 = const()[name = string("key_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_61_dilations_0 = const()[name = string("key_states_61_dilations_0"), val = tensor([1, 1])]; int32 key_states_61_groups_0 = const()[name = string("key_states_61_groups_0"), val = int32(1)]; tensor key_states_61 = conv(dilations = key_states_61_dilations_0, groups = key_states_61_groups_0, pad = key_states_61_pad_0, pad_type = key_states_61_pad_type_0, strides = key_states_61_strides_0, weight = model_model_layers_10_self_attn_k_proj_weight_palettized, x = var_2604_cast_fp16)[name = string("key_states_61")]; string value_states_61_pad_type_0 = const()[name = string("value_states_61_pad_type_0"), val = string("valid")]; tensor value_states_61_strides_0 = const()[name = string("value_states_61_strides_0"), val = tensor([1, 1])]; tensor value_states_61_pad_0 = const()[name = string("value_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_61_dilations_0 = const()[name = string("value_states_61_dilations_0"), val = tensor([1, 1])]; int32 value_states_61_groups_0 = const()[name = string("value_states_61_groups_0"), val = int32(1)]; tensor value_states_61 = conv(dilations = value_states_61_dilations_0, groups = value_states_61_groups_0, pad = value_states_61_pad_0, pad_type = value_states_61_pad_type_0, strides = value_states_61_strides_0, weight = model_model_layers_10_self_attn_v_proj_weight_palettized, x = var_2604_cast_fp16)[name = string("value_states_61")]; tensor var_2624 = const()[name = string("op_2624"), val = tensor([1, 32, 64, 64])]; tensor var_2625 = reshape(shape = var_2624, x = query_states_41)[name = string("op_2625")]; tensor var_2626 = const()[name = string("op_2626"), val = tensor([0, 1, 3, 2])]; tensor var_2628 = const()[name = string("op_2628"), val = tensor([1, 8, 64, 64])]; tensor var_2629 = reshape(shape = var_2628, x = key_states_61)[name = string("op_2629")]; tensor var_2630 = const()[name = string("op_2630"), val = tensor([0, 1, 3, 2])]; tensor var_2632 = const()[name = string("op_2632"), val = tensor([1, 8, 64, 64])]; tensor var_2633 = reshape(shape = var_2632, x = value_states_61)[name = string("op_2633")]; tensor var_2634 = const()[name = string("op_2634"), val = tensor([0, 1, 3, 2])]; tensor x1_41_begin_0 = const()[name = string("x1_41_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_41_end_0 = const()[name = string("x1_41_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_41_end_mask_0 = const()[name = string("x1_41_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_281 = transpose(perm = var_2626, x = var_2625)[name = string("transpose_38")]; tensor x1_41 = slice_by_index(begin = x1_41_begin_0, end = x1_41_end_0, end_mask = x1_41_end_mask_0, x = x_281)[name = string("x1_41")]; tensor x2_41_begin_0 = const()[name = string("x2_41_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_41_end_0 = const()[name = string("x2_41_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_41_end_mask_0 = const()[name = string("x2_41_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_41 = slice_by_index(begin = x2_41_begin_0, end = x2_41_end_0, end_mask = x2_41_end_mask_0, x = x_281)[name = string("x2_41")]; tensor var_2652 = mul(x = x1_41, y = cos_7)[name = string("op_2652")]; tensor var_2653 = mul(x = x2_41, y = sin_7)[name = string("op_2653")]; tensor var_2654 = sub(x = var_2652, y = var_2653)[name = string("op_2654")]; tensor var_2655 = mul(x = x2_41, y = cos_7)[name = string("op_2655")]; tensor var_2656 = mul(x = x1_41, y = sin_7)[name = string("op_2656")]; tensor var_2657 = add(x = var_2655, y = var_2656)[name = string("op_2657")]; bool rotated_41_interleave_0 = const()[name = string("rotated_41_interleave_0"), val = bool(false)]; tensor rotated_41 = concat(axis = var_73, interleave = rotated_41_interleave_0, values = (var_2654, var_2657))[name = string("rotated_41")]; tensor x1_43_begin_0 = const()[name = string("x1_43_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_43_end_0 = const()[name = string("x1_43_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_43_end_mask_0 = const()[name = string("x1_43_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_285 = transpose(perm = var_2630, x = var_2629)[name = string("transpose_37")]; tensor x1_43 = slice_by_index(begin = x1_43_begin_0, end = x1_43_end_0, end_mask = x1_43_end_mask_0, x = x_285)[name = string("x1_43")]; tensor x2_43_begin_0 = const()[name = string("x2_43_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_43_end_0 = const()[name = string("x2_43_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_43_end_mask_0 = const()[name = string("x2_43_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_43 = slice_by_index(begin = x2_43_begin_0, end = x2_43_end_0, end_mask = x2_43_end_mask_0, x = x_285)[name = string("x2_43")]; tensor var_2673 = mul(x = x1_43, y = cos_7)[name = string("op_2673")]; tensor var_2674 = mul(x = x2_43, y = sin_7)[name = string("op_2674")]; tensor var_2675 = sub(x = var_2673, y = var_2674)[name = string("op_2675")]; tensor var_2676 = mul(x = x2_43, y = cos_7)[name = string("op_2676")]; tensor var_2677 = mul(x = x1_43, y = sin_7)[name = string("op_2677")]; tensor var_2678 = add(x = var_2676, y = var_2677)[name = string("op_2678")]; bool rotated_43_interleave_0 = const()[name = string("rotated_43_interleave_0"), val = bool(false)]; tensor rotated_43 = concat(axis = var_73, interleave = rotated_43_interleave_0, values = (var_2675, var_2678))[name = string("rotated_43")]; tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; int32 concat_182_axis_0 = const()[name = string("concat_182_axis_0"), val = int32(0)]; bool concat_182_interleave_0 = const()[name = string("concat_182_interleave_0"), val = bool(false)]; tensor concat_182 = concat(axis = concat_182_axis_0, interleave = concat_182_interleave_0, values = (expand_dims_120, expand_dims_121, current_pos, expand_dims_123))[name = string("concat_182")]; tensor concat_183_values1_0 = const()[name = string("concat_183_values1_0"), val = tensor([0])]; tensor concat_183_values3_0 = const()[name = string("concat_183_values3_0"), val = tensor([0])]; int32 concat_183_axis_0 = const()[name = string("concat_183_axis_0"), val = int32(0)]; bool concat_183_interleave_0 = const()[name = string("concat_183_interleave_0"), val = bool(false)]; tensor concat_183 = concat(axis = concat_183_axis_0, interleave = concat_183_interleave_0, values = (expand_dims_124, concat_183_values1_0, var_597, concat_183_values3_0))[name = string("concat_183")]; tensor model_model_kv_cache_0_internal_tensor_assign_21_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_182, begin_mask = model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0, end = concat_183, end_mask = model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_21_stride_0, update = rotated_43, x = coreml_update_state_51)[name = string("model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_116_write_state")]; tensor coreml_update_state_52 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_116")]; tensor expand_dims_126 = const()[name = string("expand_dims_126"), val = tensor([26])]; tensor expand_dims_127 = const()[name = string("expand_dims_127"), val = tensor([0])]; tensor expand_dims_129 = const()[name = string("expand_dims_129"), val = tensor([0])]; tensor expand_dims_130 = const()[name = string("expand_dims_130"), val = tensor([27])]; int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_126, expand_dims_127, current_pos, expand_dims_129))[name = string("concat_186")]; tensor concat_187_values1_0 = const()[name = string("concat_187_values1_0"), val = tensor([0])]; tensor concat_187_values3_0 = const()[name = string("concat_187_values3_0"), val = tensor([0])]; int32 concat_187_axis_0 = const()[name = string("concat_187_axis_0"), val = int32(0)]; bool concat_187_interleave_0 = const()[name = string("concat_187_interleave_0"), val = bool(false)]; tensor concat_187 = concat(axis = concat_187_axis_0, interleave = concat_187_interleave_0, values = (expand_dims_130, concat_187_values1_0, var_597, concat_187_values3_0))[name = string("concat_187")]; tensor model_model_kv_cache_0_internal_tensor_assign_22_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_63 = transpose(perm = var_2634, x = var_2633)[name = string("transpose_36")]; tensor model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_186, begin_mask = model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0, end = concat_187, end_mask = model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_22_stride_0, update = value_states_63, x = coreml_update_state_52)[name = string("model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_117_write_state")]; tensor coreml_update_state_53 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_117")]; tensor var_2701_begin_0 = const()[name = string("op_2701_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_2701_end_0 = const()[name = string("op_2701_end_0"), val = tensor([11, 8, 4096, 64])]; tensor var_2701_end_mask_0 = const()[name = string("op_2701_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2701_cast_fp16 = slice_by_index(begin = var_2701_begin_0, end = var_2701_end_0, end_mask = var_2701_end_mask_0, x = coreml_update_state_53)[name = string("op_2701_cast_fp16")]; tensor K_layer_cache_21_axes_0 = const()[name = string("K_layer_cache_21_axes_0"), val = tensor([0])]; tensor K_layer_cache_21_cast_fp16 = squeeze(axes = K_layer_cache_21_axes_0, x = var_2701_cast_fp16)[name = string("K_layer_cache_21_cast_fp16")]; tensor var_2703_begin_0 = const()[name = string("op_2703_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_2703_end_0 = const()[name = string("op_2703_end_0"), val = tensor([27, 8, 4096, 64])]; tensor var_2703_end_mask_0 = const()[name = string("op_2703_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2703_cast_fp16 = slice_by_index(begin = var_2703_begin_0, end = var_2703_end_0, end_mask = var_2703_end_mask_0, x = coreml_update_state_53)[name = string("op_2703_cast_fp16")]; tensor V_layer_cache_21_axes_0 = const()[name = string("V_layer_cache_21_axes_0"), val = tensor([0])]; tensor V_layer_cache_21_cast_fp16 = squeeze(axes = V_layer_cache_21_axes_0, x = var_2703_cast_fp16)[name = string("V_layer_cache_21_cast_fp16")]; tensor x_291_axes_0 = const()[name = string("x_291_axes_0"), val = tensor([1])]; tensor x_291_cast_fp16 = expand_dims(axes = x_291_axes_0, x = K_layer_cache_21_cast_fp16)[name = string("x_291_cast_fp16")]; tensor var_2712 = const()[name = string("op_2712"), val = tensor([1, 4, 1, 1])]; tensor x_293_cast_fp16 = tile(reps = var_2712, x = x_291_cast_fp16)[name = string("x_293_cast_fp16")]; tensor var_2716 = const()[name = string("op_2716"), val = tensor([1, -1, 4096, 64])]; tensor var_2717_cast_fp16 = reshape(shape = var_2716, x = x_293_cast_fp16)[name = string("op_2717_cast_fp16")]; tensor x_297_axes_0 = const()[name = string("x_297_axes_0"), val = tensor([1])]; tensor x_297_cast_fp16 = expand_dims(axes = x_297_axes_0, x = V_layer_cache_21_cast_fp16)[name = string("x_297_cast_fp16")]; tensor var_2719 = const()[name = string("op_2719"), val = tensor([1, 4, 1, 1])]; tensor x_299_cast_fp16 = tile(reps = var_2719, x = x_297_cast_fp16)[name = string("x_299_cast_fp16")]; bool var_2726_transpose_x_0 = const()[name = string("op_2726_transpose_x_0"), val = bool(false)]; bool var_2726_transpose_y_0 = const()[name = string("op_2726_transpose_y_0"), val = bool(true)]; tensor var_2726_cast_fp16 = matmul(transpose_x = var_2726_transpose_x_0, transpose_y = var_2726_transpose_y_0, x = rotated_41, y = var_2717_cast_fp16)[name = string("op_2726_cast_fp16")]; fp16 var_2727_to_fp16 = const()[name = string("op_2727_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_21_cast_fp16 = mul(x = var_2726_cast_fp16, y = var_2727_to_fp16)[name = string("attn_weights_21_cast_fp16")]; tensor x_301_cast_fp16 = add(x = attn_weights_21_cast_fp16, y = causal_mask)[name = string("x_301_cast_fp16")]; tensor reduce_max_10_axes_0 = const()[name = string("reduce_max_10_axes_0"), val = tensor([-1])]; bool reduce_max_10_keep_dims_0 = const()[name = string("reduce_max_10_keep_dims_0"), val = bool(true)]; tensor reduce_max_10_cast_fp16 = reduce_max(axes = reduce_max_10_axes_0, keep_dims = reduce_max_10_keep_dims_0, x = x_301_cast_fp16)[name = string("reduce_max_10_cast_fp16")]; tensor x_303_cast_fp16 = sub(x = x_301_cast_fp16, y = reduce_max_10_cast_fp16)[name = string("x_303_cast_fp16")]; tensor exp_x_21_cast_fp16 = exp(x = x_303_cast_fp16)[name = string("exp_x_21_cast_fp16")]; tensor var_2738_axes_0 = const()[name = string("op_2738_axes_0"), val = tensor([-1])]; bool var_2738_keep_dims_0 = const()[name = string("op_2738_keep_dims_0"), val = bool(true)]; tensor var_2738_cast_fp16 = reduce_sum(axes = var_2738_axes_0, keep_dims = var_2738_keep_dims_0, x = exp_x_21_cast_fp16)[name = string("op_2738_cast_fp16")]; tensor var_2739_cast_fp16 = real_div(x = exp_x_21_cast_fp16, y = var_2738_cast_fp16)[name = string("op_2739_cast_fp16")]; tensor concat_192 = const()[name = string("concat_192"), val = tensor([32, 64, 4096])]; tensor reshape_30_cast_fp16 = reshape(shape = concat_192, x = var_2739_cast_fp16)[name = string("reshape_30_cast_fp16")]; tensor concat_193 = const()[name = string("concat_193"), val = tensor([32, 4096, 64])]; tensor reshape_31_cast_fp16 = reshape(shape = concat_193, x = x_299_cast_fp16)[name = string("reshape_31_cast_fp16")]; bool matmul_10_transpose_x_0 = const()[name = string("matmul_10_transpose_x_0"), val = bool(false)]; bool matmul_10_transpose_y_0 = const()[name = string("matmul_10_transpose_y_0"), val = bool(false)]; tensor matmul_10_cast_fp16 = matmul(transpose_x = matmul_10_transpose_x_0, transpose_y = matmul_10_transpose_y_0, x = reshape_30_cast_fp16, y = reshape_31_cast_fp16)[name = string("matmul_10_cast_fp16")]; tensor concat_197 = const()[name = string("concat_197"), val = tensor([1, 32, 64, 64])]; tensor reshape_32_cast_fp16 = reshape(shape = concat_197, x = matmul_10_cast_fp16)[name = string("reshape_32_cast_fp16")]; tensor var_2742_perm_0 = const()[name = string("op_2742_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2744 = const()[name = string("op_2744"), val = tensor([1, 64, 2048])]; tensor var_2742_cast_fp16 = transpose(perm = var_2742_perm_0, x = reshape_32_cast_fp16)[name = string("transpose_35")]; tensor input_145_cast_fp16 = reshape(shape = var_2744, x = var_2742_cast_fp16)[name = string("input_145_cast_fp16")]; tensor model_model_layers_10_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477616960))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479714176))))[name = string("model_model_layers_10_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_10_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_10_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_145_cast_fp16)[name = string("linear_10_cast_fp16")]; tensor hidden_states_85_cast_fp16 = add(x = hidden_states_81_cast_fp16, y = linear_10_cast_fp16)[name = string("hidden_states_85_cast_fp16")]; fp16 const_195_promoted_to_fp16 = const()[name = string("const_195_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2750_cast_fp16 = mul(x = hidden_states_85_cast_fp16, y = const_195_promoted_to_fp16)[name = string("op_2750_cast_fp16")]; bool input_147_interleave_0 = const()[name = string("input_147_interleave_0"), val = bool(false)]; tensor input_147_cast_fp16 = concat(axis = var_73, interleave = input_147_interleave_0, values = (hidden_states_85_cast_fp16, var_2750_cast_fp16))[name = string("input_147_cast_fp16")]; tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_76_to_fp16, x = input_147_cast_fp16)[name = string("normed_85_cast_fp16")]; tensor normed_87_begin_0 = const()[name = string("normed_87_begin_0"), val = tensor([0, 0, 0])]; tensor normed_87_end_0 = const()[name = string("normed_87_end_0"), val = tensor([1, 64, 2048])]; tensor normed_87_end_mask_0 = const()[name = string("normed_87_end_mask_0"), val = tensor([true, true, false])]; tensor normed_87_cast_fp16 = slice_by_index(begin = normed_87_begin_0, end = normed_87_end_0, end_mask = normed_87_end_mask_0, x = normed_85_cast_fp16)[name = string("normed_87_cast_fp16")]; tensor const_198_promoted_to_fp16 = const()[name = string("const_198_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479722432)))]; tensor x_305_cast_fp16 = mul(x = normed_87_cast_fp16, y = const_198_promoted_to_fp16)[name = string("x_305_cast_fp16")]; tensor var_2768 = const()[name = string("op_2768"), val = tensor([0, 2, 1])]; tensor input_149_axes_0 = const()[name = string("input_149_axes_0"), val = tensor([2])]; tensor var_2769 = transpose(perm = var_2768, x = x_305_cast_fp16)[name = string("transpose_34")]; tensor input_149 = expand_dims(axes = input_149_axes_0, x = var_2769)[name = string("input_149")]; string input_151_pad_type_0 = const()[name = string("input_151_pad_type_0"), val = string("valid")]; tensor input_151_strides_0 = const()[name = string("input_151_strides_0"), val = tensor([1, 1])]; tensor input_151_pad_0 = const()[name = string("input_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_151_dilations_0 = const()[name = string("input_151_dilations_0"), val = tensor([1, 1])]; int32 input_151_groups_0 = const()[name = string("input_151_groups_0"), val = int32(1)]; tensor input_151 = conv(dilations = input_151_dilations_0, groups = input_151_groups_0, pad = input_151_pad_0, pad_type = input_151_pad_type_0, strides = input_151_strides_0, weight = model_model_layers_10_mlp_gate_proj_weight_palettized, x = input_149)[name = string("input_151")]; string up_states_21_pad_type_0 = const()[name = string("up_states_21_pad_type_0"), val = string("valid")]; tensor up_states_21_strides_0 = const()[name = string("up_states_21_strides_0"), val = tensor([1, 1])]; tensor up_states_21_pad_0 = const()[name = string("up_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_21_dilations_0 = const()[name = string("up_states_21_dilations_0"), val = tensor([1, 1])]; int32 up_states_21_groups_0 = const()[name = string("up_states_21_groups_0"), val = int32(1)]; tensor up_states_21 = conv(dilations = up_states_21_dilations_0, groups = up_states_21_groups_0, pad = up_states_21_pad_0, pad_type = up_states_21_pad_type_0, strides = up_states_21_strides_0, weight = model_model_layers_10_mlp_up_proj_weight_palettized, x = input_149)[name = string("up_states_21")]; tensor gate_states_21 = silu(x = input_151)[name = string("gate_states_21")]; tensor input_153 = mul(x = gate_states_21, y = up_states_21)[name = string("input_153")]; string hidden_states_87_pad_type_0 = const()[name = string("hidden_states_87_pad_type_0"), val = string("valid")]; tensor hidden_states_87_strides_0 = const()[name = string("hidden_states_87_strides_0"), val = tensor([1, 1])]; tensor hidden_states_87_pad_0 = const()[name = string("hidden_states_87_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_87_dilations_0 = const()[name = string("hidden_states_87_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_87_groups_0 = const()[name = string("hidden_states_87_groups_0"), val = int32(1)]; tensor hidden_states_87 = conv(dilations = hidden_states_87_dilations_0, groups = hidden_states_87_groups_0, pad = hidden_states_87_pad_0, pad_type = hidden_states_87_pad_type_0, strides = hidden_states_87_strides_0, weight = model_model_layers_10_mlp_down_proj_weight_palettized, x = input_153)[name = string("hidden_states_87")]; tensor var_2791_axes_0 = const()[name = string("op_2791_axes_0"), val = tensor([2])]; tensor var_2791 = squeeze(axes = var_2791_axes_0, x = hidden_states_87)[name = string("op_2791")]; tensor var_2792 = const()[name = string("op_2792"), val = tensor([0, 2, 1])]; tensor var_2793 = transpose(perm = var_2792, x = var_2791)[name = string("transpose_33")]; tensor hidden_states_89_cast_fp16 = add(x = hidden_states_85_cast_fp16, y = var_2793)[name = string("hidden_states_89_cast_fp16")]; fp16 const_199_promoted_to_fp16 = const()[name = string("const_199_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2796_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_199_promoted_to_fp16)[name = string("op_2796_cast_fp16")]; bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; tensor input_155_cast_fp16 = concat(axis = var_73, interleave = input_155_interleave_0, values = (hidden_states_89_cast_fp16, var_2796_cast_fp16))[name = string("input_155_cast_fp16")]; tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_76_to_fp16, x = input_155_cast_fp16)[name = string("normed_89_cast_fp16")]; tensor normed_91_begin_0 = const()[name = string("normed_91_begin_0"), val = tensor([0, 0, 0])]; tensor normed_91_end_0 = const()[name = string("normed_91_end_0"), val = tensor([1, 64, 2048])]; tensor normed_91_end_mask_0 = const()[name = string("normed_91_end_mask_0"), val = tensor([true, true, false])]; tensor normed_91_cast_fp16 = slice_by_index(begin = normed_91_begin_0, end = normed_91_end_0, end_mask = normed_91_end_mask_0, x = normed_89_cast_fp16)[name = string("normed_91_cast_fp16")]; tensor const_202_promoted_to_fp16 = const()[name = string("const_202_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479726592)))]; tensor hidden_states_91_cast_fp16 = mul(x = normed_91_cast_fp16, y = const_202_promoted_to_fp16)[name = string("hidden_states_91_cast_fp16")]; tensor var_2811 = const()[name = string("op_2811"), val = tensor([0, 2, 1])]; tensor var_2813_axes_0 = const()[name = string("op_2813_axes_0"), val = tensor([2])]; tensor var_2812_cast_fp16 = transpose(perm = var_2811, x = hidden_states_91_cast_fp16)[name = string("transpose_32")]; tensor var_2813_cast_fp16 = expand_dims(axes = var_2813_axes_0, x = var_2812_cast_fp16)[name = string("op_2813_cast_fp16")]; string query_states_45_pad_type_0 = const()[name = string("query_states_45_pad_type_0"), val = string("valid")]; tensor query_states_45_strides_0 = const()[name = string("query_states_45_strides_0"), val = tensor([1, 1])]; tensor query_states_45_pad_0 = const()[name = string("query_states_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_45_dilations_0 = const()[name = string("query_states_45_dilations_0"), val = tensor([1, 1])]; int32 query_states_45_groups_0 = const()[name = string("query_states_45_groups_0"), val = int32(1)]; tensor query_states_45 = conv(dilations = query_states_45_dilations_0, groups = query_states_45_groups_0, pad = query_states_45_pad_0, pad_type = query_states_45_pad_type_0, strides = query_states_45_strides_0, weight = model_model_layers_11_self_attn_q_proj_weight_palettized, x = var_2813_cast_fp16)[name = string("query_states_45")]; string key_states_67_pad_type_0 = const()[name = string("key_states_67_pad_type_0"), val = string("valid")]; tensor key_states_67_strides_0 = const()[name = string("key_states_67_strides_0"), val = tensor([1, 1])]; tensor key_states_67_pad_0 = const()[name = string("key_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_67_dilations_0 = const()[name = string("key_states_67_dilations_0"), val = tensor([1, 1])]; int32 key_states_67_groups_0 = const()[name = string("key_states_67_groups_0"), val = int32(1)]; tensor key_states_67 = conv(dilations = key_states_67_dilations_0, groups = key_states_67_groups_0, pad = key_states_67_pad_0, pad_type = key_states_67_pad_type_0, strides = key_states_67_strides_0, weight = model_model_layers_11_self_attn_k_proj_weight_palettized, x = var_2813_cast_fp16)[name = string("key_states_67")]; string value_states_67_pad_type_0 = const()[name = string("value_states_67_pad_type_0"), val = string("valid")]; tensor value_states_67_strides_0 = const()[name = string("value_states_67_strides_0"), val = tensor([1, 1])]; tensor value_states_67_pad_0 = const()[name = string("value_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_67_dilations_0 = const()[name = string("value_states_67_dilations_0"), val = tensor([1, 1])]; int32 value_states_67_groups_0 = const()[name = string("value_states_67_groups_0"), val = int32(1)]; tensor value_states_67 = conv(dilations = value_states_67_dilations_0, groups = value_states_67_groups_0, pad = value_states_67_pad_0, pad_type = value_states_67_pad_type_0, strides = value_states_67_strides_0, weight = model_model_layers_11_self_attn_v_proj_weight_palettized, x = var_2813_cast_fp16)[name = string("value_states_67")]; tensor var_2833 = const()[name = string("op_2833"), val = tensor([1, 32, 64, 64])]; tensor var_2834 = reshape(shape = var_2833, x = query_states_45)[name = string("op_2834")]; tensor var_2835 = const()[name = string("op_2835"), val = tensor([0, 1, 3, 2])]; tensor var_2837 = const()[name = string("op_2837"), val = tensor([1, 8, 64, 64])]; tensor var_2838 = reshape(shape = var_2837, x = key_states_67)[name = string("op_2838")]; tensor var_2839 = const()[name = string("op_2839"), val = tensor([0, 1, 3, 2])]; tensor var_2841 = const()[name = string("op_2841"), val = tensor([1, 8, 64, 64])]; tensor var_2842 = reshape(shape = var_2841, x = value_states_67)[name = string("op_2842")]; tensor var_2843 = const()[name = string("op_2843"), val = tensor([0, 1, 3, 2])]; tensor x1_45_begin_0 = const()[name = string("x1_45_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_45_end_0 = const()[name = string("x1_45_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_45_end_mask_0 = const()[name = string("x1_45_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_309 = transpose(perm = var_2835, x = var_2834)[name = string("transpose_31")]; tensor x1_45 = slice_by_index(begin = x1_45_begin_0, end = x1_45_end_0, end_mask = x1_45_end_mask_0, x = x_309)[name = string("x1_45")]; tensor x2_45_begin_0 = const()[name = string("x2_45_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_45_end_0 = const()[name = string("x2_45_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_45_end_mask_0 = const()[name = string("x2_45_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_45 = slice_by_index(begin = x2_45_begin_0, end = x2_45_end_0, end_mask = x2_45_end_mask_0, x = x_309)[name = string("x2_45")]; tensor var_2861 = mul(x = x1_45, y = cos_7)[name = string("op_2861")]; tensor var_2862 = mul(x = x2_45, y = sin_7)[name = string("op_2862")]; tensor var_2863 = sub(x = var_2861, y = var_2862)[name = string("op_2863")]; tensor var_2864 = mul(x = x2_45, y = cos_7)[name = string("op_2864")]; tensor var_2865 = mul(x = x1_45, y = sin_7)[name = string("op_2865")]; tensor var_2866 = add(x = var_2864, y = var_2865)[name = string("op_2866")]; bool rotated_45_interleave_0 = const()[name = string("rotated_45_interleave_0"), val = bool(false)]; tensor rotated_45 = concat(axis = var_73, interleave = rotated_45_interleave_0, values = (var_2863, var_2866))[name = string("rotated_45")]; tensor x1_47_begin_0 = const()[name = string("x1_47_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_47_end_0 = const()[name = string("x1_47_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_47_end_mask_0 = const()[name = string("x1_47_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_313 = transpose(perm = var_2839, x = var_2838)[name = string("transpose_30")]; tensor x1_47 = slice_by_index(begin = x1_47_begin_0, end = x1_47_end_0, end_mask = x1_47_end_mask_0, x = x_313)[name = string("x1_47")]; tensor x2_47_begin_0 = const()[name = string("x2_47_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_47_end_0 = const()[name = string("x2_47_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_47_end_mask_0 = const()[name = string("x2_47_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_47 = slice_by_index(begin = x2_47_begin_0, end = x2_47_end_0, end_mask = x2_47_end_mask_0, x = x_313)[name = string("x2_47")]; tensor var_2882 = mul(x = x1_47, y = cos_7)[name = string("op_2882")]; tensor var_2883 = mul(x = x2_47, y = sin_7)[name = string("op_2883")]; tensor var_2884 = sub(x = var_2882, y = var_2883)[name = string("op_2884")]; tensor var_2885 = mul(x = x2_47, y = cos_7)[name = string("op_2885")]; tensor var_2886 = mul(x = x1_47, y = sin_7)[name = string("op_2886")]; tensor var_2887 = add(x = var_2885, y = var_2886)[name = string("op_2887")]; bool rotated_47_interleave_0 = const()[name = string("rotated_47_interleave_0"), val = bool(false)]; tensor rotated_47 = concat(axis = var_73, interleave = rotated_47_interleave_0, values = (var_2884, var_2887))[name = string("rotated_47")]; tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; int32 concat_200_axis_0 = const()[name = string("concat_200_axis_0"), val = int32(0)]; bool concat_200_interleave_0 = const()[name = string("concat_200_interleave_0"), val = bool(false)]; tensor concat_200 = concat(axis = concat_200_axis_0, interleave = concat_200_interleave_0, values = (expand_dims_132, expand_dims_133, current_pos, expand_dims_135))[name = string("concat_200")]; tensor concat_201_values1_0 = const()[name = string("concat_201_values1_0"), val = tensor([0])]; tensor concat_201_values3_0 = const()[name = string("concat_201_values3_0"), val = tensor([0])]; int32 concat_201_axis_0 = const()[name = string("concat_201_axis_0"), val = int32(0)]; bool concat_201_interleave_0 = const()[name = string("concat_201_interleave_0"), val = bool(false)]; tensor concat_201 = concat(axis = concat_201_axis_0, interleave = concat_201_interleave_0, values = (expand_dims_136, concat_201_values1_0, var_597, concat_201_values3_0))[name = string("concat_201")]; tensor model_model_kv_cache_0_internal_tensor_assign_23_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_200, begin_mask = model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0, end = concat_201, end_mask = model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_23_stride_0, update = rotated_47, x = coreml_update_state_53)[name = string("model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_118_write_state")]; tensor coreml_update_state_54 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_118")]; tensor expand_dims_138 = const()[name = string("expand_dims_138"), val = tensor([27])]; tensor expand_dims_139 = const()[name = string("expand_dims_139"), val = tensor([0])]; tensor expand_dims_141 = const()[name = string("expand_dims_141"), val = tensor([0])]; tensor expand_dims_142 = const()[name = string("expand_dims_142"), val = tensor([28])]; int32 concat_204_axis_0 = const()[name = string("concat_204_axis_0"), val = int32(0)]; bool concat_204_interleave_0 = const()[name = string("concat_204_interleave_0"), val = bool(false)]; tensor concat_204 = concat(axis = concat_204_axis_0, interleave = concat_204_interleave_0, values = (expand_dims_138, expand_dims_139, current_pos, expand_dims_141))[name = string("concat_204")]; tensor concat_205_values1_0 = const()[name = string("concat_205_values1_0"), val = tensor([0])]; tensor concat_205_values3_0 = const()[name = string("concat_205_values3_0"), val = tensor([0])]; int32 concat_205_axis_0 = const()[name = string("concat_205_axis_0"), val = int32(0)]; bool concat_205_interleave_0 = const()[name = string("concat_205_interleave_0"), val = bool(false)]; tensor concat_205 = concat(axis = concat_205_axis_0, interleave = concat_205_interleave_0, values = (expand_dims_142, concat_205_values1_0, var_597, concat_205_values3_0))[name = string("concat_205")]; tensor model_model_kv_cache_0_internal_tensor_assign_24_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_69 = transpose(perm = var_2843, x = var_2842)[name = string("transpose_29")]; tensor model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_204, begin_mask = model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0, end = concat_205, end_mask = model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_24_stride_0, update = value_states_69, x = coreml_update_state_54)[name = string("model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_119_write_state")]; tensor coreml_update_state_55 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_119")]; tensor var_2910_begin_0 = const()[name = string("op_2910_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_2910_end_0 = const()[name = string("op_2910_end_0"), val = tensor([12, 8, 4096, 64])]; tensor var_2910_end_mask_0 = const()[name = string("op_2910_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2910_cast_fp16 = slice_by_index(begin = var_2910_begin_0, end = var_2910_end_0, end_mask = var_2910_end_mask_0, x = coreml_update_state_55)[name = string("op_2910_cast_fp16")]; tensor K_layer_cache_23_axes_0 = const()[name = string("K_layer_cache_23_axes_0"), val = tensor([0])]; tensor K_layer_cache_23_cast_fp16 = squeeze(axes = K_layer_cache_23_axes_0, x = var_2910_cast_fp16)[name = string("K_layer_cache_23_cast_fp16")]; tensor var_2912_begin_0 = const()[name = string("op_2912_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_2912_end_0 = const()[name = string("op_2912_end_0"), val = tensor([28, 8, 4096, 64])]; tensor var_2912_end_mask_0 = const()[name = string("op_2912_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2912_cast_fp16 = slice_by_index(begin = var_2912_begin_0, end = var_2912_end_0, end_mask = var_2912_end_mask_0, x = coreml_update_state_55)[name = string("op_2912_cast_fp16")]; tensor V_layer_cache_23_axes_0 = const()[name = string("V_layer_cache_23_axes_0"), val = tensor([0])]; tensor V_layer_cache_23_cast_fp16 = squeeze(axes = V_layer_cache_23_axes_0, x = var_2912_cast_fp16)[name = string("V_layer_cache_23_cast_fp16")]; tensor x_319_axes_0 = const()[name = string("x_319_axes_0"), val = tensor([1])]; tensor x_319_cast_fp16 = expand_dims(axes = x_319_axes_0, x = K_layer_cache_23_cast_fp16)[name = string("x_319_cast_fp16")]; tensor var_2921 = const()[name = string("op_2921"), val = tensor([1, 4, 1, 1])]; tensor x_321_cast_fp16 = tile(reps = var_2921, x = x_319_cast_fp16)[name = string("x_321_cast_fp16")]; tensor var_2925 = const()[name = string("op_2925"), val = tensor([1, -1, 4096, 64])]; tensor var_2926_cast_fp16 = reshape(shape = var_2925, x = x_321_cast_fp16)[name = string("op_2926_cast_fp16")]; tensor x_325_axes_0 = const()[name = string("x_325_axes_0"), val = tensor([1])]; tensor x_325_cast_fp16 = expand_dims(axes = x_325_axes_0, x = V_layer_cache_23_cast_fp16)[name = string("x_325_cast_fp16")]; tensor var_2928 = const()[name = string("op_2928"), val = tensor([1, 4, 1, 1])]; tensor x_327_cast_fp16 = tile(reps = var_2928, x = x_325_cast_fp16)[name = string("x_327_cast_fp16")]; bool var_2935_transpose_x_0 = const()[name = string("op_2935_transpose_x_0"), val = bool(false)]; bool var_2935_transpose_y_0 = const()[name = string("op_2935_transpose_y_0"), val = bool(true)]; tensor var_2935_cast_fp16 = matmul(transpose_x = var_2935_transpose_x_0, transpose_y = var_2935_transpose_y_0, x = rotated_45, y = var_2926_cast_fp16)[name = string("op_2935_cast_fp16")]; fp16 var_2936_to_fp16 = const()[name = string("op_2936_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_23_cast_fp16 = mul(x = var_2935_cast_fp16, y = var_2936_to_fp16)[name = string("attn_weights_23_cast_fp16")]; tensor x_329_cast_fp16 = add(x = attn_weights_23_cast_fp16, y = causal_mask)[name = string("x_329_cast_fp16")]; tensor reduce_max_11_axes_0 = const()[name = string("reduce_max_11_axes_0"), val = tensor([-1])]; bool reduce_max_11_keep_dims_0 = const()[name = string("reduce_max_11_keep_dims_0"), val = bool(true)]; tensor reduce_max_11_cast_fp16 = reduce_max(axes = reduce_max_11_axes_0, keep_dims = reduce_max_11_keep_dims_0, x = x_329_cast_fp16)[name = string("reduce_max_11_cast_fp16")]; tensor x_331_cast_fp16 = sub(x = x_329_cast_fp16, y = reduce_max_11_cast_fp16)[name = string("x_331_cast_fp16")]; tensor exp_x_23_cast_fp16 = exp(x = x_331_cast_fp16)[name = string("exp_x_23_cast_fp16")]; tensor var_2947_axes_0 = const()[name = string("op_2947_axes_0"), val = tensor([-1])]; bool var_2947_keep_dims_0 = const()[name = string("op_2947_keep_dims_0"), val = bool(true)]; tensor var_2947_cast_fp16 = reduce_sum(axes = var_2947_axes_0, keep_dims = var_2947_keep_dims_0, x = exp_x_23_cast_fp16)[name = string("op_2947_cast_fp16")]; tensor var_2948_cast_fp16 = real_div(x = exp_x_23_cast_fp16, y = var_2947_cast_fp16)[name = string("op_2948_cast_fp16")]; tensor concat_210 = const()[name = string("concat_210"), val = tensor([32, 64, 4096])]; tensor reshape_33_cast_fp16 = reshape(shape = concat_210, x = var_2948_cast_fp16)[name = string("reshape_33_cast_fp16")]; tensor concat_211 = const()[name = string("concat_211"), val = tensor([32, 4096, 64])]; tensor reshape_34_cast_fp16 = reshape(shape = concat_211, x = x_327_cast_fp16)[name = string("reshape_34_cast_fp16")]; bool matmul_11_transpose_x_0 = const()[name = string("matmul_11_transpose_x_0"), val = bool(false)]; bool matmul_11_transpose_y_0 = const()[name = string("matmul_11_transpose_y_0"), val = bool(false)]; tensor matmul_11_cast_fp16 = matmul(transpose_x = matmul_11_transpose_x_0, transpose_y = matmul_11_transpose_y_0, x = reshape_33_cast_fp16, y = reshape_34_cast_fp16)[name = string("matmul_11_cast_fp16")]; tensor concat_215 = const()[name = string("concat_215"), val = tensor([1, 32, 64, 64])]; tensor reshape_35_cast_fp16 = reshape(shape = concat_215, x = matmul_11_cast_fp16)[name = string("reshape_35_cast_fp16")]; tensor var_2951_perm_0 = const()[name = string("op_2951_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_2953 = const()[name = string("op_2953"), val = tensor([1, 64, 2048])]; tensor var_2951_cast_fp16 = transpose(perm = var_2951_perm_0, x = reshape_35_cast_fp16)[name = string("transpose_28")]; tensor input_159_cast_fp16 = reshape(shape = var_2953, x = var_2951_cast_fp16)[name = string("input_159_cast_fp16")]; tensor model_model_layers_11_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479730752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481827968))))[name = string("model_model_layers_11_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_11_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_11_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_159_cast_fp16)[name = string("linear_11_cast_fp16")]; tensor hidden_states_93_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = linear_11_cast_fp16)[name = string("hidden_states_93_cast_fp16")]; fp16 const_213_promoted_to_fp16 = const()[name = string("const_213_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2959_cast_fp16 = mul(x = hidden_states_93_cast_fp16, y = const_213_promoted_to_fp16)[name = string("op_2959_cast_fp16")]; bool input_161_interleave_0 = const()[name = string("input_161_interleave_0"), val = bool(false)]; tensor input_161_cast_fp16 = concat(axis = var_73, interleave = input_161_interleave_0, values = (hidden_states_93_cast_fp16, var_2959_cast_fp16))[name = string("input_161_cast_fp16")]; tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_76_to_fp16, x = input_161_cast_fp16)[name = string("normed_93_cast_fp16")]; tensor normed_95_begin_0 = const()[name = string("normed_95_begin_0"), val = tensor([0, 0, 0])]; tensor normed_95_end_0 = const()[name = string("normed_95_end_0"), val = tensor([1, 64, 2048])]; tensor normed_95_end_mask_0 = const()[name = string("normed_95_end_mask_0"), val = tensor([true, true, false])]; tensor normed_95_cast_fp16 = slice_by_index(begin = normed_95_begin_0, end = normed_95_end_0, end_mask = normed_95_end_mask_0, x = normed_93_cast_fp16)[name = string("normed_95_cast_fp16")]; tensor const_216_promoted_to_fp16 = const()[name = string("const_216_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481836224)))]; tensor x_333_cast_fp16 = mul(x = normed_95_cast_fp16, y = const_216_promoted_to_fp16)[name = string("x_333_cast_fp16")]; tensor var_2977 = const()[name = string("op_2977"), val = tensor([0, 2, 1])]; tensor input_163_axes_0 = const()[name = string("input_163_axes_0"), val = tensor([2])]; tensor var_2978 = transpose(perm = var_2977, x = x_333_cast_fp16)[name = string("transpose_27")]; tensor input_163 = expand_dims(axes = input_163_axes_0, x = var_2978)[name = string("input_163")]; string input_165_pad_type_0 = const()[name = string("input_165_pad_type_0"), val = string("valid")]; tensor input_165_strides_0 = const()[name = string("input_165_strides_0"), val = tensor([1, 1])]; tensor input_165_pad_0 = const()[name = string("input_165_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_165_dilations_0 = const()[name = string("input_165_dilations_0"), val = tensor([1, 1])]; int32 input_165_groups_0 = const()[name = string("input_165_groups_0"), val = int32(1)]; tensor input_165 = conv(dilations = input_165_dilations_0, groups = input_165_groups_0, pad = input_165_pad_0, pad_type = input_165_pad_type_0, strides = input_165_strides_0, weight = model_model_layers_11_mlp_gate_proj_weight_palettized, x = input_163)[name = string("input_165")]; string up_states_23_pad_type_0 = const()[name = string("up_states_23_pad_type_0"), val = string("valid")]; tensor up_states_23_strides_0 = const()[name = string("up_states_23_strides_0"), val = tensor([1, 1])]; tensor up_states_23_pad_0 = const()[name = string("up_states_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_23_dilations_0 = const()[name = string("up_states_23_dilations_0"), val = tensor([1, 1])]; int32 up_states_23_groups_0 = const()[name = string("up_states_23_groups_0"), val = int32(1)]; tensor up_states_23 = conv(dilations = up_states_23_dilations_0, groups = up_states_23_groups_0, pad = up_states_23_pad_0, pad_type = up_states_23_pad_type_0, strides = up_states_23_strides_0, weight = model_model_layers_11_mlp_up_proj_weight_palettized, x = input_163)[name = string("up_states_23")]; tensor gate_states_23 = silu(x = input_165)[name = string("gate_states_23")]; tensor input_167 = mul(x = gate_states_23, y = up_states_23)[name = string("input_167")]; string hidden_states_95_pad_type_0 = const()[name = string("hidden_states_95_pad_type_0"), val = string("valid")]; tensor hidden_states_95_strides_0 = const()[name = string("hidden_states_95_strides_0"), val = tensor([1, 1])]; tensor hidden_states_95_pad_0 = const()[name = string("hidden_states_95_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_95_dilations_0 = const()[name = string("hidden_states_95_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_95_groups_0 = const()[name = string("hidden_states_95_groups_0"), val = int32(1)]; tensor hidden_states_95 = conv(dilations = hidden_states_95_dilations_0, groups = hidden_states_95_groups_0, pad = hidden_states_95_pad_0, pad_type = hidden_states_95_pad_type_0, strides = hidden_states_95_strides_0, weight = model_model_layers_11_mlp_down_proj_weight_palettized, x = input_167)[name = string("hidden_states_95")]; tensor var_3000_axes_0 = const()[name = string("op_3000_axes_0"), val = tensor([2])]; tensor var_3000 = squeeze(axes = var_3000_axes_0, x = hidden_states_95)[name = string("op_3000")]; tensor var_3001 = const()[name = string("op_3001"), val = tensor([0, 2, 1])]; tensor var_3002 = transpose(perm = var_3001, x = var_3000)[name = string("transpose_26")]; tensor hidden_states_97_cast_fp16 = add(x = hidden_states_93_cast_fp16, y = var_3002)[name = string("hidden_states_97_cast_fp16")]; fp16 const_217_promoted_to_fp16 = const()[name = string("const_217_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3005_cast_fp16 = mul(x = hidden_states_97_cast_fp16, y = const_217_promoted_to_fp16)[name = string("op_3005_cast_fp16")]; bool input_169_interleave_0 = const()[name = string("input_169_interleave_0"), val = bool(false)]; tensor input_169_cast_fp16 = concat(axis = var_73, interleave = input_169_interleave_0, values = (hidden_states_97_cast_fp16, var_3005_cast_fp16))[name = string("input_169_cast_fp16")]; tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_76_to_fp16, x = input_169_cast_fp16)[name = string("normed_97_cast_fp16")]; tensor normed_99_begin_0 = const()[name = string("normed_99_begin_0"), val = tensor([0, 0, 0])]; tensor normed_99_end_0 = const()[name = string("normed_99_end_0"), val = tensor([1, 64, 2048])]; tensor normed_99_end_mask_0 = const()[name = string("normed_99_end_mask_0"), val = tensor([true, true, false])]; tensor normed_99_cast_fp16 = slice_by_index(begin = normed_99_begin_0, end = normed_99_end_0, end_mask = normed_99_end_mask_0, x = normed_97_cast_fp16)[name = string("normed_99_cast_fp16")]; tensor const_220_promoted_to_fp16 = const()[name = string("const_220_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481840384)))]; tensor hidden_states_99_cast_fp16 = mul(x = normed_99_cast_fp16, y = const_220_promoted_to_fp16)[name = string("hidden_states_99_cast_fp16")]; tensor var_3020 = const()[name = string("op_3020"), val = tensor([0, 2, 1])]; tensor var_3022_axes_0 = const()[name = string("op_3022_axes_0"), val = tensor([2])]; tensor var_3021_cast_fp16 = transpose(perm = var_3020, x = hidden_states_99_cast_fp16)[name = string("transpose_25")]; tensor var_3022_cast_fp16 = expand_dims(axes = var_3022_axes_0, x = var_3021_cast_fp16)[name = string("op_3022_cast_fp16")]; string query_states_49_pad_type_0 = const()[name = string("query_states_49_pad_type_0"), val = string("valid")]; tensor query_states_49_strides_0 = const()[name = string("query_states_49_strides_0"), val = tensor([1, 1])]; tensor query_states_49_pad_0 = const()[name = string("query_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_49_dilations_0 = const()[name = string("query_states_49_dilations_0"), val = tensor([1, 1])]; int32 query_states_49_groups_0 = const()[name = string("query_states_49_groups_0"), val = int32(1)]; tensor query_states_49 = conv(dilations = query_states_49_dilations_0, groups = query_states_49_groups_0, pad = query_states_49_pad_0, pad_type = query_states_49_pad_type_0, strides = query_states_49_strides_0, weight = model_model_layers_12_self_attn_q_proj_weight_palettized, x = var_3022_cast_fp16)[name = string("query_states_49")]; string key_states_73_pad_type_0 = const()[name = string("key_states_73_pad_type_0"), val = string("valid")]; tensor key_states_73_strides_0 = const()[name = string("key_states_73_strides_0"), val = tensor([1, 1])]; tensor key_states_73_pad_0 = const()[name = string("key_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_73_dilations_0 = const()[name = string("key_states_73_dilations_0"), val = tensor([1, 1])]; int32 key_states_73_groups_0 = const()[name = string("key_states_73_groups_0"), val = int32(1)]; tensor key_states_73 = conv(dilations = key_states_73_dilations_0, groups = key_states_73_groups_0, pad = key_states_73_pad_0, pad_type = key_states_73_pad_type_0, strides = key_states_73_strides_0, weight = model_model_layers_12_self_attn_k_proj_weight_palettized, x = var_3022_cast_fp16)[name = string("key_states_73")]; string value_states_73_pad_type_0 = const()[name = string("value_states_73_pad_type_0"), val = string("valid")]; tensor value_states_73_strides_0 = const()[name = string("value_states_73_strides_0"), val = tensor([1, 1])]; tensor value_states_73_pad_0 = const()[name = string("value_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_73_dilations_0 = const()[name = string("value_states_73_dilations_0"), val = tensor([1, 1])]; int32 value_states_73_groups_0 = const()[name = string("value_states_73_groups_0"), val = int32(1)]; tensor value_states_73 = conv(dilations = value_states_73_dilations_0, groups = value_states_73_groups_0, pad = value_states_73_pad_0, pad_type = value_states_73_pad_type_0, strides = value_states_73_strides_0, weight = model_model_layers_12_self_attn_v_proj_weight_palettized, x = var_3022_cast_fp16)[name = string("value_states_73")]; tensor var_3042 = const()[name = string("op_3042"), val = tensor([1, 32, 64, 64])]; tensor var_3043 = reshape(shape = var_3042, x = query_states_49)[name = string("op_3043")]; tensor var_3044 = const()[name = string("op_3044"), val = tensor([0, 1, 3, 2])]; tensor var_3046 = const()[name = string("op_3046"), val = tensor([1, 8, 64, 64])]; tensor var_3047 = reshape(shape = var_3046, x = key_states_73)[name = string("op_3047")]; tensor var_3048 = const()[name = string("op_3048"), val = tensor([0, 1, 3, 2])]; tensor var_3050 = const()[name = string("op_3050"), val = tensor([1, 8, 64, 64])]; tensor var_3051 = reshape(shape = var_3050, x = value_states_73)[name = string("op_3051")]; tensor var_3052 = const()[name = string("op_3052"), val = tensor([0, 1, 3, 2])]; tensor x1_49_begin_0 = const()[name = string("x1_49_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_49_end_0 = const()[name = string("x1_49_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_49_end_mask_0 = const()[name = string("x1_49_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_337 = transpose(perm = var_3044, x = var_3043)[name = string("transpose_24")]; tensor x1_49 = slice_by_index(begin = x1_49_begin_0, end = x1_49_end_0, end_mask = x1_49_end_mask_0, x = x_337)[name = string("x1_49")]; tensor x2_49_begin_0 = const()[name = string("x2_49_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_49_end_0 = const()[name = string("x2_49_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_49_end_mask_0 = const()[name = string("x2_49_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_49 = slice_by_index(begin = x2_49_begin_0, end = x2_49_end_0, end_mask = x2_49_end_mask_0, x = x_337)[name = string("x2_49")]; tensor var_3070 = mul(x = x1_49, y = cos_7)[name = string("op_3070")]; tensor var_3071 = mul(x = x2_49, y = sin_7)[name = string("op_3071")]; tensor var_3072 = sub(x = var_3070, y = var_3071)[name = string("op_3072")]; tensor var_3073 = mul(x = x2_49, y = cos_7)[name = string("op_3073")]; tensor var_3074 = mul(x = x1_49, y = sin_7)[name = string("op_3074")]; tensor var_3075 = add(x = var_3073, y = var_3074)[name = string("op_3075")]; bool rotated_49_interleave_0 = const()[name = string("rotated_49_interleave_0"), val = bool(false)]; tensor rotated_49 = concat(axis = var_73, interleave = rotated_49_interleave_0, values = (var_3072, var_3075))[name = string("rotated_49")]; tensor x1_51_begin_0 = const()[name = string("x1_51_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_51_end_0 = const()[name = string("x1_51_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_51_end_mask_0 = const()[name = string("x1_51_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_341 = transpose(perm = var_3048, x = var_3047)[name = string("transpose_23")]; tensor x1_51 = slice_by_index(begin = x1_51_begin_0, end = x1_51_end_0, end_mask = x1_51_end_mask_0, x = x_341)[name = string("x1_51")]; tensor x2_51_begin_0 = const()[name = string("x2_51_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_51_end_0 = const()[name = string("x2_51_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_51_end_mask_0 = const()[name = string("x2_51_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_51 = slice_by_index(begin = x2_51_begin_0, end = x2_51_end_0, end_mask = x2_51_end_mask_0, x = x_341)[name = string("x2_51")]; tensor var_3091 = mul(x = x1_51, y = cos_7)[name = string("op_3091")]; tensor var_3092 = mul(x = x2_51, y = sin_7)[name = string("op_3092")]; tensor var_3093 = sub(x = var_3091, y = var_3092)[name = string("op_3093")]; tensor var_3094 = mul(x = x2_51, y = cos_7)[name = string("op_3094")]; tensor var_3095 = mul(x = x1_51, y = sin_7)[name = string("op_3095")]; tensor var_3096 = add(x = var_3094, y = var_3095)[name = string("op_3096")]; bool rotated_51_interleave_0 = const()[name = string("rotated_51_interleave_0"), val = bool(false)]; tensor rotated_51 = concat(axis = var_73, interleave = rotated_51_interleave_0, values = (var_3093, var_3096))[name = string("rotated_51")]; tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; int32 concat_218_axis_0 = const()[name = string("concat_218_axis_0"), val = int32(0)]; bool concat_218_interleave_0 = const()[name = string("concat_218_interleave_0"), val = bool(false)]; tensor concat_218 = concat(axis = concat_218_axis_0, interleave = concat_218_interleave_0, values = (expand_dims_144, expand_dims_145, current_pos, expand_dims_147))[name = string("concat_218")]; tensor concat_219_values1_0 = const()[name = string("concat_219_values1_0"), val = tensor([0])]; tensor concat_219_values3_0 = const()[name = string("concat_219_values3_0"), val = tensor([0])]; int32 concat_219_axis_0 = const()[name = string("concat_219_axis_0"), val = int32(0)]; bool concat_219_interleave_0 = const()[name = string("concat_219_interleave_0"), val = bool(false)]; tensor concat_219 = concat(axis = concat_219_axis_0, interleave = concat_219_interleave_0, values = (expand_dims_148, concat_219_values1_0, var_597, concat_219_values3_0))[name = string("concat_219")]; tensor model_model_kv_cache_0_internal_tensor_assign_25_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_218, begin_mask = model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0, end = concat_219, end_mask = model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_25_stride_0, update = rotated_51, x = coreml_update_state_55)[name = string("model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_120_write_state")]; tensor coreml_update_state_56 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_120")]; tensor expand_dims_150 = const()[name = string("expand_dims_150"), val = tensor([28])]; tensor expand_dims_151 = const()[name = string("expand_dims_151"), val = tensor([0])]; tensor expand_dims_153 = const()[name = string("expand_dims_153"), val = tensor([0])]; tensor expand_dims_154 = const()[name = string("expand_dims_154"), val = tensor([29])]; int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_150, expand_dims_151, current_pos, expand_dims_153))[name = string("concat_222")]; tensor concat_223_values1_0 = const()[name = string("concat_223_values1_0"), val = tensor([0])]; tensor concat_223_values3_0 = const()[name = string("concat_223_values3_0"), val = tensor([0])]; int32 concat_223_axis_0 = const()[name = string("concat_223_axis_0"), val = int32(0)]; bool concat_223_interleave_0 = const()[name = string("concat_223_interleave_0"), val = bool(false)]; tensor concat_223 = concat(axis = concat_223_axis_0, interleave = concat_223_interleave_0, values = (expand_dims_154, concat_223_values1_0, var_597, concat_223_values3_0))[name = string("concat_223")]; tensor model_model_kv_cache_0_internal_tensor_assign_26_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_75 = transpose(perm = var_3052, x = var_3051)[name = string("transpose_22")]; tensor model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_222, begin_mask = model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0, end = concat_223, end_mask = model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_26_stride_0, update = value_states_75, x = coreml_update_state_56)[name = string("model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_121_write_state")]; tensor coreml_update_state_57 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_121")]; tensor var_3119_begin_0 = const()[name = string("op_3119_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_3119_end_0 = const()[name = string("op_3119_end_0"), val = tensor([13, 8, 4096, 64])]; tensor var_3119_end_mask_0 = const()[name = string("op_3119_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3119_cast_fp16 = slice_by_index(begin = var_3119_begin_0, end = var_3119_end_0, end_mask = var_3119_end_mask_0, x = coreml_update_state_57)[name = string("op_3119_cast_fp16")]; tensor K_layer_cache_25_axes_0 = const()[name = string("K_layer_cache_25_axes_0"), val = tensor([0])]; tensor K_layer_cache_25_cast_fp16 = squeeze(axes = K_layer_cache_25_axes_0, x = var_3119_cast_fp16)[name = string("K_layer_cache_25_cast_fp16")]; tensor var_3121_begin_0 = const()[name = string("op_3121_begin_0"), val = tensor([28, 0, 0, 0])]; tensor var_3121_end_0 = const()[name = string("op_3121_end_0"), val = tensor([29, 8, 4096, 64])]; tensor var_3121_end_mask_0 = const()[name = string("op_3121_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3121_cast_fp16 = slice_by_index(begin = var_3121_begin_0, end = var_3121_end_0, end_mask = var_3121_end_mask_0, x = coreml_update_state_57)[name = string("op_3121_cast_fp16")]; tensor V_layer_cache_25_axes_0 = const()[name = string("V_layer_cache_25_axes_0"), val = tensor([0])]; tensor V_layer_cache_25_cast_fp16 = squeeze(axes = V_layer_cache_25_axes_0, x = var_3121_cast_fp16)[name = string("V_layer_cache_25_cast_fp16")]; tensor x_347_axes_0 = const()[name = string("x_347_axes_0"), val = tensor([1])]; tensor x_347_cast_fp16 = expand_dims(axes = x_347_axes_0, x = K_layer_cache_25_cast_fp16)[name = string("x_347_cast_fp16")]; tensor var_3130 = const()[name = string("op_3130"), val = tensor([1, 4, 1, 1])]; tensor x_349_cast_fp16 = tile(reps = var_3130, x = x_347_cast_fp16)[name = string("x_349_cast_fp16")]; tensor var_3134 = const()[name = string("op_3134"), val = tensor([1, -1, 4096, 64])]; tensor var_3135_cast_fp16 = reshape(shape = var_3134, x = x_349_cast_fp16)[name = string("op_3135_cast_fp16")]; tensor x_353_axes_0 = const()[name = string("x_353_axes_0"), val = tensor([1])]; tensor x_353_cast_fp16 = expand_dims(axes = x_353_axes_0, x = V_layer_cache_25_cast_fp16)[name = string("x_353_cast_fp16")]; tensor var_3137 = const()[name = string("op_3137"), val = tensor([1, 4, 1, 1])]; tensor x_355_cast_fp16 = tile(reps = var_3137, x = x_353_cast_fp16)[name = string("x_355_cast_fp16")]; bool var_3144_transpose_x_0 = const()[name = string("op_3144_transpose_x_0"), val = bool(false)]; bool var_3144_transpose_y_0 = const()[name = string("op_3144_transpose_y_0"), val = bool(true)]; tensor var_3144_cast_fp16 = matmul(transpose_x = var_3144_transpose_x_0, transpose_y = var_3144_transpose_y_0, x = rotated_49, y = var_3135_cast_fp16)[name = string("op_3144_cast_fp16")]; fp16 var_3145_to_fp16 = const()[name = string("op_3145_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_25_cast_fp16 = mul(x = var_3144_cast_fp16, y = var_3145_to_fp16)[name = string("attn_weights_25_cast_fp16")]; tensor x_357_cast_fp16 = add(x = attn_weights_25_cast_fp16, y = causal_mask)[name = string("x_357_cast_fp16")]; tensor reduce_max_12_axes_0 = const()[name = string("reduce_max_12_axes_0"), val = tensor([-1])]; bool reduce_max_12_keep_dims_0 = const()[name = string("reduce_max_12_keep_dims_0"), val = bool(true)]; tensor reduce_max_12_cast_fp16 = reduce_max(axes = reduce_max_12_axes_0, keep_dims = reduce_max_12_keep_dims_0, x = x_357_cast_fp16)[name = string("reduce_max_12_cast_fp16")]; tensor x_359_cast_fp16 = sub(x = x_357_cast_fp16, y = reduce_max_12_cast_fp16)[name = string("x_359_cast_fp16")]; tensor exp_x_25_cast_fp16 = exp(x = x_359_cast_fp16)[name = string("exp_x_25_cast_fp16")]; tensor var_3156_axes_0 = const()[name = string("op_3156_axes_0"), val = tensor([-1])]; bool var_3156_keep_dims_0 = const()[name = string("op_3156_keep_dims_0"), val = bool(true)]; tensor var_3156_cast_fp16 = reduce_sum(axes = var_3156_axes_0, keep_dims = var_3156_keep_dims_0, x = exp_x_25_cast_fp16)[name = string("op_3156_cast_fp16")]; tensor var_3157_cast_fp16 = real_div(x = exp_x_25_cast_fp16, y = var_3156_cast_fp16)[name = string("op_3157_cast_fp16")]; tensor concat_228 = const()[name = string("concat_228"), val = tensor([32, 64, 4096])]; tensor reshape_36_cast_fp16 = reshape(shape = concat_228, x = var_3157_cast_fp16)[name = string("reshape_36_cast_fp16")]; tensor concat_229 = const()[name = string("concat_229"), val = tensor([32, 4096, 64])]; tensor reshape_37_cast_fp16 = reshape(shape = concat_229, x = x_355_cast_fp16)[name = string("reshape_37_cast_fp16")]; bool matmul_12_transpose_x_0 = const()[name = string("matmul_12_transpose_x_0"), val = bool(false)]; bool matmul_12_transpose_y_0 = const()[name = string("matmul_12_transpose_y_0"), val = bool(false)]; tensor matmul_12_cast_fp16 = matmul(transpose_x = matmul_12_transpose_x_0, transpose_y = matmul_12_transpose_y_0, x = reshape_36_cast_fp16, y = reshape_37_cast_fp16)[name = string("matmul_12_cast_fp16")]; tensor concat_233 = const()[name = string("concat_233"), val = tensor([1, 32, 64, 64])]; tensor reshape_38_cast_fp16 = reshape(shape = concat_233, x = matmul_12_cast_fp16)[name = string("reshape_38_cast_fp16")]; tensor var_3160_perm_0 = const()[name = string("op_3160_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3162 = const()[name = string("op_3162"), val = tensor([1, 64, 2048])]; tensor var_3160_cast_fp16 = transpose(perm = var_3160_perm_0, x = reshape_38_cast_fp16)[name = string("transpose_21")]; tensor input_173_cast_fp16 = reshape(shape = var_3162, x = var_3160_cast_fp16)[name = string("input_173_cast_fp16")]; tensor model_model_layers_12_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481844544))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483941760))))[name = string("model_model_layers_12_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_12_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_12_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_173_cast_fp16)[name = string("linear_12_cast_fp16")]; tensor hidden_states_101_cast_fp16 = add(x = hidden_states_97_cast_fp16, y = linear_12_cast_fp16)[name = string("hidden_states_101_cast_fp16")]; fp16 const_231_promoted_to_fp16 = const()[name = string("const_231_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3168_cast_fp16 = mul(x = hidden_states_101_cast_fp16, y = const_231_promoted_to_fp16)[name = string("op_3168_cast_fp16")]; bool input_175_interleave_0 = const()[name = string("input_175_interleave_0"), val = bool(false)]; tensor input_175_cast_fp16 = concat(axis = var_73, interleave = input_175_interleave_0, values = (hidden_states_101_cast_fp16, var_3168_cast_fp16))[name = string("input_175_cast_fp16")]; tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_76_to_fp16, x = input_175_cast_fp16)[name = string("normed_101_cast_fp16")]; tensor normed_103_begin_0 = const()[name = string("normed_103_begin_0"), val = tensor([0, 0, 0])]; tensor normed_103_end_0 = const()[name = string("normed_103_end_0"), val = tensor([1, 64, 2048])]; tensor normed_103_end_mask_0 = const()[name = string("normed_103_end_mask_0"), val = tensor([true, true, false])]; tensor normed_103_cast_fp16 = slice_by_index(begin = normed_103_begin_0, end = normed_103_end_0, end_mask = normed_103_end_mask_0, x = normed_101_cast_fp16)[name = string("normed_103_cast_fp16")]; tensor const_234_promoted_to_fp16 = const()[name = string("const_234_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483950016)))]; tensor x_361_cast_fp16 = mul(x = normed_103_cast_fp16, y = const_234_promoted_to_fp16)[name = string("x_361_cast_fp16")]; tensor var_3186 = const()[name = string("op_3186"), val = tensor([0, 2, 1])]; tensor input_177_axes_0 = const()[name = string("input_177_axes_0"), val = tensor([2])]; tensor var_3187 = transpose(perm = var_3186, x = x_361_cast_fp16)[name = string("transpose_20")]; tensor input_177 = expand_dims(axes = input_177_axes_0, x = var_3187)[name = string("input_177")]; string input_179_pad_type_0 = const()[name = string("input_179_pad_type_0"), val = string("valid")]; tensor input_179_strides_0 = const()[name = string("input_179_strides_0"), val = tensor([1, 1])]; tensor input_179_pad_0 = const()[name = string("input_179_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_179_dilations_0 = const()[name = string("input_179_dilations_0"), val = tensor([1, 1])]; int32 input_179_groups_0 = const()[name = string("input_179_groups_0"), val = int32(1)]; tensor input_179 = conv(dilations = input_179_dilations_0, groups = input_179_groups_0, pad = input_179_pad_0, pad_type = input_179_pad_type_0, strides = input_179_strides_0, weight = model_model_layers_12_mlp_gate_proj_weight_palettized, x = input_177)[name = string("input_179")]; string up_states_25_pad_type_0 = const()[name = string("up_states_25_pad_type_0"), val = string("valid")]; tensor up_states_25_strides_0 = const()[name = string("up_states_25_strides_0"), val = tensor([1, 1])]; tensor up_states_25_pad_0 = const()[name = string("up_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_25_dilations_0 = const()[name = string("up_states_25_dilations_0"), val = tensor([1, 1])]; int32 up_states_25_groups_0 = const()[name = string("up_states_25_groups_0"), val = int32(1)]; tensor up_states_25 = conv(dilations = up_states_25_dilations_0, groups = up_states_25_groups_0, pad = up_states_25_pad_0, pad_type = up_states_25_pad_type_0, strides = up_states_25_strides_0, weight = model_model_layers_12_mlp_up_proj_weight_palettized, x = input_177)[name = string("up_states_25")]; tensor gate_states_25 = silu(x = input_179)[name = string("gate_states_25")]; tensor input_181 = mul(x = gate_states_25, y = up_states_25)[name = string("input_181")]; string hidden_states_103_pad_type_0 = const()[name = string("hidden_states_103_pad_type_0"), val = string("valid")]; tensor hidden_states_103_strides_0 = const()[name = string("hidden_states_103_strides_0"), val = tensor([1, 1])]; tensor hidden_states_103_pad_0 = const()[name = string("hidden_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_103_dilations_0 = const()[name = string("hidden_states_103_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_103_groups_0 = const()[name = string("hidden_states_103_groups_0"), val = int32(1)]; tensor hidden_states_103 = conv(dilations = hidden_states_103_dilations_0, groups = hidden_states_103_groups_0, pad = hidden_states_103_pad_0, pad_type = hidden_states_103_pad_type_0, strides = hidden_states_103_strides_0, weight = model_model_layers_12_mlp_down_proj_weight_palettized, x = input_181)[name = string("hidden_states_103")]; tensor var_3209_axes_0 = const()[name = string("op_3209_axes_0"), val = tensor([2])]; tensor var_3209 = squeeze(axes = var_3209_axes_0, x = hidden_states_103)[name = string("op_3209")]; tensor var_3210 = const()[name = string("op_3210"), val = tensor([0, 2, 1])]; tensor var_3211 = transpose(perm = var_3210, x = var_3209)[name = string("transpose_19")]; tensor hidden_states_105_cast_fp16 = add(x = hidden_states_101_cast_fp16, y = var_3211)[name = string("hidden_states_105_cast_fp16")]; fp16 const_235_promoted_to_fp16 = const()[name = string("const_235_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3214_cast_fp16 = mul(x = hidden_states_105_cast_fp16, y = const_235_promoted_to_fp16)[name = string("op_3214_cast_fp16")]; bool input_183_interleave_0 = const()[name = string("input_183_interleave_0"), val = bool(false)]; tensor input_183_cast_fp16 = concat(axis = var_73, interleave = input_183_interleave_0, values = (hidden_states_105_cast_fp16, var_3214_cast_fp16))[name = string("input_183_cast_fp16")]; tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_76_to_fp16, x = input_183_cast_fp16)[name = string("normed_105_cast_fp16")]; tensor normed_107_begin_0 = const()[name = string("normed_107_begin_0"), val = tensor([0, 0, 0])]; tensor normed_107_end_0 = const()[name = string("normed_107_end_0"), val = tensor([1, 64, 2048])]; tensor normed_107_end_mask_0 = const()[name = string("normed_107_end_mask_0"), val = tensor([true, true, false])]; tensor normed_107_cast_fp16 = slice_by_index(begin = normed_107_begin_0, end = normed_107_end_0, end_mask = normed_107_end_mask_0, x = normed_105_cast_fp16)[name = string("normed_107_cast_fp16")]; tensor const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483954176)))]; tensor hidden_states_107_cast_fp16 = mul(x = normed_107_cast_fp16, y = const_238_promoted_to_fp16)[name = string("hidden_states_107_cast_fp16")]; tensor var_3229 = const()[name = string("op_3229"), val = tensor([0, 2, 1])]; tensor var_3231_axes_0 = const()[name = string("op_3231_axes_0"), val = tensor([2])]; tensor var_3230_cast_fp16 = transpose(perm = var_3229, x = hidden_states_107_cast_fp16)[name = string("transpose_18")]; tensor var_3231_cast_fp16 = expand_dims(axes = var_3231_axes_0, x = var_3230_cast_fp16)[name = string("op_3231_cast_fp16")]; string query_states_53_pad_type_0 = const()[name = string("query_states_53_pad_type_0"), val = string("valid")]; tensor query_states_53_strides_0 = const()[name = string("query_states_53_strides_0"), val = tensor([1, 1])]; tensor query_states_53_pad_0 = const()[name = string("query_states_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_53_dilations_0 = const()[name = string("query_states_53_dilations_0"), val = tensor([1, 1])]; int32 query_states_53_groups_0 = const()[name = string("query_states_53_groups_0"), val = int32(1)]; tensor query_states_53 = conv(dilations = query_states_53_dilations_0, groups = query_states_53_groups_0, pad = query_states_53_pad_0, pad_type = query_states_53_pad_type_0, strides = query_states_53_strides_0, weight = model_model_layers_13_self_attn_q_proj_weight_palettized, x = var_3231_cast_fp16)[name = string("query_states_53")]; string key_states_79_pad_type_0 = const()[name = string("key_states_79_pad_type_0"), val = string("valid")]; tensor key_states_79_strides_0 = const()[name = string("key_states_79_strides_0"), val = tensor([1, 1])]; tensor key_states_79_pad_0 = const()[name = string("key_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_79_dilations_0 = const()[name = string("key_states_79_dilations_0"), val = tensor([1, 1])]; int32 key_states_79_groups_0 = const()[name = string("key_states_79_groups_0"), val = int32(1)]; tensor key_states_79 = conv(dilations = key_states_79_dilations_0, groups = key_states_79_groups_0, pad = key_states_79_pad_0, pad_type = key_states_79_pad_type_0, strides = key_states_79_strides_0, weight = model_model_layers_13_self_attn_k_proj_weight_palettized, x = var_3231_cast_fp16)[name = string("key_states_79")]; string value_states_79_pad_type_0 = const()[name = string("value_states_79_pad_type_0"), val = string("valid")]; tensor value_states_79_strides_0 = const()[name = string("value_states_79_strides_0"), val = tensor([1, 1])]; tensor value_states_79_pad_0 = const()[name = string("value_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_79_dilations_0 = const()[name = string("value_states_79_dilations_0"), val = tensor([1, 1])]; int32 value_states_79_groups_0 = const()[name = string("value_states_79_groups_0"), val = int32(1)]; tensor value_states_79 = conv(dilations = value_states_79_dilations_0, groups = value_states_79_groups_0, pad = value_states_79_pad_0, pad_type = value_states_79_pad_type_0, strides = value_states_79_strides_0, weight = model_model_layers_13_self_attn_v_proj_weight_palettized, x = var_3231_cast_fp16)[name = string("value_states_79")]; tensor var_3251 = const()[name = string("op_3251"), val = tensor([1, 32, 64, 64])]; tensor var_3252 = reshape(shape = var_3251, x = query_states_53)[name = string("op_3252")]; tensor var_3253 = const()[name = string("op_3253"), val = tensor([0, 1, 3, 2])]; tensor var_3255 = const()[name = string("op_3255"), val = tensor([1, 8, 64, 64])]; tensor var_3256 = reshape(shape = var_3255, x = key_states_79)[name = string("op_3256")]; tensor var_3257 = const()[name = string("op_3257"), val = tensor([0, 1, 3, 2])]; tensor var_3259 = const()[name = string("op_3259"), val = tensor([1, 8, 64, 64])]; tensor var_3260 = reshape(shape = var_3259, x = value_states_79)[name = string("op_3260")]; tensor var_3261 = const()[name = string("op_3261"), val = tensor([0, 1, 3, 2])]; tensor x1_53_begin_0 = const()[name = string("x1_53_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_53_end_0 = const()[name = string("x1_53_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_53_end_mask_0 = const()[name = string("x1_53_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_365 = transpose(perm = var_3253, x = var_3252)[name = string("transpose_17")]; tensor x1_53 = slice_by_index(begin = x1_53_begin_0, end = x1_53_end_0, end_mask = x1_53_end_mask_0, x = x_365)[name = string("x1_53")]; tensor x2_53_begin_0 = const()[name = string("x2_53_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_53_end_0 = const()[name = string("x2_53_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_53_end_mask_0 = const()[name = string("x2_53_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_53 = slice_by_index(begin = x2_53_begin_0, end = x2_53_end_0, end_mask = x2_53_end_mask_0, x = x_365)[name = string("x2_53")]; tensor var_3279 = mul(x = x1_53, y = cos_7)[name = string("op_3279")]; tensor var_3280 = mul(x = x2_53, y = sin_7)[name = string("op_3280")]; tensor var_3281 = sub(x = var_3279, y = var_3280)[name = string("op_3281")]; tensor var_3282 = mul(x = x2_53, y = cos_7)[name = string("op_3282")]; tensor var_3283 = mul(x = x1_53, y = sin_7)[name = string("op_3283")]; tensor var_3284 = add(x = var_3282, y = var_3283)[name = string("op_3284")]; bool rotated_53_interleave_0 = const()[name = string("rotated_53_interleave_0"), val = bool(false)]; tensor rotated_53 = concat(axis = var_73, interleave = rotated_53_interleave_0, values = (var_3281, var_3284))[name = string("rotated_53")]; tensor x1_55_begin_0 = const()[name = string("x1_55_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_55_end_0 = const()[name = string("x1_55_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_55_end_mask_0 = const()[name = string("x1_55_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_369 = transpose(perm = var_3257, x = var_3256)[name = string("transpose_16")]; tensor x1_55 = slice_by_index(begin = x1_55_begin_0, end = x1_55_end_0, end_mask = x1_55_end_mask_0, x = x_369)[name = string("x1_55")]; tensor x2_55_begin_0 = const()[name = string("x2_55_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_55_end_0 = const()[name = string("x2_55_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_55_end_mask_0 = const()[name = string("x2_55_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_55 = slice_by_index(begin = x2_55_begin_0, end = x2_55_end_0, end_mask = x2_55_end_mask_0, x = x_369)[name = string("x2_55")]; tensor var_3300 = mul(x = x1_55, y = cos_7)[name = string("op_3300")]; tensor var_3301 = mul(x = x2_55, y = sin_7)[name = string("op_3301")]; tensor var_3302 = sub(x = var_3300, y = var_3301)[name = string("op_3302")]; tensor var_3303 = mul(x = x2_55, y = cos_7)[name = string("op_3303")]; tensor var_3304 = mul(x = x1_55, y = sin_7)[name = string("op_3304")]; tensor var_3305 = add(x = var_3303, y = var_3304)[name = string("op_3305")]; bool rotated_55_interleave_0 = const()[name = string("rotated_55_interleave_0"), val = bool(false)]; tensor rotated_55 = concat(axis = var_73, interleave = rotated_55_interleave_0, values = (var_3302, var_3305))[name = string("rotated_55")]; tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; int32 concat_236_axis_0 = const()[name = string("concat_236_axis_0"), val = int32(0)]; bool concat_236_interleave_0 = const()[name = string("concat_236_interleave_0"), val = bool(false)]; tensor concat_236 = concat(axis = concat_236_axis_0, interleave = concat_236_interleave_0, values = (expand_dims_156, expand_dims_157, current_pos, expand_dims_159))[name = string("concat_236")]; tensor concat_237_values1_0 = const()[name = string("concat_237_values1_0"), val = tensor([0])]; tensor concat_237_values3_0 = const()[name = string("concat_237_values3_0"), val = tensor([0])]; int32 concat_237_axis_0 = const()[name = string("concat_237_axis_0"), val = int32(0)]; bool concat_237_interleave_0 = const()[name = string("concat_237_interleave_0"), val = bool(false)]; tensor concat_237 = concat(axis = concat_237_axis_0, interleave = concat_237_interleave_0, values = (expand_dims_160, concat_237_values1_0, var_597, concat_237_values3_0))[name = string("concat_237")]; tensor model_model_kv_cache_0_internal_tensor_assign_27_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_236, begin_mask = model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0, end = concat_237, end_mask = model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_27_stride_0, update = rotated_55, x = coreml_update_state_57)[name = string("model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_122_write_state")]; tensor coreml_update_state_58 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_122")]; tensor expand_dims_162 = const()[name = string("expand_dims_162"), val = tensor([29])]; tensor expand_dims_163 = const()[name = string("expand_dims_163"), val = tensor([0])]; tensor expand_dims_165 = const()[name = string("expand_dims_165"), val = tensor([0])]; tensor expand_dims_166 = const()[name = string("expand_dims_166"), val = tensor([30])]; int32 concat_240_axis_0 = const()[name = string("concat_240_axis_0"), val = int32(0)]; bool concat_240_interleave_0 = const()[name = string("concat_240_interleave_0"), val = bool(false)]; tensor concat_240 = concat(axis = concat_240_axis_0, interleave = concat_240_interleave_0, values = (expand_dims_162, expand_dims_163, current_pos, expand_dims_165))[name = string("concat_240")]; tensor concat_241_values1_0 = const()[name = string("concat_241_values1_0"), val = tensor([0])]; tensor concat_241_values3_0 = const()[name = string("concat_241_values3_0"), val = tensor([0])]; int32 concat_241_axis_0 = const()[name = string("concat_241_axis_0"), val = int32(0)]; bool concat_241_interleave_0 = const()[name = string("concat_241_interleave_0"), val = bool(false)]; tensor concat_241 = concat(axis = concat_241_axis_0, interleave = concat_241_interleave_0, values = (expand_dims_166, concat_241_values1_0, var_597, concat_241_values3_0))[name = string("concat_241")]; tensor model_model_kv_cache_0_internal_tensor_assign_28_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_81 = transpose(perm = var_3261, x = var_3260)[name = string("transpose_15")]; tensor model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_240, begin_mask = model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0, end = concat_241, end_mask = model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_28_stride_0, update = value_states_81, x = coreml_update_state_58)[name = string("model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_123_write_state")]; tensor coreml_update_state_59 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_123")]; tensor var_3328_begin_0 = const()[name = string("op_3328_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_3328_end_0 = const()[name = string("op_3328_end_0"), val = tensor([14, 8, 4096, 64])]; tensor var_3328_end_mask_0 = const()[name = string("op_3328_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3328_cast_fp16 = slice_by_index(begin = var_3328_begin_0, end = var_3328_end_0, end_mask = var_3328_end_mask_0, x = coreml_update_state_59)[name = string("op_3328_cast_fp16")]; tensor K_layer_cache_27_axes_0 = const()[name = string("K_layer_cache_27_axes_0"), val = tensor([0])]; tensor K_layer_cache_27_cast_fp16 = squeeze(axes = K_layer_cache_27_axes_0, x = var_3328_cast_fp16)[name = string("K_layer_cache_27_cast_fp16")]; tensor var_3330_begin_0 = const()[name = string("op_3330_begin_0"), val = tensor([29, 0, 0, 0])]; tensor var_3330_end_0 = const()[name = string("op_3330_end_0"), val = tensor([30, 8, 4096, 64])]; tensor var_3330_end_mask_0 = const()[name = string("op_3330_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3330_cast_fp16 = slice_by_index(begin = var_3330_begin_0, end = var_3330_end_0, end_mask = var_3330_end_mask_0, x = coreml_update_state_59)[name = string("op_3330_cast_fp16")]; tensor V_layer_cache_27_axes_0 = const()[name = string("V_layer_cache_27_axes_0"), val = tensor([0])]; tensor V_layer_cache_27_cast_fp16 = squeeze(axes = V_layer_cache_27_axes_0, x = var_3330_cast_fp16)[name = string("V_layer_cache_27_cast_fp16")]; tensor x_375_axes_0 = const()[name = string("x_375_axes_0"), val = tensor([1])]; tensor x_375_cast_fp16 = expand_dims(axes = x_375_axes_0, x = K_layer_cache_27_cast_fp16)[name = string("x_375_cast_fp16")]; tensor var_3339 = const()[name = string("op_3339"), val = tensor([1, 4, 1, 1])]; tensor x_377_cast_fp16 = tile(reps = var_3339, x = x_375_cast_fp16)[name = string("x_377_cast_fp16")]; tensor var_3343 = const()[name = string("op_3343"), val = tensor([1, -1, 4096, 64])]; tensor var_3344_cast_fp16 = reshape(shape = var_3343, x = x_377_cast_fp16)[name = string("op_3344_cast_fp16")]; tensor x_381_axes_0 = const()[name = string("x_381_axes_0"), val = tensor([1])]; tensor x_381_cast_fp16 = expand_dims(axes = x_381_axes_0, x = V_layer_cache_27_cast_fp16)[name = string("x_381_cast_fp16")]; tensor var_3346 = const()[name = string("op_3346"), val = tensor([1, 4, 1, 1])]; tensor x_383_cast_fp16 = tile(reps = var_3346, x = x_381_cast_fp16)[name = string("x_383_cast_fp16")]; bool var_3353_transpose_x_0 = const()[name = string("op_3353_transpose_x_0"), val = bool(false)]; bool var_3353_transpose_y_0 = const()[name = string("op_3353_transpose_y_0"), val = bool(true)]; tensor var_3353_cast_fp16 = matmul(transpose_x = var_3353_transpose_x_0, transpose_y = var_3353_transpose_y_0, x = rotated_53, y = var_3344_cast_fp16)[name = string("op_3353_cast_fp16")]; fp16 var_3354_to_fp16 = const()[name = string("op_3354_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_27_cast_fp16 = mul(x = var_3353_cast_fp16, y = var_3354_to_fp16)[name = string("attn_weights_27_cast_fp16")]; tensor x_385_cast_fp16 = add(x = attn_weights_27_cast_fp16, y = causal_mask)[name = string("x_385_cast_fp16")]; tensor reduce_max_13_axes_0 = const()[name = string("reduce_max_13_axes_0"), val = tensor([-1])]; bool reduce_max_13_keep_dims_0 = const()[name = string("reduce_max_13_keep_dims_0"), val = bool(true)]; tensor reduce_max_13_cast_fp16 = reduce_max(axes = reduce_max_13_axes_0, keep_dims = reduce_max_13_keep_dims_0, x = x_385_cast_fp16)[name = string("reduce_max_13_cast_fp16")]; tensor x_387_cast_fp16 = sub(x = x_385_cast_fp16, y = reduce_max_13_cast_fp16)[name = string("x_387_cast_fp16")]; tensor exp_x_27_cast_fp16 = exp(x = x_387_cast_fp16)[name = string("exp_x_27_cast_fp16")]; tensor var_3365_axes_0 = const()[name = string("op_3365_axes_0"), val = tensor([-1])]; bool var_3365_keep_dims_0 = const()[name = string("op_3365_keep_dims_0"), val = bool(true)]; tensor var_3365_cast_fp16 = reduce_sum(axes = var_3365_axes_0, keep_dims = var_3365_keep_dims_0, x = exp_x_27_cast_fp16)[name = string("op_3365_cast_fp16")]; tensor var_3366_cast_fp16 = real_div(x = exp_x_27_cast_fp16, y = var_3365_cast_fp16)[name = string("op_3366_cast_fp16")]; tensor concat_246 = const()[name = string("concat_246"), val = tensor([32, 64, 4096])]; tensor reshape_39_cast_fp16 = reshape(shape = concat_246, x = var_3366_cast_fp16)[name = string("reshape_39_cast_fp16")]; tensor concat_247 = const()[name = string("concat_247"), val = tensor([32, 4096, 64])]; tensor reshape_40_cast_fp16 = reshape(shape = concat_247, x = x_383_cast_fp16)[name = string("reshape_40_cast_fp16")]; bool matmul_13_transpose_x_0 = const()[name = string("matmul_13_transpose_x_0"), val = bool(false)]; bool matmul_13_transpose_y_0 = const()[name = string("matmul_13_transpose_y_0"), val = bool(false)]; tensor matmul_13_cast_fp16 = matmul(transpose_x = matmul_13_transpose_x_0, transpose_y = matmul_13_transpose_y_0, x = reshape_39_cast_fp16, y = reshape_40_cast_fp16)[name = string("matmul_13_cast_fp16")]; tensor concat_251 = const()[name = string("concat_251"), val = tensor([1, 32, 64, 64])]; tensor reshape_41_cast_fp16 = reshape(shape = concat_251, x = matmul_13_cast_fp16)[name = string("reshape_41_cast_fp16")]; tensor var_3369_perm_0 = const()[name = string("op_3369_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3371 = const()[name = string("op_3371"), val = tensor([1, 64, 2048])]; tensor var_3369_cast_fp16 = transpose(perm = var_3369_perm_0, x = reshape_41_cast_fp16)[name = string("transpose_14")]; tensor input_187_cast_fp16 = reshape(shape = var_3371, x = var_3369_cast_fp16)[name = string("input_187_cast_fp16")]; tensor model_model_layers_13_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483958336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(486055552))))[name = string("model_model_layers_13_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_13_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_13_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_187_cast_fp16)[name = string("linear_13_cast_fp16")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = linear_13_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; fp16 const_249_promoted_to_fp16 = const()[name = string("const_249_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3377_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_249_promoted_to_fp16)[name = string("op_3377_cast_fp16")]; bool input_189_interleave_0 = const()[name = string("input_189_interleave_0"), val = bool(false)]; tensor input_189_cast_fp16 = concat(axis = var_73, interleave = input_189_interleave_0, values = (hidden_states_109_cast_fp16, var_3377_cast_fp16))[name = string("input_189_cast_fp16")]; tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_76_to_fp16, x = input_189_cast_fp16)[name = string("normed_109_cast_fp16")]; tensor normed_111_begin_0 = const()[name = string("normed_111_begin_0"), val = tensor([0, 0, 0])]; tensor normed_111_end_0 = const()[name = string("normed_111_end_0"), val = tensor([1, 64, 2048])]; tensor normed_111_end_mask_0 = const()[name = string("normed_111_end_mask_0"), val = tensor([true, true, false])]; tensor normed_111_cast_fp16 = slice_by_index(begin = normed_111_begin_0, end = normed_111_end_0, end_mask = normed_111_end_mask_0, x = normed_109_cast_fp16)[name = string("normed_111_cast_fp16")]; tensor const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(486063808)))]; tensor x_389_cast_fp16 = mul(x = normed_111_cast_fp16, y = const_252_promoted_to_fp16)[name = string("x_389_cast_fp16")]; tensor var_3395 = const()[name = string("op_3395"), val = tensor([0, 2, 1])]; tensor input_191_axes_0 = const()[name = string("input_191_axes_0"), val = tensor([2])]; tensor var_3396 = transpose(perm = var_3395, x = x_389_cast_fp16)[name = string("transpose_13")]; tensor input_191 = expand_dims(axes = input_191_axes_0, x = var_3396)[name = string("input_191")]; string input_193_pad_type_0 = const()[name = string("input_193_pad_type_0"), val = string("valid")]; tensor input_193_strides_0 = const()[name = string("input_193_strides_0"), val = tensor([1, 1])]; tensor input_193_pad_0 = const()[name = string("input_193_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_193_dilations_0 = const()[name = string("input_193_dilations_0"), val = tensor([1, 1])]; int32 input_193_groups_0 = const()[name = string("input_193_groups_0"), val = int32(1)]; tensor input_193 = conv(dilations = input_193_dilations_0, groups = input_193_groups_0, pad = input_193_pad_0, pad_type = input_193_pad_type_0, strides = input_193_strides_0, weight = model_model_layers_13_mlp_gate_proj_weight_palettized, x = input_191)[name = string("input_193")]; string up_states_27_pad_type_0 = const()[name = string("up_states_27_pad_type_0"), val = string("valid")]; tensor up_states_27_strides_0 = const()[name = string("up_states_27_strides_0"), val = tensor([1, 1])]; tensor up_states_27_pad_0 = const()[name = string("up_states_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_27_dilations_0 = const()[name = string("up_states_27_dilations_0"), val = tensor([1, 1])]; int32 up_states_27_groups_0 = const()[name = string("up_states_27_groups_0"), val = int32(1)]; tensor up_states_27 = conv(dilations = up_states_27_dilations_0, groups = up_states_27_groups_0, pad = up_states_27_pad_0, pad_type = up_states_27_pad_type_0, strides = up_states_27_strides_0, weight = model_model_layers_13_mlp_up_proj_weight_palettized, x = input_191)[name = string("up_states_27")]; tensor gate_states_27 = silu(x = input_193)[name = string("gate_states_27")]; tensor input_195 = mul(x = gate_states_27, y = up_states_27)[name = string("input_195")]; string hidden_states_111_pad_type_0 = const()[name = string("hidden_states_111_pad_type_0"), val = string("valid")]; tensor hidden_states_111_strides_0 = const()[name = string("hidden_states_111_strides_0"), val = tensor([1, 1])]; tensor hidden_states_111_pad_0 = const()[name = string("hidden_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_111_dilations_0 = const()[name = string("hidden_states_111_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_111_groups_0 = const()[name = string("hidden_states_111_groups_0"), val = int32(1)]; tensor hidden_states_111 = conv(dilations = hidden_states_111_dilations_0, groups = hidden_states_111_groups_0, pad = hidden_states_111_pad_0, pad_type = hidden_states_111_pad_type_0, strides = hidden_states_111_strides_0, weight = model_model_layers_13_mlp_down_proj_weight_palettized, x = input_195)[name = string("hidden_states_111")]; tensor var_3418_axes_0 = const()[name = string("op_3418_axes_0"), val = tensor([2])]; tensor var_3418 = squeeze(axes = var_3418_axes_0, x = hidden_states_111)[name = string("op_3418")]; tensor var_3419 = const()[name = string("op_3419"), val = tensor([0, 2, 1])]; tensor var_3420 = transpose(perm = var_3419, x = var_3418)[name = string("transpose_12")]; tensor hidden_states_113_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = var_3420)[name = string("hidden_states_113_cast_fp16")]; fp16 const_253_promoted_to_fp16 = const()[name = string("const_253_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3423_cast_fp16 = mul(x = hidden_states_113_cast_fp16, y = const_253_promoted_to_fp16)[name = string("op_3423_cast_fp16")]; bool input_197_interleave_0 = const()[name = string("input_197_interleave_0"), val = bool(false)]; tensor input_197_cast_fp16 = concat(axis = var_73, interleave = input_197_interleave_0, values = (hidden_states_113_cast_fp16, var_3423_cast_fp16))[name = string("input_197_cast_fp16")]; tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_76_to_fp16, x = input_197_cast_fp16)[name = string("normed_113_cast_fp16")]; tensor normed_115_begin_0 = const()[name = string("normed_115_begin_0"), val = tensor([0, 0, 0])]; tensor normed_115_end_0 = const()[name = string("normed_115_end_0"), val = tensor([1, 64, 2048])]; tensor normed_115_end_mask_0 = const()[name = string("normed_115_end_mask_0"), val = tensor([true, true, false])]; tensor normed_115_cast_fp16 = slice_by_index(begin = normed_115_begin_0, end = normed_115_end_0, end_mask = normed_115_end_mask_0, x = normed_113_cast_fp16)[name = string("normed_115_cast_fp16")]; tensor const_256_promoted_to_fp16 = const()[name = string("const_256_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(486067968)))]; tensor hidden_states_115_cast_fp16 = mul(x = normed_115_cast_fp16, y = const_256_promoted_to_fp16)[name = string("hidden_states_115_cast_fp16")]; tensor var_3438 = const()[name = string("op_3438"), val = tensor([0, 2, 1])]; tensor var_3440_axes_0 = const()[name = string("op_3440_axes_0"), val = tensor([2])]; tensor var_3439_cast_fp16 = transpose(perm = var_3438, x = hidden_states_115_cast_fp16)[name = string("transpose_11")]; tensor var_3440_cast_fp16 = expand_dims(axes = var_3440_axes_0, x = var_3439_cast_fp16)[name = string("op_3440_cast_fp16")]; string query_states_57_pad_type_0 = const()[name = string("query_states_57_pad_type_0"), val = string("valid")]; tensor query_states_57_strides_0 = const()[name = string("query_states_57_strides_0"), val = tensor([1, 1])]; tensor query_states_57_pad_0 = const()[name = string("query_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_57_dilations_0 = const()[name = string("query_states_57_dilations_0"), val = tensor([1, 1])]; int32 query_states_57_groups_0 = const()[name = string("query_states_57_groups_0"), val = int32(1)]; tensor query_states_57 = conv(dilations = query_states_57_dilations_0, groups = query_states_57_groups_0, pad = query_states_57_pad_0, pad_type = query_states_57_pad_type_0, strides = query_states_57_strides_0, weight = model_model_layers_14_self_attn_q_proj_weight_palettized, x = var_3440_cast_fp16)[name = string("query_states_57")]; string key_states_85_pad_type_0 = const()[name = string("key_states_85_pad_type_0"), val = string("valid")]; tensor key_states_85_strides_0 = const()[name = string("key_states_85_strides_0"), val = tensor([1, 1])]; tensor key_states_85_pad_0 = const()[name = string("key_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_85_dilations_0 = const()[name = string("key_states_85_dilations_0"), val = tensor([1, 1])]; int32 key_states_85_groups_0 = const()[name = string("key_states_85_groups_0"), val = int32(1)]; tensor key_states_85 = conv(dilations = key_states_85_dilations_0, groups = key_states_85_groups_0, pad = key_states_85_pad_0, pad_type = key_states_85_pad_type_0, strides = key_states_85_strides_0, weight = model_model_layers_14_self_attn_k_proj_weight_palettized, x = var_3440_cast_fp16)[name = string("key_states_85")]; string value_states_85_pad_type_0 = const()[name = string("value_states_85_pad_type_0"), val = string("valid")]; tensor value_states_85_strides_0 = const()[name = string("value_states_85_strides_0"), val = tensor([1, 1])]; tensor value_states_85_pad_0 = const()[name = string("value_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_85_dilations_0 = const()[name = string("value_states_85_dilations_0"), val = tensor([1, 1])]; int32 value_states_85_groups_0 = const()[name = string("value_states_85_groups_0"), val = int32(1)]; tensor value_states_85 = conv(dilations = value_states_85_dilations_0, groups = value_states_85_groups_0, pad = value_states_85_pad_0, pad_type = value_states_85_pad_type_0, strides = value_states_85_strides_0, weight = model_model_layers_14_self_attn_v_proj_weight_palettized, x = var_3440_cast_fp16)[name = string("value_states_85")]; tensor var_3460 = const()[name = string("op_3460"), val = tensor([1, 32, 64, 64])]; tensor var_3461 = reshape(shape = var_3460, x = query_states_57)[name = string("op_3461")]; tensor var_3462 = const()[name = string("op_3462"), val = tensor([0, 1, 3, 2])]; tensor var_3464 = const()[name = string("op_3464"), val = tensor([1, 8, 64, 64])]; tensor var_3465 = reshape(shape = var_3464, x = key_states_85)[name = string("op_3465")]; tensor var_3466 = const()[name = string("op_3466"), val = tensor([0, 1, 3, 2])]; tensor var_3468 = const()[name = string("op_3468"), val = tensor([1, 8, 64, 64])]; tensor var_3469 = reshape(shape = var_3468, x = value_states_85)[name = string("op_3469")]; tensor var_3470 = const()[name = string("op_3470"), val = tensor([0, 1, 3, 2])]; tensor x1_57_begin_0 = const()[name = string("x1_57_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_57_end_0 = const()[name = string("x1_57_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_57_end_mask_0 = const()[name = string("x1_57_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_393 = transpose(perm = var_3462, x = var_3461)[name = string("transpose_10")]; tensor x1_57 = slice_by_index(begin = x1_57_begin_0, end = x1_57_end_0, end_mask = x1_57_end_mask_0, x = x_393)[name = string("x1_57")]; tensor x2_57_begin_0 = const()[name = string("x2_57_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_57_end_0 = const()[name = string("x2_57_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_57_end_mask_0 = const()[name = string("x2_57_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_57 = slice_by_index(begin = x2_57_begin_0, end = x2_57_end_0, end_mask = x2_57_end_mask_0, x = x_393)[name = string("x2_57")]; tensor var_3488 = mul(x = x1_57, y = cos_7)[name = string("op_3488")]; tensor var_3489 = mul(x = x2_57, y = sin_7)[name = string("op_3489")]; tensor var_3490 = sub(x = var_3488, y = var_3489)[name = string("op_3490")]; tensor var_3491 = mul(x = x2_57, y = cos_7)[name = string("op_3491")]; tensor var_3492 = mul(x = x1_57, y = sin_7)[name = string("op_3492")]; tensor var_3493 = add(x = var_3491, y = var_3492)[name = string("op_3493")]; bool rotated_57_interleave_0 = const()[name = string("rotated_57_interleave_0"), val = bool(false)]; tensor rotated_57 = concat(axis = var_73, interleave = rotated_57_interleave_0, values = (var_3490, var_3493))[name = string("rotated_57")]; tensor x1_59_begin_0 = const()[name = string("x1_59_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_59_end_0 = const()[name = string("x1_59_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_59_end_mask_0 = const()[name = string("x1_59_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_397 = transpose(perm = var_3466, x = var_3465)[name = string("transpose_9")]; tensor x1_59 = slice_by_index(begin = x1_59_begin_0, end = x1_59_end_0, end_mask = x1_59_end_mask_0, x = x_397)[name = string("x1_59")]; tensor x2_59_begin_0 = const()[name = string("x2_59_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_59_end_0 = const()[name = string("x2_59_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_59_end_mask_0 = const()[name = string("x2_59_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_59 = slice_by_index(begin = x2_59_begin_0, end = x2_59_end_0, end_mask = x2_59_end_mask_0, x = x_397)[name = string("x2_59")]; tensor var_3509 = mul(x = x1_59, y = cos_7)[name = string("op_3509")]; tensor var_3510 = mul(x = x2_59, y = sin_7)[name = string("op_3510")]; tensor var_3511 = sub(x = var_3509, y = var_3510)[name = string("op_3511")]; tensor var_3512 = mul(x = x2_59, y = cos_7)[name = string("op_3512")]; tensor var_3513 = mul(x = x1_59, y = sin_7)[name = string("op_3513")]; tensor var_3514 = add(x = var_3512, y = var_3513)[name = string("op_3514")]; bool rotated_59_interleave_0 = const()[name = string("rotated_59_interleave_0"), val = bool(false)]; tensor rotated_59 = concat(axis = var_73, interleave = rotated_59_interleave_0, values = (var_3511, var_3514))[name = string("rotated_59")]; tensor expand_dims_168 = const()[name = string("expand_dims_168"), val = tensor([14])]; tensor expand_dims_169 = const()[name = string("expand_dims_169"), val = tensor([0])]; tensor expand_dims_171 = const()[name = string("expand_dims_171"), val = tensor([0])]; tensor expand_dims_172 = const()[name = string("expand_dims_172"), val = tensor([15])]; int32 concat_254_axis_0 = const()[name = string("concat_254_axis_0"), val = int32(0)]; bool concat_254_interleave_0 = const()[name = string("concat_254_interleave_0"), val = bool(false)]; tensor concat_254 = concat(axis = concat_254_axis_0, interleave = concat_254_interleave_0, values = (expand_dims_168, expand_dims_169, current_pos, expand_dims_171))[name = string("concat_254")]; tensor concat_255_values1_0 = const()[name = string("concat_255_values1_0"), val = tensor([0])]; tensor concat_255_values3_0 = const()[name = string("concat_255_values3_0"), val = tensor([0])]; int32 concat_255_axis_0 = const()[name = string("concat_255_axis_0"), val = int32(0)]; bool concat_255_interleave_0 = const()[name = string("concat_255_interleave_0"), val = bool(false)]; tensor concat_255 = concat(axis = concat_255_axis_0, interleave = concat_255_interleave_0, values = (expand_dims_172, concat_255_values1_0, var_597, concat_255_values3_0))[name = string("concat_255")]; tensor model_model_kv_cache_0_internal_tensor_assign_29_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_29_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_29_cast_fp16 = slice_update(begin = concat_254, begin_mask = model_model_kv_cache_0_internal_tensor_assign_29_begin_mask_0, end = concat_255, end_mask = model_model_kv_cache_0_internal_tensor_assign_29_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_29_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_29_stride_0, update = rotated_59, x = coreml_update_state_59)[name = string("model_model_kv_cache_0_internal_tensor_assign_29_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_29_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_124_write_state")]; tensor coreml_update_state_60 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_124")]; tensor expand_dims_174 = const()[name = string("expand_dims_174"), val = tensor([30])]; tensor expand_dims_175 = const()[name = string("expand_dims_175"), val = tensor([0])]; tensor expand_dims_177 = const()[name = string("expand_dims_177"), val = tensor([0])]; tensor expand_dims_178 = const()[name = string("expand_dims_178"), val = tensor([31])]; int32 concat_258_axis_0 = const()[name = string("concat_258_axis_0"), val = int32(0)]; bool concat_258_interleave_0 = const()[name = string("concat_258_interleave_0"), val = bool(false)]; tensor concat_258 = concat(axis = concat_258_axis_0, interleave = concat_258_interleave_0, values = (expand_dims_174, expand_dims_175, current_pos, expand_dims_177))[name = string("concat_258")]; tensor concat_259_values1_0 = const()[name = string("concat_259_values1_0"), val = tensor([0])]; tensor concat_259_values3_0 = const()[name = string("concat_259_values3_0"), val = tensor([0])]; int32 concat_259_axis_0 = const()[name = string("concat_259_axis_0"), val = int32(0)]; bool concat_259_interleave_0 = const()[name = string("concat_259_interleave_0"), val = bool(false)]; tensor concat_259 = concat(axis = concat_259_axis_0, interleave = concat_259_interleave_0, values = (expand_dims_178, concat_259_values1_0, var_597, concat_259_values3_0))[name = string("concat_259")]; tensor model_model_kv_cache_0_internal_tensor_assign_30_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_30_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_30_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_87 = transpose(perm = var_3470, x = var_3469)[name = string("transpose_8")]; tensor model_model_kv_cache_0_internal_tensor_assign_30_cast_fp16 = slice_update(begin = concat_258, begin_mask = model_model_kv_cache_0_internal_tensor_assign_30_begin_mask_0, end = concat_259, end_mask = model_model_kv_cache_0_internal_tensor_assign_30_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_30_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_30_stride_0, update = value_states_87, x = coreml_update_state_60)[name = string("model_model_kv_cache_0_internal_tensor_assign_30_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_30_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_125_write_state")]; tensor coreml_update_state_61 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_125")]; tensor var_3537_begin_0 = const()[name = string("op_3537_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_3537_end_0 = const()[name = string("op_3537_end_0"), val = tensor([15, 8, 4096, 64])]; tensor var_3537_end_mask_0 = const()[name = string("op_3537_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3537_cast_fp16 = slice_by_index(begin = var_3537_begin_0, end = var_3537_end_0, end_mask = var_3537_end_mask_0, x = coreml_update_state_61)[name = string("op_3537_cast_fp16")]; tensor K_layer_cache_29_axes_0 = const()[name = string("K_layer_cache_29_axes_0"), val = tensor([0])]; tensor K_layer_cache_29_cast_fp16 = squeeze(axes = K_layer_cache_29_axes_0, x = var_3537_cast_fp16)[name = string("K_layer_cache_29_cast_fp16")]; tensor var_3539_begin_0 = const()[name = string("op_3539_begin_0"), val = tensor([30, 0, 0, 0])]; tensor var_3539_end_0 = const()[name = string("op_3539_end_0"), val = tensor([31, 8, 4096, 64])]; tensor var_3539_end_mask_0 = const()[name = string("op_3539_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3539_cast_fp16 = slice_by_index(begin = var_3539_begin_0, end = var_3539_end_0, end_mask = var_3539_end_mask_0, x = coreml_update_state_61)[name = string("op_3539_cast_fp16")]; tensor V_layer_cache_29_axes_0 = const()[name = string("V_layer_cache_29_axes_0"), val = tensor([0])]; tensor V_layer_cache_29_cast_fp16 = squeeze(axes = V_layer_cache_29_axes_0, x = var_3539_cast_fp16)[name = string("V_layer_cache_29_cast_fp16")]; tensor x_403_axes_0 = const()[name = string("x_403_axes_0"), val = tensor([1])]; tensor x_403_cast_fp16 = expand_dims(axes = x_403_axes_0, x = K_layer_cache_29_cast_fp16)[name = string("x_403_cast_fp16")]; tensor var_3548 = const()[name = string("op_3548"), val = tensor([1, 4, 1, 1])]; tensor x_405_cast_fp16 = tile(reps = var_3548, x = x_403_cast_fp16)[name = string("x_405_cast_fp16")]; tensor var_3552 = const()[name = string("op_3552"), val = tensor([1, -1, 4096, 64])]; tensor var_3553_cast_fp16 = reshape(shape = var_3552, x = x_405_cast_fp16)[name = string("op_3553_cast_fp16")]; tensor x_409_axes_0 = const()[name = string("x_409_axes_0"), val = tensor([1])]; tensor x_409_cast_fp16 = expand_dims(axes = x_409_axes_0, x = V_layer_cache_29_cast_fp16)[name = string("x_409_cast_fp16")]; tensor var_3555 = const()[name = string("op_3555"), val = tensor([1, 4, 1, 1])]; tensor x_411_cast_fp16 = tile(reps = var_3555, x = x_409_cast_fp16)[name = string("x_411_cast_fp16")]; bool var_3562_transpose_x_0 = const()[name = string("op_3562_transpose_x_0"), val = bool(false)]; bool var_3562_transpose_y_0 = const()[name = string("op_3562_transpose_y_0"), val = bool(true)]; tensor var_3562_cast_fp16 = matmul(transpose_x = var_3562_transpose_x_0, transpose_y = var_3562_transpose_y_0, x = rotated_57, y = var_3553_cast_fp16)[name = string("op_3562_cast_fp16")]; fp16 var_3563_to_fp16 = const()[name = string("op_3563_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_29_cast_fp16 = mul(x = var_3562_cast_fp16, y = var_3563_to_fp16)[name = string("attn_weights_29_cast_fp16")]; tensor x_413_cast_fp16 = add(x = attn_weights_29_cast_fp16, y = causal_mask)[name = string("x_413_cast_fp16")]; tensor reduce_max_14_axes_0 = const()[name = string("reduce_max_14_axes_0"), val = tensor([-1])]; bool reduce_max_14_keep_dims_0 = const()[name = string("reduce_max_14_keep_dims_0"), val = bool(true)]; tensor reduce_max_14_cast_fp16 = reduce_max(axes = reduce_max_14_axes_0, keep_dims = reduce_max_14_keep_dims_0, x = x_413_cast_fp16)[name = string("reduce_max_14_cast_fp16")]; tensor x_415_cast_fp16 = sub(x = x_413_cast_fp16, y = reduce_max_14_cast_fp16)[name = string("x_415_cast_fp16")]; tensor exp_x_29_cast_fp16 = exp(x = x_415_cast_fp16)[name = string("exp_x_29_cast_fp16")]; tensor var_3574_axes_0 = const()[name = string("op_3574_axes_0"), val = tensor([-1])]; bool var_3574_keep_dims_0 = const()[name = string("op_3574_keep_dims_0"), val = bool(true)]; tensor var_3574_cast_fp16 = reduce_sum(axes = var_3574_axes_0, keep_dims = var_3574_keep_dims_0, x = exp_x_29_cast_fp16)[name = string("op_3574_cast_fp16")]; tensor var_3575_cast_fp16 = real_div(x = exp_x_29_cast_fp16, y = var_3574_cast_fp16)[name = string("op_3575_cast_fp16")]; tensor concat_264 = const()[name = string("concat_264"), val = tensor([32, 64, 4096])]; tensor reshape_42_cast_fp16 = reshape(shape = concat_264, x = var_3575_cast_fp16)[name = string("reshape_42_cast_fp16")]; tensor concat_265 = const()[name = string("concat_265"), val = tensor([32, 4096, 64])]; tensor reshape_43_cast_fp16 = reshape(shape = concat_265, x = x_411_cast_fp16)[name = string("reshape_43_cast_fp16")]; bool matmul_14_transpose_x_0 = const()[name = string("matmul_14_transpose_x_0"), val = bool(false)]; bool matmul_14_transpose_y_0 = const()[name = string("matmul_14_transpose_y_0"), val = bool(false)]; tensor matmul_14_cast_fp16 = matmul(transpose_x = matmul_14_transpose_x_0, transpose_y = matmul_14_transpose_y_0, x = reshape_42_cast_fp16, y = reshape_43_cast_fp16)[name = string("matmul_14_cast_fp16")]; tensor concat_269 = const()[name = string("concat_269"), val = tensor([1, 32, 64, 64])]; tensor reshape_44_cast_fp16 = reshape(shape = concat_269, x = matmul_14_cast_fp16)[name = string("reshape_44_cast_fp16")]; tensor var_3578_perm_0 = const()[name = string("op_3578_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3580 = const()[name = string("op_3580"), val = tensor([1, 64, 2048])]; tensor var_3578_cast_fp16 = transpose(perm = var_3578_perm_0, x = reshape_44_cast_fp16)[name = string("transpose_7")]; tensor input_201_cast_fp16 = reshape(shape = var_3580, x = var_3578_cast_fp16)[name = string("input_201_cast_fp16")]; tensor model_model_layers_14_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(486072128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488169344))))[name = string("model_model_layers_14_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_14_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_14_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_201_cast_fp16)[name = string("linear_14_cast_fp16")]; tensor hidden_states_117_cast_fp16 = add(x = hidden_states_113_cast_fp16, y = linear_14_cast_fp16)[name = string("hidden_states_117_cast_fp16")]; fp16 const_267_promoted_to_fp16 = const()[name = string("const_267_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3586_cast_fp16 = mul(x = hidden_states_117_cast_fp16, y = const_267_promoted_to_fp16)[name = string("op_3586_cast_fp16")]; bool input_203_interleave_0 = const()[name = string("input_203_interleave_0"), val = bool(false)]; tensor input_203_cast_fp16 = concat(axis = var_73, interleave = input_203_interleave_0, values = (hidden_states_117_cast_fp16, var_3586_cast_fp16))[name = string("input_203_cast_fp16")]; tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_76_to_fp16, x = input_203_cast_fp16)[name = string("normed_117_cast_fp16")]; tensor normed_119_begin_0 = const()[name = string("normed_119_begin_0"), val = tensor([0, 0, 0])]; tensor normed_119_end_0 = const()[name = string("normed_119_end_0"), val = tensor([1, 64, 2048])]; tensor normed_119_end_mask_0 = const()[name = string("normed_119_end_mask_0"), val = tensor([true, true, false])]; tensor normed_119_cast_fp16 = slice_by_index(begin = normed_119_begin_0, end = normed_119_end_0, end_mask = normed_119_end_mask_0, x = normed_117_cast_fp16)[name = string("normed_119_cast_fp16")]; tensor const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488177600)))]; tensor x_417_cast_fp16 = mul(x = normed_119_cast_fp16, y = const_270_promoted_to_fp16)[name = string("x_417_cast_fp16")]; tensor var_3604 = const()[name = string("op_3604"), val = tensor([0, 2, 1])]; tensor input_205_axes_0 = const()[name = string("input_205_axes_0"), val = tensor([2])]; tensor var_3605 = transpose(perm = var_3604, x = x_417_cast_fp16)[name = string("transpose_6")]; tensor input_205 = expand_dims(axes = input_205_axes_0, x = var_3605)[name = string("input_205")]; string input_207_pad_type_0 = const()[name = string("input_207_pad_type_0"), val = string("valid")]; tensor input_207_strides_0 = const()[name = string("input_207_strides_0"), val = tensor([1, 1])]; tensor input_207_pad_0 = const()[name = string("input_207_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_207_dilations_0 = const()[name = string("input_207_dilations_0"), val = tensor([1, 1])]; int32 input_207_groups_0 = const()[name = string("input_207_groups_0"), val = int32(1)]; tensor input_207 = conv(dilations = input_207_dilations_0, groups = input_207_groups_0, pad = input_207_pad_0, pad_type = input_207_pad_type_0, strides = input_207_strides_0, weight = model_model_layers_14_mlp_gate_proj_weight_palettized, x = input_205)[name = string("input_207")]; string up_states_pad_type_0 = const()[name = string("up_states_pad_type_0"), val = string("valid")]; tensor up_states_strides_0 = const()[name = string("up_states_strides_0"), val = tensor([1, 1])]; tensor up_states_pad_0 = const()[name = string("up_states_pad_0"), val = tensor([0, 0, 0, 0])]; tensor up_states_dilations_0 = const()[name = string("up_states_dilations_0"), val = tensor([1, 1])]; int32 up_states_groups_0 = const()[name = string("up_states_groups_0"), val = int32(1)]; tensor up_states = conv(dilations = up_states_dilations_0, groups = up_states_groups_0, pad = up_states_pad_0, pad_type = up_states_pad_type_0, strides = up_states_strides_0, weight = model_model_layers_14_mlp_up_proj_weight_palettized, x = input_205)[name = string("up_states")]; tensor gate_states = silu(x = input_207)[name = string("gate_states")]; tensor input_209 = mul(x = gate_states, y = up_states)[name = string("input_209")]; string hidden_states_119_pad_type_0 = const()[name = string("hidden_states_119_pad_type_0"), val = string("valid")]; tensor hidden_states_119_strides_0 = const()[name = string("hidden_states_119_strides_0"), val = tensor([1, 1])]; tensor hidden_states_119_pad_0 = const()[name = string("hidden_states_119_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_119_dilations_0 = const()[name = string("hidden_states_119_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_119_groups_0 = const()[name = string("hidden_states_119_groups_0"), val = int32(1)]; tensor hidden_states_119 = conv(dilations = hidden_states_119_dilations_0, groups = hidden_states_119_groups_0, pad = hidden_states_119_pad_0, pad_type = hidden_states_119_pad_type_0, strides = hidden_states_119_strides_0, weight = model_model_layers_14_mlp_down_proj_weight_palettized, x = input_209)[name = string("hidden_states_119")]; tensor var_3627_axes_0 = const()[name = string("op_3627_axes_0"), val = tensor([2])]; tensor var_3627 = squeeze(axes = var_3627_axes_0, x = hidden_states_119)[name = string("op_3627")]; tensor var_3628 = const()[name = string("op_3628"), val = tensor([0, 2, 1])]; tensor var_3629 = transpose(perm = var_3628, x = var_3627)[name = string("transpose_5")]; tensor hidden_states_121_cast_fp16 = add(x = hidden_states_117_cast_fp16, y = var_3629)[name = string("hidden_states_121_cast_fp16")]; fp16 const_271_promoted_to_fp16 = const()[name = string("const_271_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3632_cast_fp16 = mul(x = hidden_states_121_cast_fp16, y = const_271_promoted_to_fp16)[name = string("op_3632_cast_fp16")]; bool input_211_interleave_0 = const()[name = string("input_211_interleave_0"), val = bool(false)]; tensor input_211_cast_fp16 = concat(axis = var_73, interleave = input_211_interleave_0, values = (hidden_states_121_cast_fp16, var_3632_cast_fp16))[name = string("input_211_cast_fp16")]; tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_76_to_fp16, x = input_211_cast_fp16)[name = string("normed_121_cast_fp16")]; tensor normed_begin_0 = const()[name = string("normed_begin_0"), val = tensor([0, 0, 0])]; tensor normed_end_0 = const()[name = string("normed_end_0"), val = tensor([1, 64, 2048])]; tensor normed_end_mask_0 = const()[name = string("normed_end_mask_0"), val = tensor([true, true, false])]; tensor normed_cast_fp16 = slice_by_index(begin = normed_begin_0, end = normed_end_0, end_mask = normed_end_mask_0, x = normed_121_cast_fp16)[name = string("normed_cast_fp16")]; tensor const_274_promoted_to_fp16 = const()[name = string("const_274_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488181760)))]; tensor hidden_states_123_cast_fp16 = mul(x = normed_cast_fp16, y = const_274_promoted_to_fp16)[name = string("hidden_states_123_cast_fp16")]; tensor var_3647 = const()[name = string("op_3647"), val = tensor([0, 2, 1])]; tensor var_3649_axes_0 = const()[name = string("op_3649_axes_0"), val = tensor([2])]; tensor var_3648_cast_fp16 = transpose(perm = var_3647, x = hidden_states_123_cast_fp16)[name = string("transpose_4")]; tensor var_3649_cast_fp16 = expand_dims(axes = var_3649_axes_0, x = var_3648_cast_fp16)[name = string("op_3649_cast_fp16")]; string query_states_61_pad_type_0 = const()[name = string("query_states_61_pad_type_0"), val = string("valid")]; tensor query_states_61_strides_0 = const()[name = string("query_states_61_strides_0"), val = tensor([1, 1])]; tensor query_states_61_pad_0 = const()[name = string("query_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_61_dilations_0 = const()[name = string("query_states_61_dilations_0"), val = tensor([1, 1])]; int32 query_states_61_groups_0 = const()[name = string("query_states_61_groups_0"), val = int32(1)]; tensor query_states_61 = conv(dilations = query_states_61_dilations_0, groups = query_states_61_groups_0, pad = query_states_61_pad_0, pad_type = query_states_61_pad_type_0, strides = query_states_61_strides_0, weight = model_model_layers_15_self_attn_q_proj_weight_palettized, x = var_3649_cast_fp16)[name = string("query_states_61")]; string key_states_91_pad_type_0 = const()[name = string("key_states_91_pad_type_0"), val = string("valid")]; tensor key_states_91_strides_0 = const()[name = string("key_states_91_strides_0"), val = tensor([1, 1])]; tensor key_states_91_pad_0 = const()[name = string("key_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_91_dilations_0 = const()[name = string("key_states_91_dilations_0"), val = tensor([1, 1])]; int32 key_states_91_groups_0 = const()[name = string("key_states_91_groups_0"), val = int32(1)]; tensor key_states_91 = conv(dilations = key_states_91_dilations_0, groups = key_states_91_groups_0, pad = key_states_91_pad_0, pad_type = key_states_91_pad_type_0, strides = key_states_91_strides_0, weight = model_model_layers_15_self_attn_k_proj_weight_palettized, x = var_3649_cast_fp16)[name = string("key_states_91")]; string value_states_91_pad_type_0 = const()[name = string("value_states_91_pad_type_0"), val = string("valid")]; tensor value_states_91_strides_0 = const()[name = string("value_states_91_strides_0"), val = tensor([1, 1])]; tensor value_states_91_pad_0 = const()[name = string("value_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_91_dilations_0 = const()[name = string("value_states_91_dilations_0"), val = tensor([1, 1])]; int32 value_states_91_groups_0 = const()[name = string("value_states_91_groups_0"), val = int32(1)]; tensor value_states_91 = conv(dilations = value_states_91_dilations_0, groups = value_states_91_groups_0, pad = value_states_91_pad_0, pad_type = value_states_91_pad_type_0, strides = value_states_91_strides_0, weight = model_model_layers_15_self_attn_v_proj_weight_palettized, x = var_3649_cast_fp16)[name = string("value_states_91")]; tensor var_3669 = const()[name = string("op_3669"), val = tensor([1, 32, 64, 64])]; tensor var_3670 = reshape(shape = var_3669, x = query_states_61)[name = string("op_3670")]; tensor var_3671 = const()[name = string("op_3671"), val = tensor([0, 1, 3, 2])]; tensor var_3673 = const()[name = string("op_3673"), val = tensor([1, 8, 64, 64])]; tensor var_3674 = reshape(shape = var_3673, x = key_states_91)[name = string("op_3674")]; tensor var_3675 = const()[name = string("op_3675"), val = tensor([0, 1, 3, 2])]; tensor var_3677 = const()[name = string("op_3677"), val = tensor([1, 8, 64, 64])]; tensor var_3678 = reshape(shape = var_3677, x = value_states_91)[name = string("op_3678")]; tensor var_3679 = const()[name = string("op_3679"), val = tensor([0, 1, 3, 2])]; tensor x1_61_begin_0 = const()[name = string("x1_61_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_61_end_0 = const()[name = string("x1_61_end_0"), val = tensor([1, 32, 64, 32])]; tensor x1_61_end_mask_0 = const()[name = string("x1_61_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_421 = transpose(perm = var_3671, x = var_3670)[name = string("transpose_3")]; tensor x1_61 = slice_by_index(begin = x1_61_begin_0, end = x1_61_end_0, end_mask = x1_61_end_mask_0, x = x_421)[name = string("x1_61")]; tensor x2_61_begin_0 = const()[name = string("x2_61_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_61_end_0 = const()[name = string("x2_61_end_0"), val = tensor([1, 32, 64, 64])]; tensor x2_61_end_mask_0 = const()[name = string("x2_61_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2_61 = slice_by_index(begin = x2_61_begin_0, end = x2_61_end_0, end_mask = x2_61_end_mask_0, x = x_421)[name = string("x2_61")]; tensor var_3697 = mul(x = x1_61, y = cos_7)[name = string("op_3697")]; tensor var_3698 = mul(x = x2_61, y = sin_7)[name = string("op_3698")]; tensor var_3699 = sub(x = var_3697, y = var_3698)[name = string("op_3699")]; tensor var_3700 = mul(x = x2_61, y = cos_7)[name = string("op_3700")]; tensor var_3701 = mul(x = x1_61, y = sin_7)[name = string("op_3701")]; tensor var_3702 = add(x = var_3700, y = var_3701)[name = string("op_3702")]; bool rotated_61_interleave_0 = const()[name = string("rotated_61_interleave_0"), val = bool(false)]; tensor rotated_61 = concat(axis = var_73, interleave = rotated_61_interleave_0, values = (var_3699, var_3702))[name = string("rotated_61")]; tensor x1_begin_0 = const()[name = string("x1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor x1_end_0 = const()[name = string("x1_end_0"), val = tensor([1, 8, 64, 32])]; tensor x1_end_mask_0 = const()[name = string("x1_end_mask_0"), val = tensor([true, true, true, false])]; tensor x_425 = transpose(perm = var_3675, x = var_3674)[name = string("transpose_2")]; tensor x1 = slice_by_index(begin = x1_begin_0, end = x1_end_0, end_mask = x1_end_mask_0, x = x_425)[name = string("x1")]; tensor x2_begin_0 = const()[name = string("x2_begin_0"), val = tensor([0, 0, 0, 32])]; tensor x2_end_0 = const()[name = string("x2_end_0"), val = tensor([1, 8, 64, 64])]; tensor x2_end_mask_0 = const()[name = string("x2_end_mask_0"), val = tensor([true, true, true, true])]; tensor x2 = slice_by_index(begin = x2_begin_0, end = x2_end_0, end_mask = x2_end_mask_0, x = x_425)[name = string("x2")]; tensor var_3718 = mul(x = x1, y = cos_7)[name = string("op_3718")]; tensor var_3719 = mul(x = x2, y = sin_7)[name = string("op_3719")]; tensor var_3720 = sub(x = var_3718, y = var_3719)[name = string("op_3720")]; tensor var_3721 = mul(x = x2, y = cos_7)[name = string("op_3721")]; tensor var_3722 = mul(x = x1, y = sin_7)[name = string("op_3722")]; tensor var_3723 = add(x = var_3721, y = var_3722)[name = string("op_3723")]; bool rotated_interleave_0 = const()[name = string("rotated_interleave_0"), val = bool(false)]; tensor rotated = concat(axis = var_73, interleave = rotated_interleave_0, values = (var_3720, var_3723))[name = string("rotated")]; tensor expand_dims_180 = const()[name = string("expand_dims_180"), val = tensor([15])]; tensor expand_dims_181 = const()[name = string("expand_dims_181"), val = tensor([0])]; tensor expand_dims_183 = const()[name = string("expand_dims_183"), val = tensor([0])]; tensor expand_dims_184 = const()[name = string("expand_dims_184"), val = tensor([16])]; int32 concat_272_axis_0 = const()[name = string("concat_272_axis_0"), val = int32(0)]; bool concat_272_interleave_0 = const()[name = string("concat_272_interleave_0"), val = bool(false)]; tensor concat_272 = concat(axis = concat_272_axis_0, interleave = concat_272_interleave_0, values = (expand_dims_180, expand_dims_181, current_pos, expand_dims_183))[name = string("concat_272")]; tensor concat_273_values1_0 = const()[name = string("concat_273_values1_0"), val = tensor([0])]; tensor concat_273_values3_0 = const()[name = string("concat_273_values3_0"), val = tensor([0])]; int32 concat_273_axis_0 = const()[name = string("concat_273_axis_0"), val = int32(0)]; bool concat_273_interleave_0 = const()[name = string("concat_273_interleave_0"), val = bool(false)]; tensor concat_273 = concat(axis = concat_273_axis_0, interleave = concat_273_interleave_0, values = (expand_dims_184, concat_273_values1_0, var_597, concat_273_values3_0))[name = string("concat_273")]; tensor model_model_kv_cache_0_internal_tensor_assign_31_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_31_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_31_cast_fp16 = slice_update(begin = concat_272, begin_mask = model_model_kv_cache_0_internal_tensor_assign_31_begin_mask_0, end = concat_273, end_mask = model_model_kv_cache_0_internal_tensor_assign_31_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_31_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_31_stride_0, update = rotated, x = coreml_update_state_61)[name = string("model_model_kv_cache_0_internal_tensor_assign_31_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_31_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_126_write_state")]; tensor coreml_update_state_62 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_126")]; tensor expand_dims_186 = const()[name = string("expand_dims_186"), val = tensor([31])]; tensor expand_dims_187 = const()[name = string("expand_dims_187"), val = tensor([0])]; tensor expand_dims_189 = const()[name = string("expand_dims_189"), val = tensor([0])]; tensor expand_dims_190 = const()[name = string("expand_dims_190"), val = tensor([32])]; int32 concat_276_axis_0 = const()[name = string("concat_276_axis_0"), val = int32(0)]; bool concat_276_interleave_0 = const()[name = string("concat_276_interleave_0"), val = bool(false)]; tensor concat_276 = concat(axis = concat_276_axis_0, interleave = concat_276_interleave_0, values = (expand_dims_186, expand_dims_187, current_pos, expand_dims_189))[name = string("concat_276")]; tensor concat_277_values1_0 = const()[name = string("concat_277_values1_0"), val = tensor([0])]; tensor concat_277_values3_0 = const()[name = string("concat_277_values3_0"), val = tensor([0])]; int32 concat_277_axis_0 = const()[name = string("concat_277_axis_0"), val = int32(0)]; bool concat_277_interleave_0 = const()[name = string("concat_277_interleave_0"), val = bool(false)]; tensor concat_277 = concat(axis = concat_277_axis_0, interleave = concat_277_interleave_0, values = (expand_dims_190, concat_277_values1_0, var_597, concat_277_values3_0))[name = string("concat_277")]; tensor model_model_kv_cache_0_internal_tensor_assign_32_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_stride_0"), val = tensor([1, 1, 1, 1])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_begin_mask_0"), val = tensor([false, false, false, false])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_end_mask_0"), val = tensor([false, true, false, true])]; tensor model_model_kv_cache_0_internal_tensor_assign_32_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_32_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_93 = transpose(perm = var_3679, x = var_3678)[name = string("transpose_1")]; tensor model_model_kv_cache_0_internal_tensor_assign_32_cast_fp16 = slice_update(begin = concat_276, begin_mask = model_model_kv_cache_0_internal_tensor_assign_32_begin_mask_0, end = concat_277, end_mask = model_model_kv_cache_0_internal_tensor_assign_32_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_32_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_32_stride_0, update = value_states_93, x = coreml_update_state_62)[name = string("model_model_kv_cache_0_internal_tensor_assign_32_cast_fp16")]; write_state(data = model_model_kv_cache_0_internal_tensor_assign_32_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_127_write_state")]; tensor coreml_update_state_63 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_127")]; tensor var_3746_begin_0 = const()[name = string("op_3746_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_3746_end_0 = const()[name = string("op_3746_end_0"), val = tensor([16, 8, 4096, 64])]; tensor var_3746_end_mask_0 = const()[name = string("op_3746_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3746_cast_fp16 = slice_by_index(begin = var_3746_begin_0, end = var_3746_end_0, end_mask = var_3746_end_mask_0, x = coreml_update_state_63)[name = string("op_3746_cast_fp16")]; tensor K_layer_cache_axes_0 = const()[name = string("K_layer_cache_axes_0"), val = tensor([0])]; tensor K_layer_cache_cast_fp16 = squeeze(axes = K_layer_cache_axes_0, x = var_3746_cast_fp16)[name = string("K_layer_cache_cast_fp16")]; tensor var_3748_begin_0 = const()[name = string("op_3748_begin_0"), val = tensor([31, 0, 0, 0])]; tensor var_3748_end_0 = const()[name = string("op_3748_end_0"), val = tensor([1, 8, 4096, 64])]; tensor var_3748_end_mask_0 = const()[name = string("op_3748_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3748_cast_fp16 = slice_by_index(begin = var_3748_begin_0, end = var_3748_end_0, end_mask = var_3748_end_mask_0, x = coreml_update_state_63)[name = string("op_3748_cast_fp16")]; tensor V_layer_cache_axes_0 = const()[name = string("V_layer_cache_axes_0"), val = tensor([0])]; tensor V_layer_cache_cast_fp16 = squeeze(axes = V_layer_cache_axes_0, x = var_3748_cast_fp16)[name = string("V_layer_cache_cast_fp16")]; tensor x_431_axes_0 = const()[name = string("x_431_axes_0"), val = tensor([1])]; tensor x_431_cast_fp16 = expand_dims(axes = x_431_axes_0, x = K_layer_cache_cast_fp16)[name = string("x_431_cast_fp16")]; tensor var_3757 = const()[name = string("op_3757"), val = tensor([1, 4, 1, 1])]; tensor x_433_cast_fp16 = tile(reps = var_3757, x = x_431_cast_fp16)[name = string("x_433_cast_fp16")]; tensor var_3761 = const()[name = string("op_3761"), val = tensor([1, -1, 4096, 64])]; tensor var_3762_cast_fp16 = reshape(shape = var_3761, x = x_433_cast_fp16)[name = string("op_3762_cast_fp16")]; tensor x_437_axes_0 = const()[name = string("x_437_axes_0"), val = tensor([1])]; tensor x_437_cast_fp16 = expand_dims(axes = x_437_axes_0, x = V_layer_cache_cast_fp16)[name = string("x_437_cast_fp16")]; tensor var_3764 = const()[name = string("op_3764"), val = tensor([1, 4, 1, 1])]; tensor x_439_cast_fp16 = tile(reps = var_3764, x = x_437_cast_fp16)[name = string("x_439_cast_fp16")]; bool var_3771_transpose_x_0 = const()[name = string("op_3771_transpose_x_0"), val = bool(false)]; bool var_3771_transpose_y_0 = const()[name = string("op_3771_transpose_y_0"), val = bool(true)]; tensor var_3771_cast_fp16 = matmul(transpose_x = var_3771_transpose_x_0, transpose_y = var_3771_transpose_y_0, x = rotated_61, y = var_3762_cast_fp16)[name = string("op_3771_cast_fp16")]; fp16 var_3772_to_fp16 = const()[name = string("op_3772_to_fp16"), val = fp16(0x1p-3)]; tensor attn_weights_cast_fp16 = mul(x = var_3771_cast_fp16, y = var_3772_to_fp16)[name = string("attn_weights_cast_fp16")]; tensor x_441_cast_fp16 = add(x = attn_weights_cast_fp16, y = causal_mask)[name = string("x_441_cast_fp16")]; tensor reduce_max_15_axes_0 = const()[name = string("reduce_max_15_axes_0"), val = tensor([-1])]; bool reduce_max_15_keep_dims_0 = const()[name = string("reduce_max_15_keep_dims_0"), val = bool(true)]; tensor reduce_max_15_cast_fp16 = reduce_max(axes = reduce_max_15_axes_0, keep_dims = reduce_max_15_keep_dims_0, x = x_441_cast_fp16)[name = string("reduce_max_15_cast_fp16")]; tensor x_cast_fp16 = sub(x = x_441_cast_fp16, y = reduce_max_15_cast_fp16)[name = string("x_cast_fp16")]; tensor exp_x_cast_fp16 = exp(x = x_cast_fp16)[name = string("exp_x_cast_fp16")]; tensor var_3783_axes_0 = const()[name = string("op_3783_axes_0"), val = tensor([-1])]; bool var_3783_keep_dims_0 = const()[name = string("op_3783_keep_dims_0"), val = bool(true)]; tensor var_3783_cast_fp16 = reduce_sum(axes = var_3783_axes_0, keep_dims = var_3783_keep_dims_0, x = exp_x_cast_fp16)[name = string("op_3783_cast_fp16")]; tensor var_3784_cast_fp16 = real_div(x = exp_x_cast_fp16, y = var_3783_cast_fp16)[name = string("op_3784_cast_fp16")]; tensor concat_282 = const()[name = string("concat_282"), val = tensor([32, 64, 4096])]; tensor reshape_45_cast_fp16 = reshape(shape = concat_282, x = var_3784_cast_fp16)[name = string("reshape_45_cast_fp16")]; tensor concat_283 = const()[name = string("concat_283"), val = tensor([32, 4096, 64])]; tensor reshape_46_cast_fp16 = reshape(shape = concat_283, x = x_439_cast_fp16)[name = string("reshape_46_cast_fp16")]; bool matmul_15_transpose_x_0 = const()[name = string("matmul_15_transpose_x_0"), val = bool(false)]; bool matmul_15_transpose_y_0 = const()[name = string("matmul_15_transpose_y_0"), val = bool(false)]; tensor matmul_15_cast_fp16 = matmul(transpose_x = matmul_15_transpose_x_0, transpose_y = matmul_15_transpose_y_0, x = reshape_45_cast_fp16, y = reshape_46_cast_fp16)[name = string("matmul_15_cast_fp16")]; tensor concat_287 = const()[name = string("concat_287"), val = tensor([1, 32, 64, 64])]; tensor reshape_47_cast_fp16 = reshape(shape = concat_287, x = matmul_15_cast_fp16)[name = string("reshape_47_cast_fp16")]; tensor var_3787_perm_0 = const()[name = string("op_3787_perm_0"), val = tensor([0, 2, 1, 3])]; tensor var_3789 = const()[name = string("op_3789"), val = tensor([1, 64, 2048])]; tensor var_3787_cast_fp16 = transpose(perm = var_3787_perm_0, x = reshape_47_cast_fp16)[name = string("transpose_0")]; tensor input_cast_fp16 = reshape(shape = var_3789, x = var_3787_cast_fp16)[name = string("input_cast_fp16")]; tensor model_model_layers_15_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488185920))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(490283136))))[name = string("model_model_layers_15_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_15_cast_fp16 = linear(bias = linear_0_bias_0_to_fp16, weight = model_model_layers_15_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_cast_fp16)[name = string("linear_15_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_121_cast_fp16, y = linear_15_cast_fp16)[name = string("hidden_states_cast_fp16")]; tensor var_3795_begin_0 = const()[name = string("op_3795_begin_0"), val = tensor([0, 0, 0])]; tensor var_3795_end_0 = const()[name = string("op_3795_end_0"), val = tensor([1, 1, 2048])]; tensor var_3795_end_mask_0 = const()[name = string("op_3795_end_mask_0"), val = tensor([true, false, true])]; tensor output_hidden_states = slice_by_index(begin = var_3795_begin_0, end = var_3795_end_0, end_mask = var_3795_end_mask_0, x = hidden_states_cast_fp16)[name = string("op_3795_cast_fp16")]; } -> (output_hidden_states); }